diff --git a/.env.example b/.env.example index 3634c03eb..9f0cf4cbc 100644 --- a/.env.example +++ b/.env.example @@ -114,17 +114,28 @@ PORT=3001 # Maximum inbound WebSocket frame size in bytes (default: 16777216 = 16MB) # WS_MAX_PAYLOAD_BYTES=16777216 -# Terminal stream replay ring bytes per terminal (default: 262144 = 256KB) -# TERMINAL_REPLAY_RING_MAX_BYTES=262144 - -# Terminal stream output queue bytes per attached client (default: 131072 = 128KB) -# TERMINAL_CLIENT_QUEUE_MAX_BYTES=131072 +# Terminal stream replay ring bytes per terminal (default: 1048576 = 1MB). +# This is the per-terminal retained-output ring every attach replays from: +# negotiated (pacedTerminalReplayV1) clients page it out bounded by +# credits, and its front bound is what `terminal.attach.ready` reports as +# `oldestRetainedSeq` (older output is gone; legacy inline attaches just +# get the retained tail). +# TERMINAL_REPLAY_RING_MAX_BYTES=1048576 + +# Terminal stream output queue bytes per attached client (default: 16777216 = 16MB). +# This is the SPILL bound: past it, oldest output is evicted with a +# generation-scoped gap instead of disconnecting the client. +# TERMINAL_CLIENT_QUEUE_MAX_BYTES=16777216 # Terminal stream send batch bytes per flush (default: 65536 = 64KB) # TERMINAL_STREAM_BATCH_MAX_BYTES=65536 -# Catastrophic WS bufferedAmount threshold for terminal streaming (default: 16777216 = 16MB) -# TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES=16777216 +# Catastrophic WS bufferedAmount disconnect threshold (default: 67108864 = 64MB). +# Must be STRICTLY greater than TERMINAL_CLIENT_QUEUE_MAX_BYTES — the server +# refuses to start otherwise, so normal pressure spills before any disconnect. +# The close additionally requires zero successful socket sends for the whole +# stall window (a slow-but-progressing client is not a dead socket). +# TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES=67108864 # Catastrophic stall duration before 4008 close, ms (default: 10000) # TERMINAL_WS_CATASTROPHIC_STALL_MS=10000 diff --git a/.github/workflows/rust-tests.yml b/.github/workflows/rust-tests.yml index 4f8d39d88..f9f6f3620 100644 --- a/.github/workflows/rust-tests.yml +++ b/.github/workflows/rust-tests.yml @@ -1,6 +1,7 @@ name: Rust Tests on: + workflow_dispatch: pull_request: permissions: @@ -35,7 +36,12 @@ jobs: # changed is not proof of fixture compatibility. Scoped to the # root and workspace members — the independent demo locks are # not part of the workspace dependency tree. - if git diff --name-only "origin/${{ github.base_ref }}...${{ github.sha }}" \ + # `github.base_ref` exists only for pull_request events; manual + # dispatch (the event-routing escape hatch) compares against the + # default branch instead. + BASE="${{ github.base_ref }}" + [ -n "$BASE" ] || BASE="main" + if git diff --name-only "origin/$BASE...${{ github.sha }}" \ | grep -Eq '\.rs$|(^|/)Cargo\.(toml|lock)$|^rust-toolchain|^\.cargo/|^test/fixtures/|^(pnpm-lock\.yaml|pnpm-workspace\.yaml|package(-lock)?\.json|\.npmrc)$|^(crates|packages)/.*package(-lock)?\.json$|^packages/'; then echo "rust=true" >> "$GITHUB_OUTPUT" else diff --git a/AGENTS.md b/AGENTS.md index 13369f1b6..0ad5363a8 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -264,7 +264,9 @@ stderr note). Provisioning, rotation, and revocation live in **WebSocket Protocol:** Schema-validated messages using Zod. Handshake flow: client sends `hello` with token → server validates → sends `ready`. Message types include `terminal.create/input/resize/detach/attach` and broadcasts like `sessions.updated`. The `ready` frame carries an optional additive `buildId` (the server's artifact-time-baked git commit, `"unknown"` fallback): the client bakes its own at Vite build time (`__FRESHELL_BUILD_ID__`) and, on a mismatch, reloads exactly once per tab session (sessionStorage sentinel `freshell.server-build-reload` records the last attempted server build id; the same id never reloads twice, a different (corrected) deployment re-arms the guard), self-healing stale-client contract errors; `"unknown"` on either side never triggers or clears the guard (`src/lib/server-build-check.ts`). The once-guard is per server identity: an origin fronted by mixed-build servers could oscillate, and a newer client against an older server costs one futile bounded reload per fresh tab session (both accepted for the single-server self-hosted model). -**PTY Lifecycle:** Each terminal has a unique ID. Server maintains 64KB scrollback buffer. On attach, client receives buffer snapshot then streams new output. On detach, process continues running (background session). Configurable idle timeout (15 mins default). +**PTY Lifecycle:** Each terminal has a unique ID. The server retains output in a per-terminal replay ring (default 1 MiB, `TERMINAL_REPLAY_RING_MAX_BYTES`; older output is gone). On attach, a non-negotiated (legacy) client receives the retained tail inline as a snapshot then streams new output; on detach, the process continues running (background session). Configurable idle timeout (15 mins default). + +**Negotiated Restore (pacedTerminalReplayV1):** A client whose `hello` negotiated `pacedTerminalReplayV1` restores via a bounded paced-replay session instead of the inline snapshot: `terminal.attach.ready` carries the window bounds (plus `oldestRetainedSeq`), a reconnect resumes from the client's coverage cursor (`requestedSinceSeq`/`effectiveSinceSeq`) when its checkpoint is valid, the server pushes one bounded first page, and further pages flow only on `terminal.replay.credit` continuation credits (stale generations and non-negotiated credits are inert). Retention loss mid-restore is reported as an exact `terminal.output.gap` (bounds-carrying), never as a silent truncation or a disconnect. Attach intents pick their own replay shape (viewport hydrate, transport reconnect, hidden-pane `keepalive_delta`; the load-more-history UI flow rides viewport hydrate today), and hidden panes can claim terminals (`terminalLifetimeClaimV1`) so they survive reaping while invisible. Capability negotiation gates all of this: older clients keep the exact legacy wire behavior, and no healthy process is killed or replaced because replay is missing. Per-connection output pressure spills before it disconnects: past 16 MiB of queued output (`TERMINAL_CLIENT_QUEUE_MAX_BYTES`) the oldest frames are evicted with a generation-scoped `terminal.output.gap` (`reason: queue_overflow`; the `ws.terminal_stream.queue_overflow_spill` event is emitted at eviction time, rate-limited per connection), and only a sustained over-64 MiB backlog with zero successful socket sends for the whole stall window closes with 4008 (`ws.terminal_stream.catastrophic_close` carries per-occurrence `sends_in_window` evidence plus the lifetime `total_sends`). **Claude Session Discovery:** Watches `~/.claude/projects/*/sessions/*.jsonl` for new files. Parses JSONL streams to extract messages, groups by project path. diff --git a/crates/freshell-freshagent/src/opencode_ws.rs b/crates/freshell-freshagent/src/opencode_ws.rs index 9aa0dca3a..9f10f9be3 100644 --- a/crates/freshell-freshagent/src/opencode_ws.rs +++ b/crates/freshell-freshagent/src/opencode_ws.rs @@ -10087,6 +10087,28 @@ mod tests { Some("ses_1"), "fixture: the send materialized the durable id (and committed Live{{FreshAgent}})" ); + // The materialize spawn's ownership commit (Live{FreshAgent} on the + // durable key) lands asynchronously AFTER the durable id itself; + // on a contended 2-core CI runner the test can outrun it (observed + // twice on CI: begin_handoff met a not-yet-Live key and the fixture + // assert panicked — never on the 96-core dev box). Poll the + // registry until the committed precondition is observable instead + // of assuming the spawn completed; the bound trips only if the + // commit never lands at all. + let mut commit_polls = 0u32; + loop { + let snap = registry.observe("opencode", "ses_1"); + if matches!(snap.state, freshell_ownership::OwnershipState::Live { .. }) { + break; + } + commit_polls += 1; + assert!( + commit_polls < 10_000, + "fixture: the materialize spawn never committed Live{{FreshAgent}} \ + ownership for ses_1: {snap:?}" + ); + tokio::time::sleep(std::time::Duration::from_millis(1)).await; + } // The refusal precondition: a handoff owns the durable key's // transition, so the kill's fenced stop is typed-blocked. let freshell_ownership::BeginOutcome::Granted { .. } = registry.begin_handoff( diff --git a/crates/freshell-protocol/src/client_messages.rs b/crates/freshell-protocol/src/client_messages.rs index 71776f8a5..81e06478f 100644 --- a/crates/freshell-protocol/src/client_messages.rs +++ b/crates/freshell-protocol/src/client_messages.rs @@ -32,6 +32,15 @@ pub enum ClientMessage { TerminalAttach(TerminalAttach), #[serde(rename = "terminal.interest")] TerminalInterest(TerminalInterest), + /// Responsive-terminal-restore Workstream 1 (paced replay): the + /// continuation credit a `pacedTerminalReplayV1` client sends after fully + /// consuming an ordered replay page — `consumedSeq` is the last sequence + /// it consumed, `attachRequestId` scopes it to one attach generation. + /// Additive optional; protocol version stays 10. Ignored by servers that + /// predate the capability (accept-and-strip) and by connections whose + /// own hello did not negotiate it. + #[serde(rename = "terminal.replay.credit")] + TerminalReplayCredit(TerminalReplayCredit), #[serde(rename = "terminal.autoResumeCancel")] TerminalAutoResumeCancel(TerminalAutoResumeCancel), #[serde(rename = "terminal.detach")] @@ -121,7 +130,7 @@ pub enum ClientMessage { /// The exact `type` discriminants of every client→server message, in the frozen /// inventory's order. This is the T0 conformance checklist. -pub const CLIENT_MESSAGE_TYPES: [&str; 41] = [ +pub const CLIENT_MESSAGE_TYPES: [&str; 42] = [ "amplifier.activity.list", "claude.activity.list", "client.diagnostic", @@ -160,6 +169,7 @@ pub const CLIENT_MESSAGE_TYPES: [&str; 41] = [ "terminal.input", "terminal.interest", "terminal.kill", + "terminal.replay.credit", "terminal.resize", "ui.layout.sync", "ui.screenshot.result", @@ -187,6 +197,19 @@ pub struct HelloCapabilities { /// advertises the capability back (§4.2). Absent for the frozen client. #[serde(skip_serializing_if = "Option::is_none")] pub pane_reconcile_v1: Option, + /// Paced terminal restore opt-in (responsive-terminal-restore Workstream + /// 1): the client understands bounded, ascending paced replay batches with + /// continuation credit. Additive optional — absent on the frozen client + /// and stripped-tolerant on older servers (no version bump). + #[serde(skip_serializing_if = "Option::is_none")] + pub paced_terminal_replay_v1: Option, + /// Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1): + /// the client sends `terminal.interest.claimedTerminalIds` only after the + /// `ready` echo advertises the capability back. Additive optional — absent + /// on the frozen client and stripped-tolerant on older servers (no version + /// bump). + #[serde(skip_serializing_if = "Option::is_none")] + pub terminal_lifetime_claim_v1: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -240,6 +263,14 @@ pub struct TerminalInterest { pub revision: u64, pub focused_terminal_id: Option, pub visible_terminal_ids: Vec, + /// Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1, + /// negotiated `terminalLifetimeClaimV1` only): terminals this connection + /// wants kept alive WITHOUT attaching. `None` carries no claim information + /// (the frozen shape); `Some(set)` supersedes the connection's previous + /// claim set snapshot-by-snapshot — omitting an id from a later snapshot + /// is the explicit withdrawal (release). + #[serde(skip_serializing_if = "Option::is_none")] + pub claimed_terminal_ids: Option>, } // --- client.diagnostic ------------------------------------------------------ @@ -376,6 +407,30 @@ pub struct TerminalAttach { pub expected_session_ref: Option, #[serde(skip_serializing_if = "Option::is_none")] pub max_replay_bytes: Option, + /// Paced terminal restore (responsive-terminal-restore Workstream 1): + /// the negotiated forward-page limit — an optional UPPER BOUND on each + /// paced replay page's serialized bytes, honored only on + /// pacedTerminalReplayV1 connections and clamped to the server's own + /// page-budget cap (`min(requested, server cap)`). Round-2 finding F3: + /// the field used to be emitted by the client and silently stripped + /// here — it is now part of the honest wire contract. Additive + /// optional; a missing, malformed, or non-positive value falls back to + /// the server's default exactly like the pre-contract accept-and-strip + /// behavior (the lossy deserializer keeps a wrong-typed value from + /// failing the whole attach frame). Integer-valued number spellings + /// (`2048.0`, `2e3`) carry the same value as their canonical integer + /// forms and are accepted and validated the same way (E2R1 finding 3) + /// — never silently dropped. E2R1 finding 2 (the honest bound): pages + /// are bounded by max(requested, the atomic frame size) — a single + /// frame larger than the request forms its own ATOMIC single-frame + /// page, bounded by the server's fragment cap (every frame is + /// pre-fragmented, so one frame's serialized size never exceeds it). + #[serde( + default, + deserialize_with = "lossy_positive_i64", + skip_serializing_if = "Option::is_none" + )] + pub replay_page_bytes: Option, #[serde(skip_serializing_if = "Option::is_none")] pub priority: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -404,6 +459,60 @@ pub struct TerminalAttach { pub observed_generation: Option, } +/// [`TerminalAttach::replay_page_bytes`]'s lossy deserializer (round-2 +/// finding F3): only a clean positive integer counts as a requested bound; +/// a missing, malformed (wrong-typed, fractional), or non-positive value +/// deserializes to `None` — the server's default — instead of failing the +/// whole attach frame. This preserves the pre-contract accept-and-strip +/// tolerance for buggy senders exactly. +/// +/// E2R1 finding 3: integer-VALUED number spellings (`2048.0`, `2e3`) +/// deserialize through Serde JSON's float storage variant, but they carry +/// the same VALUE as their canonical integer spellings — JSON has one +/// number type, and the TS/Zod side (`z.number().int().positive()`) plus +/// the generated JSON Schema accept that value as an integer. The lossy +/// deserializer accepts and validates them exactly like the canonical +/// form; a protocol field a client emits is never silently ignored. +fn lossy_positive_i64<'de, D>(deserializer: D) -> Result, D::Error> +where + D: serde::Deserializer<'de>, +{ + let value: Option = Option::deserialize(deserializer)?; + Ok(value + .as_ref() + .and_then(positive_integer_value) + .filter(|n| *n > 0)) +} + +/// The positive-integer VALUE of one JSON number: Serde JSON's integer +/// storage directly, or its float storage when the value is integral +/// (E2R1 finding 3 — `2048.0`/`2e3` parse as floats). Fractional, +/// non-finite, non-positive, and out-of-i64-range values are `None` (the +/// malformed/non-positive fallback, never a wrong bound). +fn positive_integer_value(v: &serde_json::Value) -> Option { + let n = v.as_number()?; + if let Some(i) = n.as_i64() { + return Some(i); + } + let f = n.as_f64()?; + (f.is_finite() && f.fract() == 0.0 && f > 0.0 && f <= i64::MAX as f64).then_some(f as i64) +} + +/// `terminal.replay.credit` (responsive-terminal-restore Workstream 1): one +/// continuation credit for a paced replay session, granted after the prior +/// page was consumed in order. `consumedSeq` must fall within the server's +/// outstanding-page window `(credited, lastSentPageEnd]`; stale generations +/// (a superseded `attachRequestId`) and out-of-window values are ignored. +#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] +#[serde(rename_all = "camelCase")] +pub struct TerminalReplayCredit { + pub terminal_id: String, + pub stream_id: String, + pub attach_request_id: String, + /// The last sequence the client fully consumed in order. + pub consumed_seq: i64, +} + #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] #[serde(rename_all = "camelCase")] pub struct TerminalDetach { diff --git a/crates/freshell-protocol/src/lib.rs b/crates/freshell-protocol/src/lib.rs index 130f04baf..84f5f899d 100644 --- a/crates/freshell-protocol/src/lib.rs +++ b/crates/freshell-protocol/src/lib.rs @@ -41,7 +41,7 @@ pub use settings::*; pub const WS_PROTOCOL_VERSION: u32 = 10; /// Every `type` discriminant the protocol speaks, both directions, sorted. -/// (41 client→server + 67 server→client = 108.) +/// (42 client→server + 67 server→client = 109.) pub fn all_message_types() -> Vec<&'static str> { let mut types: Vec<&'static str> = client_messages::CLIENT_MESSAGE_TYPES .iter() diff --git a/crates/freshell-protocol/src/server_messages.rs b/crates/freshell-protocol/src/server_messages.rs index 07257c343..bf64eccca 100644 --- a/crates/freshell-protocol/src/server_messages.rs +++ b/crates/freshell-protocol/src/server_messages.rs @@ -355,6 +355,34 @@ pub enum TerminalOutputGapReason { QueueOverflow, ReplayWindowExceeded, ReplayBudgetExceeded, + /// Responsive-terminal-restore W1 (round-4, plan:146): the paced + /// session's FIXED delivery boundary was reached with output staged + /// beyond it — the connection missed the declared interval's sequenced + /// output, but the ring RETAINED it (delivery loss, not retention + /// loss). Emitted ONLY on connections that negotiated + /// `pacedTerminalReplayV1` (the paced completion core is its only + /// emitter). The client repairs from its surface cursor (the same + /// checkpoint-cursor delta repair as `queue_overflow`): a finite + /// delivery window cannot guarantee convergence against indefinitely + /// faster output production, so the bounded session reports the exact + /// interval and the client's bounded baseline recovery fetches it. + HandoffBoundaryReached, +} + +/// `terminal.attach.ready.replayResetReason` — why the attach's effective +/// replay position was reset instead of honoring the requested one +/// (responsive-terminal-restore shared contract). +#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] +#[serde(rename_all = "snake_case")] +pub enum TerminalReplayResetReason { + /// The geometry-authority check rejected the requested position (the + /// only pre-restore-contract value; const on the wire). + GeometryAuthorityUnknown, + /// The requested position predates the retained replay window + /// (retention loss). Emitted ONLY on connections that negotiated + /// `pacedTerminalReplayV1` — task 3's negotiated retention-gap emission; + /// this increment only extends the value space, no emitter sets it yet. + RetentionLost, } #[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)] @@ -986,6 +1014,19 @@ pub struct ReadyCapabilities { pub pane_reconcile_fresh_agent_v1: Option, #[serde(skip_serializing_if = "Option::is_none")] pub terminal_interest_v1: Option, + /// Paced terminal restore (responsive-terminal-restore Workstream 1): + /// `Some(true)` iff the connection's `hello` opted in via + /// `capabilities.pacedTerminalReplayV1` — omitted from the wire entirely + /// otherwise (frozen-client inertness). + #[serde(skip_serializing_if = "Option::is_none")] + pub paced_terminal_replay_v1: Option, + /// Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1): + /// `Some(true)` iff the connection's `hello` opted in via + /// `capabilities.terminalLifetimeClaimV1` — omitted from the wire entirely + /// otherwise (frozen-client inertness). Present iff the client may send + /// `terminal.interest.claimedTerminalIds`. + #[serde(skip_serializing_if = "Option::is_none")] + pub terminal_lifetime_claim_v1: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] @@ -1226,9 +1267,16 @@ pub struct TerminalAttachReady { pub geometry_authority: Option, #[serde(skip_serializing_if = "Option::is_none")] pub geometry_epoch: Option, - /// const `"geometry_authority_unknown"`. + /// Restore contract (responsive-terminal-restore): the earliest sequence + /// position still available for replay — the retained ring's front + /// `seqStart`, or `head_seq + 1` when nothing older than the head is + /// retained. Emitted ONLY on connections that negotiated + /// `pacedTerminalReplayV1`; omitted otherwise so the frozen client's + /// frame stays byte-identical. + #[serde(skip_serializing_if = "Option::is_none")] + pub oldest_retained_seq: Option, #[serde(skip_serializing_if = "Option::is_none")] - pub replay_reset_reason: Option, + pub replay_reset_reason: Option, #[serde(skip_serializing_if = "Option::is_none")] pub requested_since_seq: Option, #[serde(skip_serializing_if = "Option::is_none")] @@ -1421,6 +1469,18 @@ pub struct TerminalOutputGap { pub to_seq: i64, #[serde(skip_serializing_if = "Option::is_none")] pub attach_request_id: Option, + /// Restore contract (responsive-terminal-restore): the terminal's current + /// `headSeq` at gap-emission time. Emitted ONLY on connections that + /// negotiated `pacedTerminalReplayV1`; omitted otherwise so the frozen + /// client's gap frame stays byte-identical. + #[serde(skip_serializing_if = "Option::is_none")] + pub head_seq: Option, + /// Restore contract (responsive-terminal-restore): the earliest sequence + /// position still available for replay (the retained ring's front + /// `seqStart`, or `head_seq + 1` when the ring is empty) at + /// gap-emission time. Emitted ONLY on negotiated connections. + #[serde(skip_serializing_if = "Option::is_none")] + pub oldest_retained_seq: Option, } #[derive(Debug, Clone, PartialEq, Serialize, Deserialize)] diff --git a/crates/freshell-protocol/tests/inventory.rs b/crates/freshell-protocol/tests/inventory.rs index cbf5ed663..18feb1179 100644 --- a/crates/freshell-protocol/tests/inventory.rs +++ b/crates/freshell-protocol/tests/inventory.rs @@ -31,12 +31,12 @@ fn client_types_match_inventory_exactly() { let inv = inventory(); assert_eq!( inv["clientToServer"]["count"].as_u64(), - Some(41), - "inventory declares 41 client→server types" + Some(42), + "inventory declares 42 client→server types" ); let expected = json_type_set(&inv["clientToServer"]["types"]); let actual: BTreeSet = CLIENT_MESSAGE_TYPES.iter().map(|s| s.to_string()).collect(); - assert_eq!(actual.len(), 41, "crate declares 41 client types (no dups)"); + assert_eq!(actual.len(), 42, "crate declares 42 client types (no dups)"); assert_eq!( actual, expected, "CLIENT_MESSAGE_TYPES must equal the frozen inventory (no missing/extra)" @@ -61,14 +61,14 @@ fn server_types_match_inventory_exactly() { } #[test] -fn combined_surface_is_108() { +fn combined_surface_is_109() { let all = all_message_types(); - assert_eq!(all.len(), 108, "41 client + 67 server = 108 discriminants"); + assert_eq!(all.len(), 109, "42 client + 67 server = 109 discriminants"); // sorted + unique let unique: BTreeSet<&str> = all.iter().copied().collect(); assert_eq!( unique.len(), - 108, + 109, "no discriminant collides across directions" ); } diff --git a/crates/freshell-protocol/tests/pane_reconcile.rs b/crates/freshell-protocol/tests/pane_reconcile.rs index 804985adf..632dbe689 100644 --- a/crates/freshell-protocol/tests/pane_reconcile.rs +++ b/crates/freshell-protocol/tests/pane_reconcile.rs @@ -81,6 +81,8 @@ fn ready_capabilities_advertise_pane_reconcile_v1_when_negotiated() { pane_reconcile_v1: Some(true), pane_reconcile_fresh_agent_v1: None, terminal_interest_v1: None, + paced_terminal_replay_v1: None, + terminal_lifetime_claim_v1: None, }), runtime_owners: None, }; diff --git a/crates/freshell-protocol/tests/roundtrip.rs b/crates/freshell-protocol/tests/roundtrip.rs index a8e19f991..db2d6e8de 100644 --- a/crates/freshell-protocol/tests/roundtrip.rs +++ b/crates/freshell-protocol/tests/roundtrip.rs @@ -193,6 +193,69 @@ fn ready_carries_build_id_and_omits_it_when_absent() { } } +#[test] +fn hello_roundtrips_paced_terminal_replay_v1_opt_in() { + // Workstream 1 (responsive terminal restore) negotiation: the client opt-in + // rides `hello.capabilities.pacedTerminalReplayV1`. Additive optional — a + // negotiating hello round-trips byte-identically... + let wire = r#"{"type":"hello","protocolVersion":10,"token":"t","capabilities":{"terminalOutputBatchV1":true,"pacedTerminalReplayV1":true}}"#; + match client_roundtrip(wire, "hello") { + ClientMessage::Hello(h) => { + assert_eq!( + h.capabilities.and_then(|c| c.paced_terminal_replay_v1), + Some(true), + "the paced-replay opt-in must parse through the typed struct" + ); + } + other => panic!("expected Hello, got {other:?}"), + } + + // ...and a non-negotiating hello never invents the key (the frozen + // client's wire shape is unchanged). + let wire = r#"{"type":"hello","protocolVersion":10,"token":"t","capabilities":{"terminalOutputBatchV1":true}}"#; + match client_roundtrip(wire, "hello") { + ClientMessage::Hello(h) => { + assert_eq!( + h.capabilities.and_then(|c| c.paced_terminal_replay_v1), + None, + "an absent pacedTerminalReplayV1 must stay absent (skip_serializing_if)" + ); + } + other => panic!("expected Hello, got {other:?}"), + } +} + +#[test] +fn ready_roundtrips_paced_terminal_replay_v1_echo_and_omission() { + // The negotiated echo rides `ready.capabilities` — only for a connection + // whose hello opted in. + let wire = r#"{"type":"ready","timestamp":"2026-09-19T00:00:00.000Z","serverInstanceId":"srv-abc","bootId":"boot-1","capabilities":{"pacedTerminalReplayV1":true}}"#; + match server_roundtrip(wire, "ready") { + ServerMessage::Ready(r) => { + assert_eq!( + r.capabilities.and_then(|c| c.paced_terminal_replay_v1), + Some(true), + "the negotiated paced-replay echo must parse through the typed struct" + ); + } + other => panic!("expected Ready, got {other:?}"), + } + + // A non-negotiating capabilities object stays byte-identical to today's + // output — no paced key is invented for the frozen client. + let wire = r#"{"type":"ready","timestamp":"2026-09-19T00:00:00.000Z","serverInstanceId":"srv-abc","bootId":"boot-1","capabilities":{"paneReconcileV1":true}}"#; + match server_roundtrip(wire, "ready") { + ServerMessage::Ready(r) => { + assert_eq!( + r.capabilities.and_then(|c| c.paced_terminal_replay_v1), + None, + "a non-paced negotiation must not invent pacedTerminalReplayV1" + ); + } + other => panic!("expected Ready, got {other:?}"), + } +} + #[test] fn terminal_inventory_and_settings_parse_from_transcript() { let transcript = read_json("port/oracle/fixtures/handshake-transcript.json"); @@ -335,6 +398,69 @@ fn rich_client_messages() { other => panic!("expected TerminalAttach, got {other:?}"), } + // terminal.attach (inbound) — the negotiated forward-page bound + // (round-2 finding F3): present-and-positive round-trips on the + // frozen contract; malformed and non-positive values fall back to + // None (the server default) instead of failing the whole attach + // frame. The invalid shapes are deliberately OUTSIDE the frozen + // contract (the Zod schema rejects them), so they are parsed + // directly — the tolerance is the server-side accept-and-strip + // parity, not a contract case. + let wire = r#"{"type":"terminal.attach","terminalId":"t1","intent":"viewport_hydrate","cols":80,"rows":24,"attachRequestId":"a1","replayPageBytes":2048}"#; + match client_roundtrip(wire, "terminal.attach") { + ClientMessage::TerminalAttach(a) => { + assert_eq!(a.replay_page_bytes, Some(2048)); + } + other => panic!("expected TerminalAttach, got {other:?}"), + } + for invalid in ["\"2048\"", "0", "-5", "1.5"] { + let wire = format!( + r#"{{"type":"terminal.attach","terminalId":"t1","intent":"viewport_hydrate","cols":80,"rows":24,"attachRequestId":"a1","replayPageBytes":{invalid}}}"# + ); + match serde_json::from_str::(&wire) + .expect("the invalid bound must not fail the attach frame") + { + ClientMessage::TerminalAttach(a) => { + assert_eq!( + a.replay_page_bytes, None, + "an invalid replayPageBytes ({invalid}) falls back to the server default" + ); + } + other => panic!("expected TerminalAttach, got {other:?}"), + } + } + + // E2R1 finding 3: integer-VALUED number spellings. JSON has one + // number type, so `2048.0` and `2e3` carry the same VALUE as the + // canonical `2048`/`2000`, and the TS/Zod contract + // (`z.number().int().positive()`) accepts that value as an integer — + // the frozen JSON Schema agrees (an integer-valued float satisfies + // `"type": "integer"`). The lossy deserializer must accept and + // validate them the same way, never silently drop the requested + // bound to the server default. (These parse directly, not via + // `client_roundtrip`, because the re-serialized canonical form + // legitimately differs from the non-canonical input spelling.) + for (spelling, expected) in [("2048.0", 2048i64), ("2e3", 2000i64)] { + let wire = format!( + r#"{{"type":"terminal.attach","terminalId":"t1","intent":"viewport_hydrate","cols":80,"rows":24,"attachRequestId":"a1","replayPageBytes":{spelling}}}"# + ); + let raw: Value = serde_json::from_str(&wire).expect("the spelling is JSON"); + let schema = inbound_schema()["schemas"]["ClientMessageSchema"].clone(); + assert_conforms(&validator(&schema), &raw, "replayPageBytes {spelling}"); + match serde_json::from_str::(&wire) + .expect("an integer-valued spelling must not fail the attach frame") + { + ClientMessage::TerminalAttach(a) => { + assert_eq!( + a.replay_page_bytes, + Some(expected), + "the {spelling} spelling keeps the client's requested bound" + ); + } + other => panic!("expected TerminalAttach, got {other:?}"), + } + } + // ping — unit variant. match client_roundtrip(r#"{"type":"ping"}"#, "ping") { ClientMessage::Ping => {} @@ -597,6 +723,104 @@ fn client_sessions_prefs_roundtrips_and_conforms() { assert_conforms(&validator(&schema), &back, "sessions.prefs"); } +#[test] +fn attach_ready_roundtrips_restore_bounds_and_retention_lost_reset_reason() { + // Responsive-terminal-restore shared contract: negotiated connections get + // the sequence-bounds fields additively — `oldestRetainedSeq` (earliest + // sequence position still available for replay) and the extended + // `replayResetReason` value space. + let wire = r#"{"type":"terminal.attach.ready","terminalId":"t1","streamId":"s1","headSeq":41,"replayFromSeq":7,"replayToSeq":41,"attachRequestId":"a1","requestedSinceSeq":0,"effectiveSinceSeq":0,"oldestRetainedSeq":7}"#; + match server_roundtrip(wire, "terminal.attach.ready") { + ServerMessage::TerminalAttachReady(r) => { + assert_eq!(r.oldest_retained_seq, Some(7)); + assert_eq!(r.replay_reset_reason, None); + } + other => panic!("expected TerminalAttachReady, got {other:?}"), + } + + // The new `retention_lost` reset-reason value round-trips through the + // typed field (task 3 emits it with the negotiated retention gap; this + // contract increment only extends the value space). + let wire = r#"{"type":"terminal.attach.ready","terminalId":"t1","streamId":"s1","headSeq":41,"replayFromSeq":42,"replayToSeq":41,"replayResetReason":"retention_lost","oldestRetainedSeq":42}"#; + match server_roundtrip(wire, "terminal.attach.ready") { + ServerMessage::TerminalAttachReady(r) => { + assert_eq!( + r.replay_reset_reason, + Some(TerminalReplayResetReason::RetentionLost) + ); + assert_eq!(r.oldest_retained_seq, Some(42)); + } + other => panic!("expected TerminalAttachReady, got {other:?}"), + } + + // The pre-existing reset-reason value keeps round-tripping. + let wire = r#"{"type":"terminal.attach.ready","terminalId":"t1","streamId":"s1","headSeq":9,"replayFromSeq":1,"replayToSeq":9,"replayResetReason":"geometry_authority_unknown"}"#; + match server_roundtrip(wire, "terminal.attach.ready") { + ServerMessage::TerminalAttachReady(r) => { + assert_eq!( + r.replay_reset_reason, + Some(TerminalReplayResetReason::GeometryAuthorityUnknown) + ); + } + other => panic!("expected TerminalAttachReady, got {other:?}"), + } + + // The frozen-client shape stays byte-identical: no new keys are invented + // for a connection that did not negotiate the restore contract. + let wire = r#"{"type":"terminal.attach.ready","terminalId":"t1","streamId":"s1","headSeq":3,"replayFromSeq":1,"replayToSeq":3}"#; + match server_roundtrip(wire, "terminal.attach.ready") { + ServerMessage::TerminalAttachReady(r) => { + assert_eq!(r.oldest_retained_seq, None); + assert_eq!(r.replay_reset_reason, None); + } + other => panic!("expected TerminalAttachReady, got {other:?}"), + } +} + +#[test] +fn output_gap_roundtrips_restore_bounds_and_omits_them_for_frozen_clients() { + // Negotiated shape: a queue-overflow gap carries the terminal's current + // `headSeq` and earliest-replayable `oldestRetainedSeq` at emission time. + let wire = r#"{"type":"terminal.output.gap","terminalId":"t1","streamId":"s1","fromSeq":1,"toSeq":9,"reason":"queue_overflow","attachRequestId":"a1","headSeq":12,"oldestRetainedSeq":2}"#; + match server_roundtrip(wire, "terminal.output.gap") { + ServerMessage::TerminalOutputGap(g) => { + assert_eq!(g.head_seq, Some(12)); + assert_eq!(g.oldest_retained_seq, Some(2)); + } + other => panic!("expected TerminalOutputGap, got {other:?}"), + } + + // Non-negotiated shape: both fields stay absent — the frozen client's + // gap frame is byte-identical to the pre-contract wire. + let wire = r#"{"type":"terminal.output.gap","terminalId":"t1","streamId":"s1","fromSeq":1,"toSeq":9,"reason":"queue_overflow"}"#; + match server_roundtrip(wire, "terminal.output.gap") { + ServerMessage::TerminalOutputGap(g) => { + assert_eq!(g.head_seq, None); + assert_eq!(g.oldest_retained_seq, None); + } + other => panic!("expected TerminalOutputGap, got {other:?}"), + } +} + +#[test] +fn terminal_replay_credit_roundtrips_and_conforms() { + // Responsive-terminal-restore Workstream 1 (paced replay): the + // continuation-credit message a pacedTerminalReplayV1 client sends after + // consuming an ordered replay page — additive optional, protocol version + // stays 10. Task 3 owns the Rust struct + Zod schema in lockstep; the + // client's sending behavior is task 4. + let wire = r#"{"type":"terminal.replay.credit","terminalId":"t1","streamId":"s1","attachRequestId":"a1","consumedSeq":41}"#; + match client_roundtrip(wire, "terminal.replay.credit") { + ClientMessage::TerminalReplayCredit(credit) => { + assert_eq!(credit.terminal_id, "t1"); + assert_eq!(credit.stream_id, "s1"); + assert_eq!(credit.attach_request_id, "a1"); + assert_eq!(credit.consumed_seq, 41); + } + other => panic!("expected TerminalReplayCredit, got {other:?}"), + } +} + #[test] fn terminal_created_roundtrips_with_and_without_notice() { // Base shape: notice omitted — byte-identical to today's frame on the wire. diff --git a/crates/freshell-protocol/tests/terminal_lifetime_claim.rs b/crates/freshell-protocol/tests/terminal_lifetime_claim.rs new file mode 100644 index 000000000..ab8ce42a2 --- /dev/null +++ b/crates/freshell-protocol/tests/terminal_lifetime_claim.rs @@ -0,0 +1,127 @@ +//! Hidden-pane lifetime-claim wire frames (responsive-terminal-restore +//! Workstream 1): the `terminalLifetimeClaimV1` capability on `hello`, its +//! advertisement on `ready`, and the additive optional +//! `terminal.interest.claimedTerminalIds` field. Additive surface only; +//! `protocolVersion` stays put (no bump — older peers accept-and-strip). + +use freshell_protocol::{ClientMessage, Ready, ReadyCapabilities, ServerMessage}; +use serde_json::json; + +// --- capability (hello) ------------------------------------------------------ + +#[test] +fn hello_capabilities_parse_terminal_lifetime_claim_v1() { + let wire = json!({ + "type": "hello", + "protocolVersion": freshell_protocol::WS_PROTOCOL_VERSION, + "token": "t", + "capabilities": { "terminalLifetimeClaimV1": true } + }); + let msg: ClientMessage = serde_json::from_value(wire).expect("hello parses"); + let ClientMessage::Hello(hello) = msg else { + panic!("expected hello"); + }; + assert_eq!( + hello + .capabilities + .and_then(|c| c.terminal_lifetime_claim_v1), + Some(true) + ); +} + +#[test] +fn hello_capabilities_omit_terminal_lifetime_claim_v1_when_absent() { + // Frozen-client shape: no terminalLifetimeClaimV1 anywhere. Round-trip + // must not invent the field (skip_serializing_if). + let wire = json!({ + "type": "hello", + "protocolVersion": freshell_protocol::WS_PROTOCOL_VERSION, + "token": "t", + "capabilities": { "terminalOutputBatchV1": true } + }); + let msg: ClientMessage = serde_json::from_value(wire.clone()).expect("hello parses"); + let back = serde_json::to_value(&msg).expect("serializes"); + assert_eq!(back, wire); +} + +// --- advertisement (ready) --------------------------------------------------- + +#[test] +fn ready_capabilities_advertise_terminal_lifetime_claim_v1_when_negotiated() { + let ready = Ready { + timestamp: "2026-09-20T00:00:00.000Z".to_string(), + boot_id: Some("boot-1".to_string()), + server_instance_id: Some("srv-1".to_string()), + build_id: None, + capabilities: Some(ReadyCapabilities { + pane_reconcile_v1: None, + pane_reconcile_fresh_agent_v1: None, + terminal_interest_v1: Some(true), + paced_terminal_replay_v1: None, + terminal_lifetime_claim_v1: Some(true), + }), + runtime_owners: None, + }; + let wire = serde_json::to_value(ServerMessage::Ready(ready)).expect("serializes"); + assert_eq!( + wire["capabilities"], + json!({ "terminalInterestV1": true, "terminalLifetimeClaimV1": true }) + ); +} + +#[test] +fn ready_capabilities_omit_terminal_lifetime_claim_v1_when_none() { + let ready = Ready { + timestamp: "2026-09-20T00:00:00.000Z".to_string(), + boot_id: Some("boot-1".to_string()), + server_instance_id: Some("srv-1".to_string()), + build_id: None, + capabilities: Some(ReadyCapabilities { + pane_reconcile_v1: Some(true), + pane_reconcile_fresh_agent_v1: None, + terminal_interest_v1: None, + paced_terminal_replay_v1: None, + terminal_lifetime_claim_v1: None, + }), + runtime_owners: None, + }; + let wire = serde_json::to_value(ServerMessage::Ready(ready)).expect("serializes"); + assert_eq!(wire["capabilities"], json!({ "paneReconcileV1": true })); +} + +// --- terminal.interest.claimedTerminalIds ------------------------------------- + +#[test] +fn terminal_interest_parses_claimed_terminal_ids() { + let wire = json!({ + "type": "terminal.interest", + "revision": 1, + "focusedTerminalId": null, + "visibleTerminalIds": [], + "claimedTerminalIds": ["T-hidden-1", "T-hidden-2"] + }); + let msg: ClientMessage = serde_json::from_value(wire).expect("interest parses"); + let ClientMessage::TerminalInterest(interest) = msg else { + panic!("expected terminal.interest"); + }; + assert_eq!( + interest.claimed_terminal_ids.as_deref(), + Some(["T-hidden-1".to_string(), "T-hidden-2".to_string()].as_slice()) + ); +} + +#[test] +fn terminal_interest_round_trip_keeps_frozen_shape_without_claims() { + // The frozen client's snapshot must serialize byte-identically: the new + // field is skipped when None (older servers accept-and-strip anyway, but + // the wire stays the committed shape). + let wire = json!({ + "type": "terminal.interest", + "revision": 1, + "focusedTerminalId": "A", + "visibleTerminalIds": ["A"] + }); + let msg: ClientMessage = serde_json::from_value(wire.clone()).expect("interest parses"); + let back = serde_json::to_value(&msg).expect("serializes"); + assert_eq!(back, wire); +} diff --git a/crates/freshell-server/src/main.rs b/crates/freshell-server/src/main.rs index 01eb4f155..49bb98972 100644 --- a/crates/freshell-server/src/main.rs +++ b/crates/freshell-server/src/main.rs @@ -2052,6 +2052,44 @@ async fn main() -> ExitCode { as std::pin::Pin + Send>> }) }); + // TERM-09 backpressure config, fail-fast validated + // (responsive-terminal-restore Workstream 3): refuse to boot on a + // configuration whose disconnect threshold sits at or below the spill + // bound — the production incident's inverted shape. Structured event for + // the JSONL log (names the offending env vars), plain stderr line for + // the console, matching the AUTH_TOKEN refusal pattern. + let term09 = match resolve_term09_config() { + Ok(config) => config, + Err(error) => { + tracing::error!(error = %error, "server.config.term09_invalid"); + eprintln!("{error}"); + return ExitCode::FAILURE; + } + }; + // Responsive-terminal-restore round-5 (finding 1, degenerate + // settings): clamp the paced-replay page budget to the queue-derived + // admission ceiling — the ONE place both knobs are known. The drain's + // reserve-then-admit gate grants a page while backlog + reservations + // + page stay at-or-below the watermark (queue cap / 2); a page + // budget above that ceiling could only admit into a fully drained + // queue, and one above the whole cap would self-spill on admission + // (a 64 KiB queue — the supported floor — is smaller than the default + // 128 KiB page). With default settings this is a no-op (128 KiB << + // the 8 MiB watermark); the clamp only bites the small-queue + // settings, honestly bounding pages to what the queue can carry. + { + let paced_page_ceiling = + freshell_ws::backpressure::paced_page_budget_ceiling(term09.queue_max_bytes); + if registry.paced_page_max_bytes() > paced_page_ceiling { + tracing::info!( + page_budget = registry.paced_page_max_bytes(), + clamped_page_budget = paced_page_ceiling, + queue_max_bytes = term09.queue_max_bytes, + "server.config.paced_page_budget_clamped" + ); + registry.set_paced_page_max_bytes(paced_page_ceiling); + } + } let ws_state = WsState { auto_resume_tx, auto_resume_cancels: Default::default(), @@ -2110,7 +2148,7 @@ async fn main() -> ExitCode { hello_timeout_ms: resolve_hello_timeout_ms(), allowed_origins: Arc::new(resolve_allowed_origins()), ws_max_payload_bytes: resolve_ws_max_payload_bytes(), - term09: freshell_ws::backpressure::Term09Config::from_env(), + term09, create_protect, // THE kata-enn3 pin: the WS door holds the SAME gate Arc as the // REST door (never a second budget minted here). @@ -3471,6 +3509,23 @@ fn resolve_ws_max_payload_bytes() -> usize { .unwrap_or(16 * 1024 * 1024) } +/// TERM-09 terminal-stream backpressure config from env, fail-fast validated +/// (responsive-terminal-restore Workstream 3): normal output pressure must +/// reach bounded admission/spill (eviction + generation-scoped gap) +/// STRICTLY before any pressure-related disconnect, so a configuration whose +/// disconnect threshold sits at or below the spill bound refuses to boot — +/// the caller logs a structured error naming the offending env vars and +/// exits before binding any port. Validation lives HERE (the boot/env path), +/// not inside `WsState`: test harnesses inject `Term09Config` values +/// directly, including deliberately inverted shapes that make the monitor's +/// last-resort window observable end to end. +fn resolve_term09_config( +) -> Result { + let config = freshell_ws::backpressure::Term09Config::from_env(); + config.validate()?; + Ok(config) +} + /// SAFE-03: resolve the WS Origin allow-list from process env, mirroring /// `server/auth.ts#parseAllowedOrigins` (`ALLOWED_ORIGINS`) plus /// `server/network-manager.ts`'s user-facing `EXTRA_ALLOWED_ORIGINS` knob @@ -4540,6 +4595,96 @@ mod tests { } } + /// TERM-09 boot wiring (responsive-terminal-restore Workstream 3): the + /// boot resolution must fail fast on a configuration that would restore + /// the spill≥disconnect inversion, naming the offending env vars — and + /// accept every correctly-ordered shape. Env-dependent cases in ONE test + /// fn (whole-process env mutation; no other test in this crate reads the + /// TERMINAL_* vars). + #[test] + fn term09_boot_resolution_fails_fast_on_inverted_env() { + let _queue = EnvVarGuard::unset("TERMINAL_CLIENT_QUEUE_MAX_BYTES"); + let _catastrophic = EnvVarGuard::unset("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES"); + let _stall = EnvVarGuard::unset("TERMINAL_WS_CATASTROPHIC_STALL_MS"); + + // Defaults must boot and satisfy the strict ordering. + let defaults = resolve_term09_config().expect("defaults must resolve"); + assert!(defaults.catastrophic_buffered_bytes > defaults.queue_max_bytes); + + // An inverted override (disconnect below spill — the production + // incident's shape) refuses to boot, naming both env vars. + let _inverted = EnvVarGuard::set("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES", "1048576"); + let err = resolve_term09_config().expect_err("inverted env must refuse boot"); + let message = err.to_string(); + assert!( + message.contains("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES") + && message.contains("TERMINAL_CLIENT_QUEUE_MAX_BYTES"), + "the boot error must name the offending env vars: {message}" + ); + drop(_inverted); + + // An equalized pair (disconnect == spill) also refuses. + let _equalized_queue = EnvVarGuard::set("TERMINAL_CLIENT_QUEUE_MAX_BYTES", "268435456"); + let _equalized_cat = + EnvVarGuard::set("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES", "268435456"); + resolve_term09_config().expect_err("equalized env must refuse boot"); + drop(_equalized_queue); + drop(_equalized_cat); + + // A correctly-ordered override boots. + let _ordered_queue = EnvVarGuard::set("TERMINAL_CLIENT_QUEUE_MAX_BYTES", "2097152"); + let _ordered_cat = EnvVarGuard::set("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES", "8388608"); + let tuned = resolve_term09_config().expect("ordered env must boot"); + assert_eq!(tuned.queue_max_bytes, 2 * 1024 * 1024); + assert_eq!(tuned.catastrophic_buffered_bytes, 8 * 1024 * 1024); + } + + /// Responsive-terminal-restore round-5 (finding 1, degenerate + /// settings): the boot wiring after `resolve_term09_config` clamps + /// the paced page budget to the queue-derived admission ceiling, so a + /// supported floor-sized queue never faces a default page larger + /// than the queue itself — the clamp the reserve-then-admit drain + /// gate relies on for its always-grantable, never-self-spilling + /// admissions. + #[test] + fn paced_page_budget_clamps_to_the_queue_admission_ceiling() { + // The exact relationship the boot applies (budget vs ceiling). + let clamp = |budget: i64, queue_max_bytes: usize| { + budget.min(freshell_ws::backpressure::paced_page_budget_ceiling( + queue_max_bytes, + )) + }; + // Default settings: the default 128 KiB budget is far below the + // 16 MiB queue's 8 MiB ceiling — the clamp is a no-op. + assert_eq!( + clamp( + freshell_terminal::DEFAULT_PACED_PAGE_MAX_BYTES, + 16 * 1024 * 1024, + ), + freshell_terminal::DEFAULT_PACED_PAGE_MAX_BYTES, + ); + // THE DEGENERATE FLOOR: the supported 64 KiB queue (the TERM-09 + // env floor) is SMALLER than the default 128 KiB page — the boot + // clamp caps the registry's budget at the 32 KiB watermark. + assert_eq!( + clamp( + freshell_terminal::DEFAULT_PACED_PAGE_MAX_BYTES, + freshell_ws::backpressure::TERM09_QUEUE_MAX_BYTES_FLOOR, + ), + 32 * 1024, + "a 64 KiB queue must never carry a 128 KiB page" + ); + // The wiring's mechanism: the registry carries the clamped value + // through the real setter/getter pair the boot uses. + let registry = freshell_terminal::TerminalRegistry::new(); + let clamped = clamp( + registry.paced_page_max_bytes(), + freshell_ws::backpressure::TERM09_QUEUE_MAX_BYTES_FLOOR, + ); + registry.set_paced_page_max_bytes(clamped); + assert_eq!(registry.paced_page_max_bytes(), 32 * 1024); + } + fn env_test_temp_dir(tag: &str) -> std::path::PathBuf { let dir = std::env::temp_dir().join(format!( "frs-main-{tag}-{}-{}", diff --git a/crates/freshell-server/src/session_name_native_tests.rs b/crates/freshell-server/src/session_name_native_tests.rs index 6f0df1b91..6d9b12d69 100644 --- a/crates/freshell-server/src/session_name_native_tests.rs +++ b/crates/freshell-server/src/session_name_native_tests.rs @@ -609,7 +609,7 @@ async fn undelivered_failures_bound_at_three_cycles_six_reads_and_exhaust_perman // by their constants; this hook only accelerates eligibility). crate::session_names::set_test_hooks( dir.path(), - vec![crate::session_names::TestHook::NativeRetryFloorMs(60)], + vec![crate::session_names::TestHook::NativeRetryFloorMs(2_000)], ); let store = open_store(dir.path()); let target = armed_pending(&store, "h-exhaust", "/h/.claude").await; @@ -643,11 +643,15 @@ async fn undelivered_failures_bound_at_three_cycles_six_reads_and_exhaust_perman .is_none(), "not yet due consumes nothing" ); - tokio::time::sleep(Duration::from_millis(80)).await; + // The 2s floor gives the not-yet-due asserts above real margin: a + // parallel-workspace run can starve a single await well past a 60ms + // floor (observed twice under full-suite load), which made the + // "cycle 2 waits for the retry floor" assert fail on working code. + tokio::time::sleep(Duration::from_millis(2_050)).await; // Exhaust cycles 2 and 3 through the real dispatch loop. super::run_cycle(&store, backend.as_ref() as &dyn NativeNameBackend, &target).await; - tokio::time::sleep(Duration::from_millis(80)).await; + tokio::time::sleep(Duration::from_millis(2_050)).await; super::run_cycle(&store, backend.as_ref() as &dyn NativeNameBackend, &target).await; assert_eq!(backend.writes(), 3, "at most three writes per revision"); assert_eq!(backend.reads(), 3); @@ -1366,7 +1370,7 @@ async fn native_work_snapshot_reads_the_test_offset_aware_clock() { let dir = temp_data_dir(); crate::session_names::set_test_hooks( dir.path(), - vec![crate::session_names::TestHook::NativeRetryFloorMs(60)], + vec![crate::session_names::TestHook::NativeRetryFloorMs(2_000)], ); let store = open_store(dir.path()); let target = armed_pending(&store, "h-clock", "/h/.claude").await; @@ -1401,7 +1405,7 @@ async fn native_work_snapshot_reads_the_test_offset_aware_clock() { crate::session_names::set_test_hooks( dir.path(), vec![ - crate::session_names::TestHook::NativeRetryFloorMs(60), + crate::session_names::TestHook::NativeRetryFloorMs(2_000), crate::session_names::TestHook::ClockOffsetMs(60_000), ], ); diff --git a/crates/freshell-terminal/src/batch.rs b/crates/freshell-terminal/src/batch.rs index b5b06c73c..3955be8d0 100644 --- a/crates/freshell-terminal/src/batch.rs +++ b/crates/freshell-terminal/src/batch.rs @@ -133,6 +133,41 @@ fn measure_json_bytes(value: &Value) -> usize { .len() } +/// The FIXED scaffold of the legacy `terminal.output` envelope — every byte +/// of [`measure_legacy_output_bytes`] except the `seqStart`/`seqEnd` digit +/// widths and the JSON-escaped `data` literal — measured once so a paced +/// page walk can account frames INCREMENTALLY (responsive-terminal-restore): +/// +/// `measure(seq_start, seq_end, data) == scaffold + digits(seq_start) +/// + digits(seq_end) + json_escaped_len(data)` +/// +/// exactly, with no per-candidate re-serialization of the accumulated run. +pub(crate) fn legacy_envelope_scaffold_bytes( + terminal_id: &str, + stream_id: &str, + attach_request_id: Option<&str>, + source: Option<&str>, +) -> usize { + // seqs "0"/"0" contribute one digit each; the empty data contributes + // exactly its two quote characters. + measure_legacy_output_bytes(terminal_id, stream_id, 0, 0, "", attach_request_id, source) + .saturating_sub(4) +} + +/// The compact-JSON byte length of a string payload INCLUDING its quotes +/// (exactly the `"data":` segment the envelope measure accounts). +pub(crate) fn json_escaped_len(data: &str) -> usize { + serde_json::to_string(data) + .expect("terminal data is always serializable") + .len() +} + +/// Decimal digit count of an i64 (0 has one digit; negatives never occur for +/// seqs but stay honest). +pub(crate) fn digit_count(n: i64) -> usize { + n.checked_abs().unwrap_or(i64::MAX).to_string().len() +} + /// `defaultPayloadForFrame` (`output-batch.ts:83-99`) measured as the legacy /// `terminal.output` envelope — the merge-budget size for `data`. fn measure_legacy_output_bytes( diff --git a/crates/freshell-terminal/src/fragment.rs b/crates/freshell-terminal/src/fragment.rs index 4c150bdec..057b218f2 100644 --- a/crates/freshell-terminal/src/fragment.rs +++ b/crates/freshell-terminal/src/fragment.rs @@ -48,13 +48,76 @@ pub fn attach_request_id_reserve_value() -> String { /// `TERMINAL_STREAM_BATCH_MAX_BYTES = max(1024, env.TERMINAL_STREAM_BATCH_MAX_BYTES /// || MAX_REALTIME_MESSAGE_BYTES)` (`constants.ts:3-6`). Env override honored for /// fidelity; unset -> 16384. +/// +/// Round-2 finding F2 — the frame-fits-page invariant, enforced at the +/// source: the result is CLAMPED to [`PACED_PAGE_BUDGET_FLOOR_BYTES`] (the +/// smallest paced page budget any supported TERM-09 queue setting can +/// produce), so no env override can mint a terminal.output frame whose +/// serialized size exceeds the page budget floor. Every PTY byte is +/// ingested through this cap (the OutputFramer fragments at construction), +/// which makes the paced page builder's over-budget first frame — its +/// atomic single-frame page — unreachable under supported settings; that +/// arm stays as defense-in-depth. pub fn terminal_stream_batch_max_bytes() -> usize { - let from_env = std::env::var("TERMINAL_STREAM_BATCH_MAX_BYTES") + terminal_stream_batch_max_bytes_for_env(std::env::var("TERMINAL_STREAM_BATCH_MAX_BYTES")) +} + +/// [`terminal_stream_batch_max_bytes`] resolved against an injected env +/// read (the pure core — testable without process-env races). +pub fn terminal_stream_batch_max_bytes_for_env( + from_env: Result, +) -> usize { + let from_env = from_env .ok() .and_then(|v| v.trim().parse::().ok()) .filter(|n| n.is_finite() && *n > 0.0) .map(|n| n.floor() as usize); - from_env.unwrap_or(MAX_REALTIME_MESSAGE_BYTES).max(1024) + from_env + .unwrap_or(MAX_REALTIME_MESSAGE_BYTES) + .clamp(1024, PACED_PAGE_BUDGET_FLOOR_BYTES) +} + +/// The paced page-budget FLOOR (responsive-terminal-restore round-2 +/// finding F2): the smallest page budget any supported deployment can hand +/// the paced page builder. The page budget is clamped at server boot to +/// `paced_page_budget_ceiling(queue_max_bytes) = queue/2` +/// (`freshell_ws::backpressure`), and the queue cap is floor-enforced at +/// 64 KiB (`TERM09_QUEUE_MAX_BYTES_FLOOR`, `Term09Config::validate`) — so +/// 32 KiB is the minimum ceiling any supported settings can produce. The +/// fragment cap is clamped to this floor so the frame-fits-page invariant +/// holds structurally, whatever `TERMINAL_STREAM_BATCH_MAX_BYTES` says. +/// (freshell-ws pins the cross-crate agreement: +/// `paced_page_budget_ceiling(TERM09_QUEUE_MAX_BYTES_FLOOR) == this`.) +pub const PACED_PAGE_BUDGET_FLOOR_BYTES: usize = 32 * 1024; + +/// The fixed page-envelope slack atop one frame's fragment-cap measure +/// (E2R1 finding 2): a page message wraps the frame with the real +/// envelope — a plain `terminal.output` (already accounted by the +/// fragment measure's worst-case seq and 512-char attachRequestId +/// reserve) or a `terminal.output.batch` (whose envelope adds the type +/// delta, `serializedBytes` digits, the `source` literal, and one +/// segment's metadata on top of the legacy measure). 512 bytes provably +/// covers both forms for a maximal fragment (pinned by +/// [`paced_atomic_page_serialized_ceiling`]'s test and the registry's +/// atomic-page ceiling test). +pub const ATOMIC_PAGE_ENVELOPE_OVERHEAD_BYTES: usize = 512; + +/// The worst-case serialized size of ONE paced replay page's ATOMIC +/// single-frame result (E2R1 finding 2): the page builder always +/// includes the first frame of a window even when that frame alone +/// exceeds the requested page budget — a single frame larger than the +/// request forms its own atomic page — and every frame is +/// pre-fragmented to at most [`terminal_stream_batch_max_bytes`] +/// (measured as the full `terminal.output` payload with a worst-case +/// seq and a 512-char attachRequestId reserve). A page therefore never +/// exceeds this ceiling WHATEVER the client requested: pages are bounded +/// by max(requested, the atomic frame size), and the atomic frame size +/// is bounded by the fragment cap plus the page envelope. Connection +/// drain admission reserves THIS ceiling when the requested page budget +/// sits below it, so a sub-cap request can never under-reserve what its +/// drain can actually admit. +pub fn paced_atomic_page_serialized_ceiling() -> usize { + terminal_stream_batch_max_bytes() + ATOMIC_PAGE_ENVELOPE_OVERHEAD_BYTES } /// `measureSerializedJsonBytes` — UTF-8 byte length of the compact JSON serialization. @@ -169,11 +232,171 @@ mod tests { fn batch_max_defaults_to_16k() { // Env unset in the test harness -> max(1024, 16384). assert_eq!( - terminal_stream_batch_max_bytes(), + terminal_stream_batch_max_bytes_for_env(Err(std::env::VarError::NotPresent)), MAX_REALTIME_MESSAGE_BYTES ); } + #[test] + fn fragment_cap_is_clamped_to_the_paced_page_budget_floor() { + // Round-2 finding F2 (the boot-clamp, applied at the source): an + // env override LARGER than the paced page budget floor must not + // survive — the effective fragment cap can never exceed the + // smallest page budget any supported queue setting can produce, + // or the env could mint a frame the page builder cannot pack. + assert_eq!( + terminal_stream_batch_max_bytes_for_env(Ok("1048576".to_string())), + PACED_PAGE_BUDGET_FLOOR_BYTES, + "a 1 MiB TERMINAL_STREAM_BATCH_MAX_BYTES override clamps to the 32 KiB page floor" + ); + // A value BELOW the floor keeps fidelity (the clamp is an upper + // bound, not a fixed size). + assert_eq!( + terminal_stream_batch_max_bytes_for_env(Ok("2048".to_string())), + 2048 + ); + // Garbage and non-positive values keep the default. + assert_eq!( + terminal_stream_batch_max_bytes_for_env(Ok("not-a-number".to_string())), + MAX_REALTIME_MESSAGE_BYTES + ); + assert_eq!( + terminal_stream_batch_max_bytes_for_env(Ok("0".to_string())), + MAX_REALTIME_MESSAGE_BYTES + ); + } + + #[test] + fn a_control_heavy_chunk_measures_far_above_the_page_floor_but_splits_within_the_clamp() { + // The round-2 reviewer's reachability arithmetic, pinned as the + // class-closing evidence: the PTY reads at most 8 KiB per chunk, + // and a control-char-heavy chunk serializes to ~6 bytes per char + // under JSON escaping (~48 KiB for 8192 chars) — FAR above the + // 32 KiB page floor. But every PTY byte is ingested through the + // fragment splitter whose budget measure is the full serialized + // terminal.output JSON with the worst-case seq placeholder and + // the 512-char attachRequestId reserve, so the emitted FRAMES can + // never exceed the clamped cap — the frame-fits-page invariant. + let chunk = "\u{1}".repeat(8192); + let measured = measure_terminal_output_budget_payload_bytes("term", "stream", &chunk); + assert!( + measured > 48 * 1024, + "the reviewer's ~49 KiB measure: 8192 control chars escape to ~6 bytes each ({measured})" + ); + assert!( + measured > PACED_PAGE_BUDGET_FLOOR_BYTES, + "the raw chunk measurably exceeds the page floor — only the splitter's fragments reach the ring ({measured})" + ); + let fragments = fragment_terminal_output_for_payload_budget( + &chunk, + terminal_stream_batch_max_bytes_for_env(Ok("1048576".to_string())), + |c| measure_terminal_output_budget_payload_bytes("term", "stream", c), + ) + .expect("the clamped budget always fits one code point"); + assert!(fragments.len() > 1, "the chunk must split"); + assert_eq!(fragments.concat(), chunk, "the split is lossless"); + for fragment in &fragments { + let frame_bytes = + measure_terminal_output_budget_payload_bytes("term", "stream", fragment); + assert!( + frame_bytes <= PACED_PAGE_BUDGET_FLOOR_BYTES, + "every frame fits the smallest supported page budget ({frame_bytes})" + ); + } + } + + #[test] + fn paced_atomic_page_ceiling_covers_a_maximal_fragment_in_every_page_form() { + // E2R1 finding 2: the ceiling is the honest bound for the page + // builder's ATOMIC single-frame result — a frame larger than the + // requested page budget forms its own page, and that page must + // fit cap + envelope slack in EVERY wire form the projection can + // take for it. Build a MAXIMAL fragment (its budgeted + // terminal.output measure fits the fragment cap; one more ASCII + // char would not) and prove both forms. + let cap = terminal_stream_batch_max_bytes(); + let ceiling = paced_atomic_page_serialized_ceiling(); + assert!(ceiling > cap, "the ceiling is the cap plus envelope slack"); + + let mut len = cap; + while measure_terminal_output_budget_payload_bytes( + "term-atomic", + "stream", + &"A".repeat(len), + ) > cap + { + len -= 1; + } + let data = "A".repeat(len); + let measured = measure_terminal_output_budget_payload_bytes("term-atomic", "stream", &data); + assert!( + measured <= cap, + "the maximal fragment fits the fragment cap" + ); + assert!( + measure_terminal_output_budget_payload_bytes( + "term-atomic", + "stream", + &"A".repeat(len + 1) + ) > cap, + "maximality: one more char would exceed the cap" + ); + + // Form 1 — the plain `terminal.output` page message, stamped with + // the WORST-CASE attachRequestId (512 chars) and real seq digits + // (far under the measure's placeholder width). + let page = json!({ + "type": "terminal.output", + "terminalId": "term-atomic", + "streamId": "stream", + "seqStart": 1, + "seqEnd": 2, + "data": data, + "attachRequestId": attach_request_id_reserve_value(), + "source": "replay", + }); + let plain_bytes = measure_serialized_json_bytes(&page); + assert!( + plain_bytes <= ceiling, + "the plain page form fits the atomic ceiling ({plain_bytes} > {ceiling})" + ); + + // Form 2 — the real batch projection over the same frame, at the + // production budget (batch_max = the fragment cap): whatever wire + // shape the single maximal frame takes — the full + // `terminal.output.batch`, or the oversize single-segment + // fallback — its serialized size must fit the ceiling. + let mut scanner = crate::barrier_scanner::BarrierScanner::new(); + let frame = crate::batch::BatchInputFrame::classified(7, &data, &mut scanner, "stream"); + let payloads = crate::batch::frames_to_wire_payloads( + &[frame], + "term-atomic", + "arid-max", + "replay", + cap as i64, + ); + assert_eq!( + payloads.len(), + 1, + "one maximal frame projects to exactly one page payload" + ); + let batch_bytes = measure_serialized_json_bytes(&payloads[0]); + assert!( + batch_bytes <= ceiling, + "the batch page form fits the atomic ceiling ({batch_bytes} > {ceiling})" + ); + // The envelope slack is honest, not arbitrary: the batch form's + // fixed envelope (type delta, serializedBytes digits, the source + // literal, one segment's metadata) fits inside the documented + // overhead constant. + assert!( + batch_bytes.saturating_sub(measured) <= ATOMIC_PAGE_ENVELOPE_OVERHEAD_BYTES, + "the batch envelope overhead stays within the documented slack ({} > {})", + batch_bytes.saturating_sub(measured), + ATOMIC_PAGE_ENVELOPE_OVERHEAD_BYTES + ); + } + #[test] fn measure_matches_json_stringify_byte_length() { // Control-char escaping parity with JSON.stringify: "\r\n" -> 4 escaped bytes. diff --git a/crates/freshell-terminal/src/lib.rs b/crates/freshell-terminal/src/lib.rs index 7bd20bbc1..c497ae406 100644 --- a/crates/freshell-terminal/src/lib.rs +++ b/crates/freshell-terminal/src/lib.rs @@ -57,12 +57,15 @@ pub use batch::{ }; pub use chunk_ring::{snapshot_seed_if_ring_empty, ChunkRingBuffer}; pub use decode::Utf8StreamDecoder; +pub use fragment::{paced_atomic_page_serialized_ceiling, PACED_PAGE_BUDGET_FLOOR_BYTES}; pub use framing::{reassemble_stream, OutputFramer}; pub use mode_tracker::ModeTracker; pub use pty::{build_child_env, build_child_env_from_process, MessageSink, PtyTerminal}; pub use registry::{ - compute_scrollback_max_bytes, ActivityEvent, ActivityObserver, AttachOutcome, FrameSink, - InputOutcome, StuckTransition, TerminalRegistry, DEFAULT_STUCK_WINDOW_MS, + compute_scrollback_max_bytes, ActivityEvent, ActivityObserver, AttachOutcome, ClaimState, + FrameSink, InputOutcome, PacedAttachOptions, PacedAttachStart, PacedExitNotify, + PacedGapExitReason, PacedPage, PacedSessionDesc, PacedTailCompletion, ReplayBounds, + StuckTransition, TerminalRegistry, DEFAULT_PACED_PAGE_MAX_BYTES, DEFAULT_STUCK_WINDOW_MS, STUCK_ACTIVITY_FRESH_MS, }; pub use replay_ring::{ReplayDeque, ReplayFrame, ReplayRing}; diff --git a/crates/freshell-terminal/src/output_queue.rs b/crates/freshell-terminal/src/output_queue.rs index caf54da36..82fda7b87 100644 --- a/crates/freshell-terminal/src/output_queue.rs +++ b/crates/freshell-terminal/src/output_queue.rs @@ -17,9 +17,18 @@ use freshell_protocol::ServerMessage; -/// Default cap (legacy: `client-output-queue.ts:33` -/// `DEFAULT_TERMINAL_CLIENT_QUEUE_MAX_BYTES = 32 * 1024 * 1024`). -pub const DEFAULT_TERMINAL_CLIENT_QUEUE_MAX_BYTES: usize = 32 * 1024 * 1024; +/// Default cap (responsive-terminal-restore Workstream 3: the SPILL bound — +/// eviction + generation-scoped gap — that normal output pressure reaches +/// strictly before any pressure-related disconnect; the disconnect threshold +/// lives in `freshell-ws::backpressure::Term09Config` and must stay strictly +/// above this). Legacy `client-output-queue.ts:33` shipped 32 MiB, which sat +/// ABOVE legacy's 16 MiB catastrophic-disconnect threshold — the inversion +/// that disconnected the production incident's ~21-25 MB backlog instead of +/// spilling it. 16 MiB keeps a multi-second grace buffer for a slow-but- +/// draining client while the byte-fair scheduler keeps other panes +/// responsive; the incident backlog spills gracefully here instead of +/// disconnecting. +pub const DEFAULT_TERMINAL_CLIENT_QUEUE_MAX_BYTES: usize = 16 * 1024 * 1024; /// The identity fields a queued output frame needs so a gap event can be /// built if it's later evicted. Mirrors the fields `ReplayFrame` carries in diff --git a/crates/freshell-terminal/src/registry.rs b/crates/freshell-terminal/src/registry.rs index e97ad7103..52cf5a1f2 100644 --- a/crates/freshell-terminal/src/registry.rs +++ b/crates/freshell-terminal/src/registry.rs @@ -42,7 +42,7 @@ //! this crate keeps its no-tokio boundary (`freshell-ws` backs the sink with a tokio //! mpsc sender feeding the socket). -use std::collections::{HashMap, VecDeque}; +use std::collections::{BTreeSet, HashMap, VecDeque}; use std::io; use std::sync::atomic::{AtomicI64, AtomicU64, Ordering}; use std::sync::{Arc, Mutex}; @@ -51,7 +51,8 @@ use freshell_platform::SpawnSpec; use freshell_protocol::{ GeometryAuthority, InventoryTerminal, OutputSource, ServerMessage, SessionLocator, TerminalAttachIntent, TerminalAttachReady, TerminalExit, TerminalModesSync, TerminalOutput, - TerminalRunStatus, TerminalStuck, + TerminalOutputGap, TerminalOutputGapReason, TerminalReplayResetReason, TerminalRunStatus, + TerminalStuck, }; use crate::barrier_scanner::{BarrierReason, BarrierScanner, ScannerState}; @@ -81,6 +82,15 @@ const MIN_SCROLLBACK_CHARS: i64 = 64 * 1024; const MAX_SCROLLBACK_CHARS: i64 = 4 * 1024 * 1024; /// `APPROX_CHARS_PER_LINE` (`terminal-registry.ts:60`). const APPROX_CHARS_PER_LINE: i64 = 300; +/// Responsive-terminal-restore Workstream 1: the default serialized-byte +/// budget of ONE paced replay page (the plan's "128 KiB initial terminal +/// batch target" — a per-page delivery bound, NOT a claim about total +/// reconstruction size). The budget covers the JSON envelope, escaping, +/// and batch-segment metadata; a single frame whose own envelope exceeds it +/// forms its own atomic single-frame page (guaranteed progress). Held as a +/// registry-level atomic (like `scrollback_max_bytes`) so focused tests can +/// shrink it per-instance without env races. +pub const DEFAULT_PACED_PAGE_MAX_BYTES: i64 = 128 * 1024; /// `computeScrollbackMaxChars(settings)` (`terminal-registry.ts:1328-1333`): /// `settings.terminal.scrollback` LINES converted to an approximate **CHAR** @@ -123,6 +133,290 @@ fn now_ms() -> i64 { freshell_platform::clock::now_ms() } +/// Current restore-contract sequence bounds for one terminal's retained +/// replay ring (responsive-terminal-restore shared contract): the +/// terminal's head sequence and the earliest sequence position still +/// available for replay. Payload-free by design — the writer's gap frames +/// stamp these bounds without copying any retained output. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ReplayBounds { + /// Highest produced `seqEnd` (drives `attach.ready.headSeq`). + pub head_seq: i64, + /// The retained ring's front `seqStart`, or `head_seq + 1` when the ring + /// is empty (nothing older than the head is retained). + pub oldest_retained_seq: i64, +} + +/// Hidden-pane lifetime-claim observability +/// ([`TerminalRegistry::claim_state`], responsive-terminal-restore Workstream +/// 1): the claim set size and the row's fast-reap eligibility. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct ClaimState { + /// Distinct connections currently claiming this terminal's lifetime. + pub claimers: usize, + /// Whether the row is configured-threshold reap-eligible (explicitly + /// released — or never attached/claimed). + pub released_by_client: bool, +} + +/// Paced-attach inputs beyond the legacy wire fields +/// (responsive-terminal-restore round-2): the negotiated forward-page +/// limit the client requested (`replayPageBytes` — an optional UPPER BOUND +/// on each paced page's serialized bytes, clamped to the registry's own +/// [`TerminalRegistry::paced_page_max_bytes`] cap; finding F3) and the +/// connection's staged-exit notification hook (finding F1 — installed with +/// the paced subscriber, atomically with the attach, so a natural exit +/// while the deferral is armed can move the connection's session into its +/// exit-drain). Both are inert for non-paced attaches. +#[derive(Clone, Default)] +pub struct PacedAttachOptions { + /// The attach's `replayPageBytes`: `Some(positive)` bounds every page + /// of the session at `min(requested, registry cap)`; `None` (or a + /// non-positive value, filtered at the protocol layer) keeps the + /// server default. + pub replay_page_bytes: Option, + /// The connection's exit-sequencing hook (finding F1): invoked by + /// `finish_pty_exit` — OUTSIDE the terminal lock — when a natural + /// exit STAGES this subscriber's exit behind its still-armed + /// deferral. Receives (terminal_id, exit_code); the ws layer routes + /// it to the owning connection's dispatch loop, which hands the + /// session to the exit-drain. Must never acquire the originating + /// terminal's lock (it fires from the PTY reader thread). + pub paced_exit_notify: Option, +} + +/// The staged-exit notification hook for a paced subscriber (finding F1). +pub type PacedExitNotify = Arc; + +/// E2R3 (the atomic exit-transition decision): the phase-transition +/// facts for one paced subscriber, read under ONE registry lock hold — +/// the staged-exit state AND the terminal's head TOGETHER, so no window +/// exists between the read and the caller's commitment of the +/// disposition in which a concurrently staged exit can change which +/// transition was correct. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct PacedExitTransition { + /// The terminal's FROZEN final head when a natural exit is STAGED + /// for this subscriber ([`TerminalRegistry::finish_pty_exit`] + /// staged it behind the still-armed deferral): the credited phase + /// extends through it, and terminal.exit rides the credit that + /// acknowledges the page reaching it. `None` atomically confirms + /// NO exit is staged as of this decision's lock hold — an exit + /// staging after the hold is post-decision content (at the + /// transfer site: the drain's documented uncredited tail + /// semantics). The head is frozen by the exit, so a `Some` value + /// is stable for the subscriber's remainder. + pub exit_head: Option, +} + +/// E2R3 (the atomic exit-transition decision) test-support: the armed +/// payload of the ONE-SHOT natural-exit staging hook — WHICH +/// connection's subscriber stages WHICH exit code when the next hook +/// site fires (see +/// [`TerminalRegistry::set_paced_exit_stage_hook_for_tests`]). +#[derive(Debug, Clone, Copy)] +struct PacedExitStageHook { + conn_id: u64, + exit_code: i64, +} + +/// The ws pacing coordinator's session description for one paced replay +/// (responsive-terminal-restore Workstream 1): everything the coordinator +/// needs to gate continuation credits and drive page reads. Produced by the +/// paced attach, owned by the connection (one session per +/// (connection, terminal); a re-attach replaces it). +#[derive(Debug, Clone, PartialEq, Eq)] +pub struct PacedSessionDesc { + pub terminal_id: String, + pub stream_id: String, + /// The attach generation this session serves; stale-credit rejection key. + pub attach_request_id: String, + /// FIXED catch-up target: `head_seq` at attach time. Ongoing output never + /// extends it — frames past the target are tail-phase delivery. + pub target: i64, + /// The effective replay baseline (the requested `sinceSeq` clamped to + /// ≥ 0, reset to `oldest-1` on attach-time retention loss). + pub effective_since: i64, + /// The production cursor: the last seq sent in a page (== the first + /// page's last seq; `effective_since` when the first page is empty). + pub page_end: i64, + /// The session's page budget (round-2 finding F3): the clamped + /// effective bound (`min(requested replayPageBytes, registry cap)`, + /// the registry cap when absent) that sized the FIRST page and bounds + /// every later page — the whole session honors the request, not just + /// the first page. + pub page_budget: i64, + /// Serialized wire bytes of the first page (observability). + pub page_bytes: u64, +} + +/// What a negotiated (paced) attach returns to the ws layer INSTEAD of an +/// inline replay burst: the session description plus the FIRST page's wire +/// messages. The ws layer sinks the page AFTER the terminal lock is +/// released; subsequent pages are produced on continuation credit and the +/// tail range `(target, head]` is drained as ordinary delivery. +#[derive(Debug, Clone, PartialEq)] +pub struct PacedAttachStart { + pub session: PacedSessionDesc, + pub first_page: Vec, +} + +/// Result of [`TerminalRegistry::next_replay_page`] — one bounded, +/// ascending page of the `(from_seq, target]` window, the window's +/// completion, an exact retention-loss report, or the session's +/// disappearance. +#[derive(Debug, Clone, PartialEq)] +#[must_use] +pub enum PacedPage { + /// One packed page: ascending wire messages, the page's last seq (the + /// session's new cursor), and the page's total serialized bytes. + Frames { + messages: Vec, + end_seq: i64, + serialized_bytes: u64, + }, + /// `from_seq >= target` — the replay window is fully delivered. + Done, + /// Retention evicted the frames the session needs next. The exact lost + /// interval is `[lost_from, lost_to]` and the session continues from + /// `resume_from` (the new ring front − 1); `head_seq`/`oldest_retained_seq` + /// are the task-2 bounds fields for the negotiated gap frame. + Expired { + lost_from: i64, + lost_to: i64, + resume_from: i64, + head_seq: i64, + oldest_retained_seq: i64, + }, + /// The terminal (or this connection's subscriber) is gone — cancel the + /// session; there is no deferral left to clear. + Gone, +} + +/// Result of [`TerminalRegistry::complete_paced_tail`] and +/// [`TerminalRegistry::handoff_paced_tail`] — the terminal phase's PAGED, +/// FIXED-BOUNDARY completion: the drain pages toward the FIXED target +/// captured once at drain start (the caller passes it; it is NEVER +/// re-captured), each call delivers at most one budget-bounded page (the +/// lock held only per page), and when the target is covered the +/// COMPLETION BOUNDARY B is captured ONCE under that same hold (the +/// `TargetCovered` verdict — "completion start", recorded on the +/// subscriber as `paced_handoff_boundary`). The post-target handoff then +/// delivers the staged window up to B and ONLY B — one budget-bounded +/// chunk per lock hold, the lock released between chunks, never a bulk +/// retained-suffix clone, NEVER re-reading the terminal's current head +/// (the recurring moving-head chase cause; the handoff structurally +/// contains no head read). The completing hold clears the deferral +/// ATOMICALLY in the same lock hold as its delivery, and the frames the +/// producer staged past B flow through the normal live fan-out path in +/// that same hold (the completing sweep) — in seq order, so the boundary +/// is exact and ordered and can neither lose nor duplicate a frame. +/// Retention overrunning the handoff cursor mid-handoff is the plan:146 +/// bounded-baseline exit: the exact bounds-carrying gap, then the session +/// COMPLETES AT THE RING FRONT (the retained window swept through the +/// normal live path, the gap recorded) — never a resumption toward an +/// unreachable B. +/// Why a paced tail completed with a recorded gap (round-5 finding 3, +/// diagnostics): `GapCompleted` is produced by two SEMANTICALLY +/// different exits, and every structured log event about one carries +/// this mandatory reason so consumers can filter them apart. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum PacedGapExitReason { + /// A GENUINE retention overrun (the plan:146 bounded-baseline + /// exit): the ring evicted part of the range the handoff still + /// owed — the lost interval is UNFETCHABLE (sunk as a + /// `replay_window_exceeded` retention gap). + RetentionOverrun, + /// The ordinary fixed-boundary residual exit: the completing chunk + /// covered B with frames staged past it — the declared interval is + /// RETAINED and fetchable (sunk as a `handoff_boundary_reached` + /// delivery gap; the client's checkpoint-cursor repair fetches it). + HandoffBoundaryResidual, +} + +impl PacedGapExitReason { + pub fn as_str(self) -> &'static str { + match self { + Self::RetentionOverrun => "retention_overrun", + Self::HandoffBoundaryResidual => "handoff_boundary_residual", + } + } +} + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[must_use] +pub enum PacedTailCompletion { + /// Nothing was staged beyond the cursor within the completion's own + /// bounds: the deferral cleared, the session is complete. + CaughtUp, + /// One budget-bounded page of the staged range was delivered + /// through the subscriber's sink. The deferral STAYS armed (frames + /// keep staging, never interleaving with the pages); the caller + /// advances its cursor and calls again. + Handoff { end_seq: i64, serialized_bytes: u64 }, + /// The CLEARED verdict: everything staged up to the completing hold + /// was delivered in that hold — the page/chunk up to the session's + /// fixed boundary (the drain target or B), plus everything the + /// producer staged past it through the normal live fan-out path (the + /// completing sweep) — and the deferral was CLEARED in the same hold. + /// The hold is the atomic boundary: post-hold ingests fan out + /// directly (their seqs are past everything delivered), so the + /// boundary can neither lose nor duplicate a frame. `end_seq` is the + /// highest delivered seq. Ends the paced session. + Completed { end_seq: i64, serialized_bytes: u64 }, + /// The FIXED-TARGET HANDOFF (only + /// [`TerminalRegistry::complete_paced_tail`] reports this): this + /// call's page covered the FIXED drain target while the producer + /// staged frames BEYOND it. NO bulk re-fan, NO clear: the deferral + /// STAYS armed (live ingest keeps staging, so ordering with the + /// remainder is preserved) and the caller hands the staged post-target + /// remainder to [`TerminalRegistry::handoff_paced_tail`], which pages + /// it toward the completion boundary B captured ONCE under THIS hold + /// (recorded on the subscriber — the ONE sanctioned head read, at + /// completion start). The caller advances its cursor to `end_seq` and + /// switches to handoff calls. + TargetCovered { end_seq: i64, serialized_bytes: u64 }, + /// The plan:146 bounded-baseline exit: retention overran the handoff + /// cursor mid-handoff (`handoff_paced_tail`) or the completing chunk + /// covered the fixed boundary with a staged residual + /// (`complete_at_fixed_boundary`) — see [`PacedGapExitReason`] for the + /// two producers' semantics. In both shapes the EXACT bounds-carrying + /// gaps were sunk through the subscriber's sink FIRST (ordered ahead + /// of everything else), the deferral was CLEARED in the same hold, + /// and the session COMPLETES with the gap recorded — never a + /// resumption of the paged handoff toward an unreachable B (a finite + /// retention window cannot guarantee convergence against indefinitely + /// faster output production — plan:146). Ends the paced session. + GapCompleted { + lost_from: i64, + lost_to: i64, + end_seq: i64, + serialized_bytes: u64, + /// The mandatory diagnostics discriminator (round-5 finding 3): + /// which of the two gap exits produced this verdict. + reason: PacedGapExitReason, + }, + /// Retention evicted part of the range the DRAIN phase still owes + /// toward its fixed target — the exact bounds-carrying interval as + /// [`PacedPage::Expired`]: the caller must emit the retention gap and + /// continue the drain from the resumed front (the target stays FIXED, + /// so this continuation is bounded). Only + /// [`TerminalRegistry::complete_paced_tail`] reports this. A retention + /// advance must NEVER become a silent forward jump. + Expired { + lost_from: i64, + lost_to: i64, + resume_from: i64, + head_seq: i64, + oldest_retained_seq: i64, + }, + /// The terminal, this connection's subscriber, or the session's + /// attach generation is gone — cancel the session; there is no + /// deferral left for THIS generation to clear (a superseding + /// generation owns its own). + Gone, +} + /// One attached connection's subscription to a terminal's live stream. struct Subscriber { /// Where this connection's frames go (its socket, via a tokio mpsc in `freshell-ws`). @@ -136,6 +430,65 @@ struct Subscriber { /// this is set AND `attach_request_id` is present (`broker.ts:1315-1343`); otherwise /// the connection receives legacy per-frame `terminal.output` (the T1 default). terminal_output_batch_v1: bool, + /// `hello.capabilities.pacedTerminalReplayV1` for this connection + /// (responsive-terminal-restore Workstream 1): parked on the subscriber + /// exactly like `terminal_output_batch_v1`. Gates the registry's paced + /// page reads — a subscriber that did not negotiate never serves them. + paced_terminal_replay_v1: bool, + /// Restore contract (responsive-terminal-restore): while a paced replay + /// session is active for this (connection, terminal), [`ingest`] does NOT + /// fan output out to this subscriber — the retained ring IS the staging + /// and the session's pages deliver the range in seq order. Armed by the + /// paced attach, cleared ATOMICALLY under the per-terminal lock by the + /// session's completing verdicts (`complete_paced_tail`'s page that + /// covers everything staged under its hold; `handoff_paced_tail`'s + /// completing hold at the FIXED boundary, or its plan:146 ring-front + /// exit) — the SAME lock hold that delivers everything staged up to the + /// clear — so the flag-clear boundary can neither lose nor duplicate a + /// frame: everything appended before the clear is paged, handed off, or + /// swept by the completing hold's normal-live-path delivery, and + /// everything appended after it is fanned out directly. The staged + /// post-target remainder is NEVER bulk-cloned: it flows through the + /// bounded post-target handoff, one page-budget-sized chunk per lock + /// hold. + paced_deferred: bool, + /// The FIXED completion boundary B (round-4): the head captured ONCE + /// under the lock hold that covers the drain's fixed target (the + /// `TargetCovered` verdict — "completion start"). The post-target + /// handoff ([`TerminalRegistry::handoff_paced_tail`]) pages the + /// staged window up to B and ONLY B; the field exists so the boundary + /// survives across the handoff's per-chunk lock holds WITHOUT any + /// re-read of the terminal's CURRENT head (the recurring moving-head + /// chase cause). Set exactly once per session (the TargetCovered + /// hold), consumed by the handoff's completing verdicts, swept with + /// the subscriber by re-attach/socket close. + paced_handoff_boundary: Option, + /// Round-2 finding F1: the STAGED natural-exit code — set by + /// `finish_pty_exit` when the terminal exits while this subscriber's + /// deferral is still armed. The staged exit is NOT sunk then: the + /// session's drain drives the deferred final output to completion, + /// and the completing verdict delivers the staged exit in the SAME + /// lock hold (ordered after the final pages through the same sink) + /// and retires the subscriber — the client observes final output + /// THEN exit (plan:190 sequenced exit delivery). Cleared with the + /// subscriber on delivery, detach, socket close, or supersede. + paced_exit_pending: Option, + /// Round-2 finding F1: the connection's staged-exit notification + /// hook (see [`PacedAttachOptions::paced_exit_notify`]) — invoked + /// once, OUTSIDE the terminal lock, at staging time so the ws layer + /// can move a still-CREDITED session into its exit-drain. + paced_exit_notify: Option, + /// TERM-07 seam: the attach's `maxReplayBytes` request, threaded through + /// BOTH attach paths and recorded here with NO delivery-behavior change + /// this increment. The plan's binding rule preserves the field's legacy + /// serialized-tail-budget meaning and forbids interpreting it under the + /// paced capability (no newest-tail selection exists until the + /// validated-baseline/screen-snapshot increment); the paced-start + /// observability event reports it (from the wire frame, in the ws + /// layer), and the increment-3 snapshot work consumes this record. + #[allow(dead_code)] + // the increment-3 snapshot work reads it; nothing may read it THIS increment + max_replay_bytes: Option, } /// One retained produced frame plus its persistent barrier classification (the ring's @@ -314,6 +667,16 @@ struct TerminalShared { create_request_id: Option, /// Attached connections, keyed by connection id (multi-client fan-out, `§7.3`). subscribers: HashMap, + /// Connections claiming this terminal's LIFETIME (responsive-terminal- + /// restore Workstream 1) without attaching — negotiated hidden panes. + /// Membership is per-connection and sweeps away with the socket + /// (`remove_connection`); a dropped claimer does NOT restore + /// `released_by_client` (transport loss is not release — identical to a + /// dropped subscriber). Only an explicit withdrawal + /// ([`TerminalRegistry::withdraw_claim`]) of the LAST claim (with no + /// subscribers) re-marks the row released, mirroring detach's + /// last-reference logic. + claims: BTreeSet, /// Whether the client EXPLICITLY released its last reference to this /// terminal (`terminal.detach` emptying `subscribers`) — as opposed to /// merely losing its socket (`remove_connection`: browser closed, laptop @@ -348,6 +711,17 @@ struct TerminalShared { } impl TerminalShared { + /// The earliest sequence position still available for replay (restore + /// contract, responsive-terminal-restore): the retained ring's front + /// `seqStart`, or `head_seq + 1` when the ring is empty (nothing older + /// than the head is retained). Caller holds the terminal lock. + fn oldest_retained_seq(&self) -> i64 { + self.replay + .front() + .map(|f| f.output.seq_start) + .unwrap_or(self.head_seq + 1) + } + /// `single_client` while at most one socket is attached; `multi_client_unknown` /// once a second attaches (`§5.3`, `broker.ts:394-395`). The client uses this to /// decide checkpoint/delta-replay validity, so it must reflect reality. @@ -722,6 +1096,10 @@ pub struct TerminalRegistry { /// Captured into each new terminal's `max_replay_chars` at [`Self::create`] /// time (TERM-13) -- see [`compute_scrollback_max_bytes`]. scrollback_max_bytes: Arc, + /// Responsive-terminal-restore Workstream 1: the serialized-byte budget + /// of ONE paced replay page ([`DEFAULT_PACED_PAGE_MAX_BYTES`]). Atomic so + /// focused tests shrink it per-instance without env races. + paced_page_max_bytes: Arc, /// TERM-15/TERM-16 activity tap (see [`ActivityEvent`]). Set once at boot /// by the activity hub; `None` (the default) keeps every fire point a /// cheap no-op. RwLock: read per event, written once. @@ -798,9 +1176,22 @@ pub struct TerminalRegistry { /// [`IdentityReadoptPauseHook`]). Never set in production. identity_readopt_pause: Arc>>, /// b8ke ext r13 F1: the create POST-CLAIM park seam (see - /// [`Self::set_terminal_create_postclaim_pause_for_tests`]). Never set - /// in production. + /// [`Self::set_terminal_create_postclaim_pause_for_tests`]). Never + /// set in production. terminal_create_postclaim_pause: Arc>>, + /// E2R3 (the atomic exit-transition decision) test-support: the + /// ONE-SHOT deterministic natural-exit staging hook. A hook site — + /// a staged-exit read on the transition-decision path — fires it + /// INSIDE its own terminal-lock hold, immediately AFTER its read + /// completes, and the registry then performs the natural-exit + /// STAGING for the armed (connection, code): the deterministic + /// model of the PTY reader staging its terminal's exit + /// CONCURRENTLY with the decision's use of its read (never a + /// sleep-based race). `None` (the default, never armed in + /// production) keeps every hook site a cheap no-op. Hosted here + /// (the `terminal_create_pause` idiom) so every cloned handle + /// observes a test-armed hook. + paced_exit_stage_hook: Arc>>, } /// The retained coordinator claim for one sessionRef-owning terminal (kata @@ -826,7 +1217,7 @@ impl Default for TerminalRegistry { /// `attach.ready` + replay were enqueued to the caller's sink) — `false` draws the /// reference's `INVALID_TERMINAL_ID` reply (attach to an unknown terminal; an /// exited-but-still-registered terminal is `found: true` + a synthetic exit). -#[derive(Debug, Clone, Copy, PartialEq, Eq)] +#[derive(Debug, Clone, PartialEq)] #[must_use] pub struct AttachOutcome { pub found: bool, @@ -834,6 +1225,12 @@ pub struct AttachOutcome { /// [`TerminalRegistry::attach_with_geometry`]. Plain [`TerminalRegistry::attach`] /// has no geometry input and returns `None`. pub geometry: Option, + /// A negotiated (pacedTerminalReplayV1 + attachRequestId) attach to a + /// RUNNING terminal: the paced session start — the first page's wire + /// messages plus the session description — INSTEAD of an inline replay + /// burst. `None` on every legacy path (non-negotiated, missing + /// attachRequestId, already-exited terminal). + pub paced: Option, } /// Outcome of [`TerminalRegistry::input`]: whether the terminal existed (the @@ -1067,6 +1464,7 @@ impl TerminalRegistry { auto_kill_idle_minutes: Arc::new(AtomicI64::new(DEFAULT_AUTO_KILL_IDLE_MINUTES)), stuck_window_ms: Arc::new(AtomicI64::new(DEFAULT_STUCK_WINDOW_MS)), scrollback_max_bytes: Arc::new(AtomicI64::new(DEFAULT_MAX_SCROLLBACK_CHARS)), + paced_page_max_bytes: Arc::new(AtomicI64::new(DEFAULT_PACED_PAGE_MAX_BYTES)), activity_observer: Arc::new(std::sync::RwLock::new(None)), respawn_liveness_window_ms: Arc::new(AtomicI64::new( DEFAULT_RESPAWN_LIVENESS_WINDOW_MS, @@ -1082,6 +1480,7 @@ impl TerminalRegistry { terminal_attach_pause: Arc::new(std::sync::RwLock::new(None)), identity_readopt_pause: Arc::new(std::sync::RwLock::new(None)), terminal_create_postclaim_pause: Arc::new(std::sync::RwLock::new(None)), + paced_exit_stage_hook: Arc::new(Mutex::new(None)), } } @@ -1304,6 +1703,170 @@ impl TerminalRegistry { self.scrollback_max_bytes.load(Ordering::Relaxed) } + /// Responsive-terminal-restore Workstream 1: update the serialized-byte + /// budget of one paced replay page. Applied to every page produced after + /// the call (the per-page read takes the terminal lock briefly, so a + /// live change is safe). Tests use small values for deterministic + /// multi-page fixtures. + pub fn set_paced_page_max_bytes(&self, max_bytes: i64) { + self.paced_page_max_bytes + .store(max_bytes, Ordering::Relaxed); + } + + /// The current paced-page serialized-byte budget + /// ([`DEFAULT_PACED_PAGE_MAX_BYTES`] unless configured). + pub fn paced_page_max_bytes(&self) -> i64 { + self.paced_page_max_bytes.load(Ordering::Relaxed) + } + + /// E2R1 finding 1: the staged natural-exit code for one paced + /// subscriber — `Some(code)` when the terminal exited naturally + /// while THIS subscriber's paced deferral was still armed + /// ([`Self::finish_pty_exit`] staged the exit instead of delivering + /// it). The STAGING is the authority for the ws layer's + /// exit-sequencing decisions: it happens under this terminal's lock + /// BEFORE the connection's notify hook fires, so a query anywhere + /// after that point observes it regardless of the notify's dispatch + /// order. `None` when the terminal is gone, the subscriber is gone, + /// or no exit is staged. + pub fn staged_paced_exit(&self, terminal_id: &str, conn_id: u64) -> Option { + let shared = { + let inner = self.inner.lock().expect("registry lock"); + inner + .terminals + .get(terminal_id) + .map(|handle| Arc::clone(&handle.shared))? + }; + let mut s = shared.lock().expect("terminal lock"); + let staged = s + .subscribers + .get(&conn_id) + .and_then(|sub| sub.paced_exit_pending); + // E2R3 test-support hook site: the one-shot staging fires AFTER + // this read, INSIDE the same lock hold — the deterministic model + // of the PTY reader staging its natural exit concurrently with + // the caller's use of the state this read just returned (the + // pre-fix arm's check-then-act window). Inert in production. + self.fire_paced_exit_stage_hook(&mut s, terminal_id); + staged + } + + /// E2R3 test-support: arm the ONE-SHOT natural-exit staging hook. + /// The next staged-exit read at a hook site performs the staging for + /// `conn_id` with `exit_code` INSIDE that read's terminal-lock hold, + /// immediately AFTER the read observes the pre-staging state — + /// deterministically modeling a PTY reader staging a natural exit + /// concurrently with the decision path's use of its read (the + /// transition-race tests' interleave; never a sleep-based race). The + /// hook fires once and clears itself; re-arming replaces a + /// not-yet-fired payload. + pub fn set_paced_exit_stage_hook_for_tests(&self, conn_id: u64, exit_code: i64) { + *self + .paced_exit_stage_hook + .lock() + .expect("paced exit stage hook") = Some(PacedExitStageHook { conn_id, exit_code }); + } + + /// E2R3 test-support: stage a natural exit for ONE paced subscriber + /// from OUTSIDE any terminal lock — the deterministic staging the + /// boundary test performs strictly AFTER an atomic transfer decision + /// (an exit staged past the decision is the drain's documented + /// uncredited tail content). Mirrors `finish_pty_exit`'s + /// subscriber-relevant subset: the terminal flips to naturally + /// `Exited` (monotone — a later real exit is inert) and the exit code + /// stages on the DEFERRED subscriber only. `false` when the terminal + /// or subscriber is gone, the subscriber is not deferred, or the + /// terminal already exited. + pub fn stage_natural_exit_for_test( + &self, + terminal_id: &str, + conn_id: u64, + exit_code: i64, + ) -> bool { + let Some(shared) = self.shared_for(terminal_id) else { + return false; + }; + let mut s = shared.lock().expect("terminal lock"); + stage_natural_exit_locked(&mut s, conn_id, exit_code) + } + + /// E2R3 test-support: fire the armed one-shot staging hook at a hook + /// site — INSIDE the caller's terminal-lock hold, immediately after + /// the site's staged-exit read. Inert unless a test armed the hook; + /// the staging is [`stage_natural_exit_locked`] (the real staging + /// path's subscriber-relevant subset — callable here because the + /// hook site already holds the lock). + fn fire_paced_exit_stage_hook(&self, s: &mut TerminalShared, terminal_id: &str) { + let Some(hook) = self + .paced_exit_stage_hook + .lock() + .expect("paced exit stage hook") + .take() + else { + return; + }; + if stage_natural_exit_locked(s, hook.conn_id, hook.exit_code) { + tracing::info!( + terminal_id = %terminal_id, + conn_id = hook.conn_id, + exit_code = hook.exit_code, + "terminal.paced_exit_staged_by_test_hook" + ); + } + } + + /// E2R3 (the atomic exit-transition decision): read one + /// subscriber's staged-exit state and the terminal's head under a + /// SINGLE terminal-lock hold and return the transition decision. + /// This is the ONE authority the ws layer's drive sites sequence + /// on — the credit path's pre-arm, the session start's arm, and + /// (load-bearing) the post-drive disposition that commits + /// extend-vs-transfer: because the staged-exit read and the + /// decision value leave the lock TOGETHER, no concurrently + /// staging PTY reader can invalidate the read before its caller + /// commits the disposition. The pre-fix arm read the staging and + /// the head in two separate holds and its disposition then decided + /// from that stale read after a drive — the check-then-act window + /// this closes structurally. + /// + /// A terminal/subscriber that is gone decides `exit_head: None` + /// (nothing staged); the caller's drive or drain discovers the + /// disappearance exactly as before. + pub fn paced_exit_transition(&self, terminal_id: &str, conn_id: u64) -> PacedExitTransition { + let Some(shared) = self.shared_for(terminal_id) else { + return PacedExitTransition { exit_head: None }; + }; + let mut s = shared.lock().expect("terminal lock"); + let exit_head = s + .subscribers + .get(&conn_id) + .and_then(|sub| sub.paced_exit_pending) + .map(|_| s.head_seq); + // E2R3 test-support hook site: the one-shot staging fires + // AFTER this read, INSIDE the same lock hold — the + // deterministic model of the PTY reader staging its natural + // exit concurrently with THIS decision's use of its read. The + // transition-race tests' interleave. Inert in production. + self.fire_paced_exit_stage_hook(&mut s, terminal_id); + PacedExitTransition { exit_head } + } + + /// The EFFECTIVE page budget for one paced attach (round-2 finding + /// F3): the negotiated `replayPageBytes` request honored as an + /// optional UPPER BOUND clamped to the registry's own cap — + /// `min(requested, cap)` when present and positive, the cap when + /// absent (missing/invalid requests already arrived as `None` via the + /// protocol layer's lossy deserializer; the `> 0` filter here also + /// guards direct embedders). One session keeps ONE budget: the caller + /// records this on [`PacedSessionDesc::page_budget`] so credits and + /// the tail drain page at the SAME bound that sized the first page. + pub fn effective_paced_page_budget(&self, paced: &PacedAttachOptions) -> i64 { + match paced.replay_page_bytes.filter(|requested| *requested > 0) { + Some(requested) => self.paced_page_max_bytes().min(requested), + None => self.paced_page_max_bytes(), + } + } + /// Reconciliation §7.5: shrink/grow the liveness window a generation must /// survive to reset the respawn counter (tests use small values). pub fn set_respawn_liveness_window_ms(&self, ms: i64) { @@ -1705,6 +2268,7 @@ impl TerminalRegistry { resume_session_id: resume_session_id.map(str::to_string), create_request_id: create_request_id.map(str::to_string), subscribers: HashMap::new(), + claims: BTreeSet::new(), released_by_client: true, name_ref: None, naming_handle: None, @@ -1827,7 +2391,7 @@ impl TerminalRegistry { /// `reconcileTerminalSessionAssociation`, a repair channel that was dead /// while this frame hardcoded `None`). /// - /// 9 arguments (`clippy::too_many_arguments`): every one is a distinct, + /// 10 arguments (`clippy::too_many_arguments`): every one is a distinct, /// non-optional attach input with exactly one call site outside tests /// (`freshell_ws::terminal::handle_attach`, which forwards the parsed /// `terminal.attach` frame fields 1:1) — a params struct would just @@ -1836,6 +2400,13 @@ impl TerminalRegistry { /// marker (mode replay-sync); when `Some(true)` and an /// `attach_request_id` is present, the tracker-synthesized mode preamble /// is emitted once, strictly between `attach.ready` and the replay. + /// `max_replay_bytes` is the attach's TERM-07 budget request — recorded + /// on the subscriber with no delivery-behavior change (see + /// [`Subscriber::max_replay_bytes`]). `paced` carries the negotiated + /// round-2 paced-attach inputs ([`PacedAttachOptions`]): the + /// `replayPageBytes` forward-page upper bound, honored as + /// `min(requested, registry cap)` on the paced path and recorded on + /// the session for every later page (round-2 finding F3). #[allow(clippy::too_many_arguments)] pub fn attach( &self, @@ -1845,8 +2416,11 @@ impl TerminalRegistry { attach_request_id: Option, since_seq: i64, terminal_output_batch_v1: bool, + paced_terminal_replay_v1: bool, session_ref: Option, surface_reset: Option, + max_replay_bytes: Option, + paced: PacedAttachOptions, ) -> AttachOutcome { // Take the terminal's shared Arc under the registry lock, then drop the // registry lock so we hold ONLY the per-terminal lock during the handoff. @@ -1858,6 +2432,7 @@ impl TerminalRegistry { return AttachOutcome { found: false, geometry: None, + paced: None, } } } @@ -1870,10 +2445,13 @@ impl TerminalRegistry { attach_request_id, since_seq, terminal_output_batch_v1, + paced_terminal_replay_v1, session_ref, surface_reset, + max_replay_bytes, shared, None, + paced, ) } @@ -1894,17 +2472,21 @@ impl TerminalRegistry { attach_request_id: Option, since_seq: i64, terminal_output_batch_v1: bool, + paced_terminal_replay_v1: bool, session_ref: Option, surface_reset: Option, + max_replay_bytes: Option, intent: TerminalAttachIntent, cols: u16, rows: u16, + paced: PacedAttachOptions, ) -> AttachOutcome { let inner = self.inner.lock().expect("registry lock"); let Some(handle) = inner.terminals.get(terminal_id) else { return AttachOutcome { found: false, geometry: Some(AttachResizeStatus::Missing), + paced: None, }; }; @@ -1915,10 +2497,13 @@ impl TerminalRegistry { attach_request_id, since_seq, terminal_output_batch_v1, + paced_terminal_replay_v1, session_ref, surface_reset, + max_replay_bytes, Arc::clone(&handle.shared), Some((intent, cols, rows, handle.pty.as_ref())), + paced, ) } @@ -1931,15 +2516,47 @@ impl TerminalRegistry { attach_request_id: Option, since_seq: i64, terminal_output_batch_v1: bool, + paced_terminal_replay_v1: bool, session_ref: Option, surface_reset: Option, + max_replay_bytes: Option, shared: Arc>, geometry: Option<(TerminalAttachIntent, u16, u16, Option<&PtyTerminal>)>, + paced: PacedAttachOptions, ) -> AttachOutcome { let mut s = shared.lock().expect("terminal lock"); let geometry = geometry.map(|(intent, cols, rows, pty)| { apply_attach_geometry(&mut s, intent, cols, rows, pty) }); + + // Responsive-terminal-restore Workstream 1: the paced path replaces + // the inline full-replay burst ONLY for negotiated connections — and + // only for a RUNNING terminal with an attachRequestId to correlate + // continuation credits (the already-Exited path keeps the frozen + // inline replay + synthetic exit, in their legacy order; a paced + // attach without an attachRequestId cannot be credited and falls + // back to the legacy inline replay, byte-identical to today). + let paced_path = paced_terminal_replay_v1 + && attach_request_id.is_some() + && s.status == TerminalRunStatus::Running; + + if paced_path { + return self.paced_attach_to_shared( + s, + terminal_id, + conn_id, + sink, + attach_request_id, + since_seq, + terminal_output_batch_v1, + session_ref, + surface_reset, + max_replay_bytes, + geometry, + paced, + ); + } + let effective_since = since_seq.max(0); // Snapshot the replay window: every retained frame newer than the client's @@ -1958,6 +2575,15 @@ impl TerminalRegistry { _ => (head_seq + 1, head_seq), }; + // Restore contract (responsive-terminal-restore): on NEGOTIATED + // attaches only, report the earliest sequence position still + // available for replay — the RING's front (not the attach's replay + // slice), or head+1 when nothing older than the head is retained. + // Captured under the same lock as the replay snapshot so the bound is + // consistent with it. Non-negotiated attaches leave it `None` (the + // frozen client's ready frame stays byte-identical). + let oldest_retained_seq = paced_terminal_replay_v1.then(|| s.oldest_retained_seq()); + // Register BEFORE enqueuing so any live frame the reader appends after we // release the lock is delivered strictly after this replay (the reader is // blocked on this same lock until we return). @@ -1967,6 +2593,12 @@ impl TerminalRegistry { sink: Arc::clone(&sink), attach_request_id: attach_request_id.clone(), terminal_output_batch_v1, + paced_terminal_replay_v1, + paced_deferred: false, + paced_handoff_boundary: None, + paced_exit_pending: None, + paced_exit_notify: None, + max_replay_bytes, }, ); // Somebody attached => this terminal is wanted. A later socket drop @@ -1985,6 +2617,7 @@ impl TerminalRegistry { effective_since_seq: Some(effective_since), geometry_authority: Some(s.geometry_authority()), geometry_epoch: Some(s.geometry_epoch), + oldest_retained_seq, replay_reset_reason: None, requested_since_seq: Some(since_seq), session_ref, @@ -2083,6 +2716,186 @@ impl TerminalRegistry { AttachOutcome { found: true, geometry, + paced: None, + } + } + + /// The PACED attach handoff (responsive-terminal-restore Workstream 1): + /// the negotiated Running-terminal replacement for the inline + /// full-replay burst. Under the same per-terminal lock as the legacy + /// path: apply geometry, resolve the retention-adjusted baseline, arm + /// the subscriber's deferral (the ring becomes the staging), sink the + /// ready/sync/retention-gap prelude, and SELECT the first page's frames + /// (never a full-ring clone). The first page travels back to the ws + /// caller — it is sunk only after the lock is released; the ws pacing + /// coordinator owns the session (credits, tail drain, completion). + /// + /// Retention loss at attach (the requested baseline predates the + /// retained ring): emit the negotiated `terminal.output.gap` with reason + /// `replay_window_exceeded` for the exact lost interval + /// `[effective+1, oldest-1]` plus the task-2 bounds fields, stamp the + /// ready frame's `replayResetReason: retention_lost`, and CONTINUE from + /// what is retained (baseline `oldest-1`) — nothing is killed, nothing + /// stalls; the client shows the honest incomplete-history state. + /// Non-negotiated attaches keep today's silent behavior exactly (see + /// the legacy branch above). + #[allow(clippy::too_many_arguments)] + fn paced_attach_to_shared( + &self, + mut s: std::sync::MutexGuard<'_, TerminalShared>, + terminal_id: &str, + conn_id: u64, + sink: FrameSink, + attach_request_id: Option, + since_seq: i64, + terminal_output_batch_v1: bool, + session_ref: Option, + surface_reset: Option, + max_replay_bytes: Option, + geometry: Option, + paced: PacedAttachOptions, + ) -> AttachOutcome { + let effective_requested = since_seq.max(0); + let head_seq = s.head_seq; + let oldest = s.oldest_retained_seq(); + + // Retention loss: the first position the client needs + // (`effective+1`) predates the retained ring. + let retention_lost = effective_requested + 1 < oldest; + let baseline = if retention_lost { + oldest - 1 + } else { + effective_requested + }; + let arid = attach_request_id + .clone() + .expect("the paced path requires an attachRequestId"); + + // Arm the deferral with the subscriber installation: from this + // point until the session completes, ingest does NOT fan out to + // this subscriber — the pages and the tail deliver its range in + // seq order (the ring is the staging). + s.subscribers.insert( + conn_id, + Subscriber { + sink: Arc::clone(&sink), + attach_request_id: Some(arid.clone()), + terminal_output_batch_v1, + paced_terminal_replay_v1: true, + paced_deferred: true, + paced_handoff_boundary: None, + paced_exit_pending: None, + paced_exit_notify: paced.paced_exit_notify.clone(), + max_replay_bytes, + }, + ); + s.released_by_client = false; + + // replayFrom/To describe the FULL window the session will deliver + // (the same first/last-span meaning as legacy, projected onto the + // paced range): `(baseline, head]`, or the empty span when the + // baseline already sits at the head. + let (replay_from, replay_to) = if baseline >= head_seq { + (head_seq + 1, head_seq) + } else { + (baseline + 1, head_seq) + }; + + let ready = ServerMessage::TerminalAttachReady(TerminalAttachReady { + head_seq, + replay_from_seq: replay_from, + replay_to_seq: replay_to, + stream_id: s.stream_id.clone(), + terminal_id: terminal_id.to_string(), + attach_request_id: Some(arid.clone()), + effective_since_seq: Some(baseline), + geometry_authority: Some(s.geometry_authority()), + geometry_epoch: Some(s.geometry_epoch), + oldest_retained_seq: Some(oldest), + replay_reset_reason: retention_lost.then_some(TerminalReplayResetReason::RetentionLost), + requested_since_seq: Some(since_seq), + session_ref, + }); + sink(ready); + + // The modes.sync preamble, byte-identical to the legacy block (a + // fresh surface still needs the emulator-mode prelude; the sync + // stays ahead of the pages by admission order — the first page is + // sunk only after this lock is released). + if surface_reset == Some(true) { + let data = s.modes.synthesize(); + if !data.is_empty() { + sink(ServerMessage::TerminalModesSync(TerminalModesSync { + terminal_id: terminal_id.to_string(), + attach_request_id: arid.clone(), + stream_id: s.stream_id.clone(), + data, + })); + } + } + + // The negotiated retention gap: ordered ahead of the pages by + // admission (sunk here under the lock; the pages are sunk after it). + if retention_lost { + sink(ServerMessage::TerminalOutputGap(TerminalOutputGap { + terminal_id: terminal_id.to_string(), + stream_id: s.stream_id.clone(), + attach_request_id: Some(arid.clone()), + from_seq: effective_requested + 1, + to_seq: oldest - 1, + reason: TerminalOutputGapReason::ReplayWindowExceeded, + head_seq: Some(head_seq), + oldest_retained_seq: Some(oldest), + })); + } + + // Select the first page (bounded by the session's EFFECTIVE page + // budget — round-2 finding F3: the negotiated `replayPageBytes` + // request clamped to the registry's cap), cloning ONLY the + // selected frames — never the whole ring. The SAME budget is + // recorded on the session so credits and the tail drain page at + // the requested bound too — the whole session honors the request, + // not just the first page. + let budget = self.effective_paced_page_budget(&paced); + let first_page = if baseline < head_seq { + paced_page_build( + &s, + conn_id, + baseline, + head_seq, + budget, + OutputSource::Replay, + ) + .map(|build| build.messages) + .unwrap_or_default() + } else { + Vec::new() + }; + let page_end = first_page + .last() + .and_then(page_last_seq) + .unwrap_or(baseline.max(head_seq)); + let page_bytes = first_page + .iter() + .map(|m| serde_json::to_string(m).map(|j| j.len()).unwrap_or(0)) + .sum::() as u64; + + AttachOutcome { + found: true, + geometry, + paced: Some(PacedAttachStart { + session: PacedSessionDesc { + terminal_id: terminal_id.to_string(), + stream_id: s.stream_id.clone(), + attach_request_id: arid, + target: head_seq, + effective_since: baseline, + page_end, + page_budget: budget, + page_bytes, + }, + first_page, + }), } } @@ -2101,6 +2914,12 @@ impl TerminalRegistry { let mut s = shared.lock().expect("terminal lock"); if s.subscribers.remove(&conn_id).is_some() && s.subscribers.is_empty() + // A connection may still CLAIM this terminal's lifetime + // (hidden pane, responsive-terminal-restore WS1): an explicit + // detach releases only when the last reference of EVERY kind + // is gone. Without claims this is always true, so the + // pre-claim detach semantics are byte-identical. + && s.claims.is_empty() && s.status == TerminalRunStatus::Running { // DEV-0009: a freshly-detached terminal gets a full idle @@ -2123,54 +2942,677 @@ impl TerminalRegistry { } } - /// On socket close: sweep `conn_id` out of EVERY terminal's subscriber set. All - /// PTYs keep running (background sessions), reattachable by a future socket. - pub fn remove_connection(&self, conn_id: u64) { - self.active_connections.fetch_sub(1, Ordering::Relaxed); - let shareds: Vec>> = { + /// Hidden-pane lifetime claim (responsive-terminal-restore Workstream 1): + /// mark `terminal_id` wanted for `conn_id` WITHOUT attaching. Clears + /// `released_by_client` under the terminal lock (if currently true) and + /// records the claim per-connection. Never adds a subscriber, never + /// grants replay or output delivery, never touches geometry or stream + /// identity. Unknown ids are a no-op (`false`) — stale client layouts may + /// claim rows that died server-side. Claims coexist with subscriptions: + /// an attach to a claimed terminal behaves exactly as any attach, and + /// the claim stays recorded until a later interest snapshot supersedes it + /// or the socket sweeps it. + pub fn claim_terminal(&self, terminal_id: &str, conn_id: u64) -> bool { + let Some(shared) = self.shared_for(terminal_id) else { + return false; + }; + let mut s = shared.lock().expect("terminal lock"); + s.claims.insert(conn_id); + // Wanted, exactly as a successful attach would mark it. + s.released_by_client = false; + true + } + + /// Explicit withdrawal of a hidden-pane lifetime claim (a later interest + /// snapshot no longer lists the id). Mirrors detach's last-reference + /// release logic: when the LAST claim goes away and no subscribers + /// remain, the terminal is genuinely orphaned — restore + /// `released_by_client` with the same DEV-0009 fresh-idle-grace bump + /// detach grants. A withdrawal that leaves other claimers (or any + /// subscriber) keeps the terminal wanted. + pub fn withdraw_claim(&self, terminal_id: &str, conn_id: u64) { + let Some(shared) = self.shared_for(terminal_id) else { + return; + }; + let mut s = shared.lock().expect("terminal lock"); + if s.claims.remove(&conn_id) + && s.claims.is_empty() + && s.subscribers.is_empty() + && s.status == TerminalRunStatus::Running + { + // DEV-0009: the release transition grants one full idle threshold + // of grace (same rationale as detach). + s.last_meaningful_activity_at = s.last_meaningful_activity_at.max(now_ms()); + s.released_by_client = true; + } + } + + /// Lifetime-claim observability (responsive-terminal-restore Workstream + /// 1): how many distinct connections currently claim `terminal_id`, and + /// whether the row is fast-reap eligible (`released_by_client`). `None` + /// when the terminal does not exist. Test/diagnostic seam — no production + /// decision path reads this. + pub fn claim_state(&self, terminal_id: &str) -> Option { + let shared = self.shared_for(terminal_id)?; + let s = shared.lock().expect("terminal lock"); + Some(ClaimState { + claimers: s.claims.len(), + released_by_client: s.released_by_client, + }) + } + + /// Restore-contract sequence bounds for one terminal's retained replay + /// ring (responsive-terminal-restore): current `head_seq` plus the + /// earliest sequence position still available for replay. Takes the + /// per-terminal lock briefly and copies NO payloads. `None` when the + /// terminal does not exist. + /// + /// Callers must NOT hold the calling connection's writer admission lock: + /// the terminal side (subscriber fan-out, attach replay) acquires that + /// lock while holding THIS per-terminal lock, so resolving bounds under + /// the admission lock would invert the established lock order. + pub fn replay_bounds(&self, terminal_id: &str) -> Option { + let shared = { let inner = self.inner.lock().expect("registry lock"); inner .terminals - .values() + .get(terminal_id) .map(|h| Arc::clone(&h.shared)) - .collect() + }?; + let s = shared.lock().expect("terminal lock"); + Some(ReplayBounds { + head_seq: s.head_seq, + oldest_retained_seq: s.oldest_retained_seq(), + }) + } + + /// Paced replay page read (responsive-terminal-restore Workstream 1): + /// ONE bounded, ascending page of the `(from_seq, target]` window for a + /// deferred subscriber's session, taking the per-terminal lock briefly + /// (never across pages). The page is packed until the serialized budget + /// (envelope + escaping + batch metadata, via the batch builder's + /// accounting) is reached; a single frame whose own envelope exceeds the + /// budget forms its own atomic single-frame page. Frames are selected + /// BEFORE cloning — no full-ring snapshot per page. + /// + /// `Done` when `from_seq >= target`; `Expired` when retention evicted the + /// next needed frames (exact lost interval `[from_seq+1, new_front-1]`, + /// `resume_from = new_front-1`); `Gone` when the terminal or the + /// subscriber disappeared (cancel the session). + /// + /// Callers must NOT hold the calling connection's writer admission lock + /// (same lock-order rule as [`Self::replay_bounds`]). + pub fn next_replay_page( + &self, + terminal_id: &str, + conn_id: u64, + from_seq: i64, + target: i64, + max_serialized_bytes: i64, + ) -> PacedPage { + let Some(shared) = self.shared_for(terminal_id) else { + return PacedPage::Gone; }; - for shared in shareds { - let mut s = shared.lock().expect("terminal lock"); - if s.subscribers.remove(&conn_id).is_some() - && s.subscribers.is_empty() - && s.status == TerminalRunStatus::Running - { - // DEV-0009: a freshly-detached terminal gets a full idle - // threshold of grace — its meaningful clock may have expired - // while a watcher was attached (attached => reaper-exempt). - // The `.is_some()` gate is essential here: this sweep visits - // EVERY terminal, and an unconditional bump would reset the - // countdown of unrelated, already-detached terminals on - // every socket close. As in `detach`, the wedge-backstop - // output clocks (`last_output_activity_at`, - // `last_meaningful_output_at`) are deliberately NOT - // bumped: a page refresh (socket drop + re-attach) must - // not reset stuck detection. - s.last_meaningful_activity_at = s.last_meaningful_activity_at.max(now_ms()); + let s = shared.lock().expect("terminal lock"); + let Some(sub) = s.subscribers.get(&conn_id) else { + return PacedPage::Gone; + }; + if !sub.paced_terminal_replay_v1 { + return PacedPage::Gone; + } + if from_seq >= target { + return PacedPage::Done; + } + let oldest = s.oldest_retained_seq(); + if from_seq + 1 < oldest { + return PacedPage::Expired { + lost_from: from_seq + 1, + lost_to: oldest - 1, + resume_from: oldest - 1, + head_seq: s.head_seq, + oldest_retained_seq: oldest, + }; + } + match paced_page_build( + &s, + conn_id, + from_seq, + target, + max_serialized_bytes, + OutputSource::Replay, + ) { + Some(build) => PacedPage::Frames { + messages: build.messages, + end_seq: build.end_seq, + serialized_bytes: build.serialized_bytes, + }, + // Defensive: the window is non-empty and retained (checked + // above), so an empty build means nothing the walk could select — + // treat the window as delivered rather than stalling the session. + None => PacedPage::Done, + } + } + + /// The paced drain's PAGED, FIXED-TARGET completion + /// (responsive-terminal-restore W1): the caller (the connection's + /// off-dispatch drain task) captures the drain target ONCE — the head + /// when the drain phase starts — and this method pages the staged + /// range `(from_seq, min(head, to_seq_inclusive)]` toward that FIXED + /// target, ONE budget-bounded page per call, the lock held only per + /// page (never across pages, so PTY ingestion interleaves and no + /// full-suffix clone is ever built under a single hold). The caller + /// loops until a completing verdict: + /// + /// - `CaughtUp` — the ring was drained past `from_seq`: the ATOMIC + /// deferral clear under this lock hold; live output resumes direct + /// fan-out. + /// - `Completed` — this call's page covered the terminal's CURRENT + /// head (everything staged was delivered in this hold): the + /// ATOMIC deferral clear in the SAME lock hold. Live output + /// resumes direct fan-out. + /// - `TargetCovered` — the page covered the FIXED target while the + /// producer staged frames BEYOND it. NO bulk re-fan, NO clear: the + /// deferral STAYS armed and the staged post-target remainder is + /// handed to [`Self::handoff_paced_tail`]. This verdict is + /// COMPLETION START (round-4): the head read under THIS hold is the + /// completion boundary B, captured ONCE and recorded on the + /// subscriber — the handoff pages up to B and ONLY B (never the + /// moving current head), so a producer appending at-or-above drain + /// speed can never postpone completion. NO RECAPTURE, ever: a + /// target covered is a target handed off. + /// - `Handoff` — one budget-bounded page delivered; the deferral + /// STAYS armed (ordering with live frames is preserved by the + /// deferral itself: ingest stages without fanning out, so live + /// output can never overtake an un-sent page); advance the cursor + /// and call again. + /// - `Expired` — retention advanced past `from_seq`: the exact + /// bounds-carrying interval for the omitted range; the caller emits + /// the retention gap and continues from the resumed front — never + /// a silent forward jump to the ring front. + /// - `Gone` — the terminal, the subscriber, or the SESSION'S ATTACH + /// GENERATION is gone. The generation guard is load-bearing for + /// the OFF-DISPATCH drain: the drain task runs concurrently with + /// the connection dispatcher, so a re-attach mid-drain replaces + /// the subscriber's attach generation while the old drain is + /// still paging — an unguarded read would page with the NEW + /// generation's stamp and its "drained"/completing verdicts would + /// clear the NEW session's deferral, letting live output overtake + /// the new session's pages. The superseded drain cancels without + /// touching anything. + /// + /// Each page read's own budget bounds the per-iteration sink work; + /// there is NO completing-call bulk re-fan (the round-3 finding) — + /// every byte the completion admits goes through [`paced_page_build`]'s + /// budget — and the caller's backpressure gate (the drain task awaits + /// the connection queue's real capacity between pages) bounds the + /// loop's admission against the connection queue. + /// + /// Callers must NOT hold the calling connection's writer admission + /// lock (same lock-order rule as [`Self::replay_bounds`]). + pub fn complete_paced_tail( + &self, + terminal_id: &str, + conn_id: u64, + expected_attach_request_id: &str, + from_seq: i64, + to_seq_inclusive: i64, + max_serialized_bytes: i64, + ) -> PacedTailCompletion { + let Some(shared) = self.shared_for(terminal_id) else { + return PacedTailCompletion::Gone; + }; + let mut s = shared.lock().expect("terminal lock"); + let sub = match s.subscribers.get(&conn_id) { + Some(sub) => sub, + None => return PacedTailCompletion::Gone, + }; + if !sub.paced_terminal_replay_v1 { + return PacedTailCompletion::Gone; + } + // The superseded-generation guard (see the doc comment): only the + // session that OWNS this subscriber's current attach generation + // may page or clear. + if sub.attach_request_id.as_deref() != Some(expected_attach_request_id) { + return PacedTailCompletion::Gone; + } + let head_seq = s.head_seq; + if from_seq >= head_seq { + // THE ATOMIC CLEAR: nothing is staged beyond `from_seq`, and the + // last page was sunk before this call — no un-sent page exists, + // so direct fan-out from here on can never overtake a page. + s.subscribers + .get_mut(&conn_id) + .expect("subscriber checked above") + .paced_deferred = false; + // Finding F1: a staged natural exit rides the completing hold — + // ordered after everything already sunk, then the subscriber + // retires. + deliver_staged_paced_exit(&mut s, conn_id); + return PacedTailCompletion::CaughtUp; + } + let oldest = s.oldest_retained_seq(); + if from_seq + 1 < oldest { + // Retention advanced past the drain cursor mid-drain: report the + // EXACT lost interval (never silently start at the ring front) + // and let the caller resume from the new front. + return PacedTailCompletion::Expired { + lost_from: from_seq + 1, + lost_to: oldest - 1, + resume_from: oldest - 1, + head_seq, + oldest_retained_seq: oldest, + }; + } + let page_target = head_seq.min(to_seq_inclusive); + let built = paced_page_build( + &s, + conn_id, + from_seq, + page_target, + max_serialized_bytes, + OutputSource::Live, + ); + let Some(build) = built else { + // Defensive: the window is non-empty and retained (checked + // above), so an empty build means nothing the walk could + // select — treat the window as delivered rather than stalling + // the session. If anything remains staged beyond the fixed target, + // the post-target handoff owns it (with the boundary captured + // ONCE under this hold, exactly like the covering-page arm); + // otherwise clear now. + if to_seq_inclusive >= head_seq { + s.subscribers + .get_mut(&conn_id) + .expect("subscriber checked above") + .paced_deferred = false; + deliver_staged_paced_exit(&mut s, conn_id); + return PacedTailCompletion::Completed { + end_seq: from_seq, + serialized_bytes: 0, + }; } + s.subscribers + .get_mut(&conn_id) + .expect("subscriber checked above") + .paced_handoff_boundary = Some(head_seq); + return PacedTailCompletion::TargetCovered { + end_seq: from_seq, + serialized_bytes: 0, + }; + }; + let sink = Arc::clone(&s.subscribers.get(&conn_id).expect("checked above").sink); + for message in build.messages { + sink(message); + } + if build.end_seq >= head_seq { + // THE CLEARED-AT-PAGE COMPLETION: this page covered the head + // read under THIS hold, so everything staged ≤ the head was + // delivered in this hold. Clear the deferral in the same hold: + // ingest is blocked on the lock, so nothing can be fanned out + // between the page's sink and the clear; everything appended + // later fans out at its own ingest, strictly after the page in + // queue order. The boundary can neither lose nor duplicate a + // frame. + s.subscribers + .get_mut(&conn_id) + .expect("subscriber checked above") + .paced_deferred = false; + // Finding F1: the staged exit (if the terminal exited while + // the deferral was armed) rides this completing hold — after + // the final pages, exactly once. + deliver_staged_paced_exit(&mut s, conn_id); + return PacedTailCompletion::Completed { + end_seq: build.end_seq, + serialized_bytes: build.serialized_bytes, + }; + } + if build.end_seq >= to_seq_inclusive { + // THE FIXED-TARGET HANDOFF — COMPLETION START (round-4): the + // page covered the FIXED target while the producer staged + // frames beyond it. NO bulk re-fan (round-3 finding: a + // retained-suffix clone under this hold admits up to a ring + // outside every page budget) and NO clear (the staged + // remainder must be delivered before live output may overtake + // it). The deferral STAYS armed — ingest keeps staging in seq + // order — and the caller delivers the staged post-target + // remainder through [`Self::handoff_paced_tail`]. THE + // COMPLETION BOUNDARY B is the head captured ONCE under THIS + // hold (the one sanctioned head read, at completion start — + // recorded on the subscriber; the handoff re-reads NOTHING): + // the handoff pages up to B and ONLY B, so a producer + // appending at-or-above drain speed can never move the + // completion boundary (the moving-head chase is structurally + // gone). + let sub = s + .subscribers + .get_mut(&conn_id) + .expect("subscriber checked above"); + sub.paced_handoff_boundary = Some(head_seq); + return PacedTailCompletion::TargetCovered { + end_seq: build.end_seq, + serialized_bytes: build.serialized_bytes, + }; + } + // More remains staged below the fixed target: the deferral stays + // armed, the caller pages on. + PacedTailCompletion::Handoff { + end_seq: build.end_seq, + serialized_bytes: build.serialized_bytes, } } - /// `terminal.input` write path (`terminal-registry.ts:3867-3894`): write bytes to - /// the PTY; bump `lastActivityAt` and the DEV-0009 meaningful-activity reap clock. - /// Unknown terminal => `InputOutcome { found: false }` (kata dtfn: previously a - /// silent no-op; the caller now replies on the wire). - pub fn input(&self, terminal_id: &str, data: &[u8]) -> InputOutcome { - let (found, tapped_mode) = { - let mut inner = self.inner.lock().expect("registry lock"); - match inner.terminals.get_mut(terminal_id) { - Some(handle) => { - if let Some(pty) = handle.pty.as_mut() { - let _ = pty.write_input(data); - } else { - // Headless rows have no PTY to receive bytes. Unreachable in - // production today (`register_headless` has no production + /// The BOUNDED post-target handoff (responsive-terminal-restore W1, + /// round-4 fix): after [`Self::complete_paced_tail`] covers the drain's + /// FIXED target, the frames the producer staged BEYOND it are delivered + /// here, ONE budget-bounded chunk per call, through the same normal + /// live page projection the drain's own pages use ([`paced_page_build`], + /// source `live`). THE COMPLETION BOUNDARY IS FIXED: B is the head + /// captured ONCE at completion start (the `TargetCovered` hold, + /// recorded on the subscriber as `paced_handoff_boundary`) and this + /// method pages up to B and ONLY B — it contains NO read of the + /// terminal's current head anywhere (the recurring moving-head-chase + /// cause is structurally removed), so a producer appending at-or-above + /// drain speed can never move the completion boundary or postpone + /// completion past the B-derived page bound. The caller (the + /// connection's off-dispatch drain task) loops under its real + /// backpressure gate: the lock is held only per chunk, the connection + /// queue's actual consumption bounds the loop's admission, and each + /// hold admits at most one page budget — never a bulk retained-suffix + /// clone. + /// + /// - `Handoff` — one chunk of `(from_seq, B]` delivered; the deferral + /// STAYS armed; advance the cursor and call again. + /// - `Completed` — THE COMPLETING HOLD: this chunk covered B. The + /// frames the producer staged past B (its ingests already passed + /// while the deferral held them back) flow through the NORMAL LIVE + /// FAN-OUT PATH in this same hold (the completing sweep — the same + /// per-frame projection [`ingest`] uses, never a page build, never a + /// retained-suffix clone), and the deferral was CLEARED in the same + /// hold. The hold is the atomic boundary: everything staged up to it + /// was delivered in it, post-hold ingests fan out directly, and the + /// boundary can neither lose nor duplicate a frame. Ends the paced + /// session. + /// - `GapCompleted` — the plan:146 bounded-baseline exit: retention + /// overran the handoff cursor mid-handoff. The EXACT bounds-carrying + /// gap for the evicted interval was sunk through the subscriber's + /// sink FIRST, then the retained window (the ring front through the + /// head) flowed through the same normal live path in this hold, and + /// the deferral cleared — the session COMPLETES AT THE RING FRONT + /// with the gap recorded; the paged handoff never resumes toward the + /// (unreachable) B. Ends the paced session. + /// - `CaughtUp` — nothing is staged within the fixed boundary beyond + /// the cursor: the same atomic clear, nothing delivered this call. + /// - `Gone` — the terminal, the subscriber, the deferral (another + /// path already cleared it), the completion boundary (no + /// `TargetCovered` hold ever armed one — a protocol violation), or + /// the session's attach generation is gone; cancel. + /// + /// Callers must NOT hold the calling connection's writer admission + /// lock (same lock-order rule as [`Self::replay_bounds`]). + pub fn handoff_paced_tail( + &self, + terminal_id: &str, + conn_id: u64, + expected_attach_request_id: &str, + from_seq: i64, + max_serialized_bytes: i64, + ) -> PacedTailCompletion { + let Some(shared) = self.shared_for(terminal_id) else { + return PacedTailCompletion::Gone; + }; + let mut s = shared.lock().expect("terminal lock"); + let sub = match s.subscribers.get(&conn_id) { + Some(sub) => sub, + None => return PacedTailCompletion::Gone, + }; + if !sub.paced_terminal_replay_v1 { + return PacedTailCompletion::Gone; + } + // The deferral must still be armed: a cleared deferral means live + // fan-out already resumed (ingest delivers directly) — delivering + // here would duplicate. + if !sub.paced_deferred { + return PacedTailCompletion::Gone; + } + if sub.attach_request_id.as_deref() != Some(expected_attach_request_id) { + return PacedTailCompletion::Gone; + } + // THE FIXED COMPLETION BOUNDARY, captured ONCE at completion start + // (the TargetCovered hold). This method NEVER reads the terminal's + // current head: the boundary is subscriber state, so no per-call + // re-capture path exists at all. + let Some(boundary) = sub.paced_handoff_boundary else { + // Protocol violation: the handoff phase only begins after the + // TargetCovered verdict armed the boundary. Refuse rather than + // guessing a target (the guessing is the convicted chase). + return PacedTailCompletion::Gone; + }; + if from_seq >= boundary { + // Nothing is staged within the fixed boundary beyond the + // cursor (defensive: the completing chunk covers B before the + // cursor can reach it). The same completing-hold rule applies + // past the boundary: a residual is declared, never swept. + let back = s + .replay + .back() + .map(|f| (f.output.seq_start, f.output.seq_end)) + .unwrap_or((boundary, boundary)); + if back.0 > boundary { + return complete_at_fixed_boundary(&mut s, conn_id, boundary, back, from_seq, 0); + } + let sub = s + .subscribers + .get_mut(&conn_id) + .expect("subscriber checked above"); + sub.paced_deferred = false; + sub.paced_handoff_boundary = None; + // Finding F1: the staged exit rides this completing hold too. + deliver_staged_paced_exit(&mut s, conn_id); + return PacedTailCompletion::CaughtUp; + } + let oldest = s.oldest_retained_seq(); + if from_seq + 1 < oldest { + // THE plan:146 BOUNDED-BASELINE EXIT: retention overran the + // handoff cursor mid-handoff. TWO exact bounds-carrying gaps are + // sunk THROUGH THE SUBSCRIBER'S SINK in THIS hold — ordered + // ahead of everything else: (1) the retention gap for the + // EVICTED interval (from_seq+1, oldest-1] — unfetchable, the + // honest retention-loss UX; (2) the delivery gap for the + // RETAINED window the session will not deliver (oldest, back] + // — fetchable, the client's checkpoint-cursor repair fetches it + // as a fresh bounded paced session. Then the deferral clears + // and the session COMPLETES AT THE RING FRONT with both gaps + // recorded: the paged handoff does NOT resume toward B (a + // finite retention window cannot guarantee convergence against + // indefinitely faster output production), and nothing is swept + // in this hold (plan:145: no pinned replay backlog, no + // admission outside the page budget). (The gap frames' + // headSeq bounds field is the retained ring's back seq_end — + // an observability value read from the ring like `oldest` + // itself, never a completion target.) + let sink = Arc::clone(&s.subscribers.get(&conn_id).expect("checked above").sink); + let arid = s + .subscribers + .get(&conn_id) + .expect("checked above") + .attach_request_id + .clone(); + let back_seq = s + .replay + .back() + .map(|f| f.output.seq_end) + .unwrap_or(from_seq); + sink(ServerMessage::TerminalOutputGap(TerminalOutputGap { + terminal_id: terminal_id.to_string(), + stream_id: s.stream_id.clone(), + attach_request_id: arid.clone(), + from_seq: from_seq + 1, + to_seq: oldest - 1, + reason: TerminalOutputGapReason::ReplayWindowExceeded, + head_seq: Some(back_seq), + oldest_retained_seq: Some(oldest), + })); + if back_seq >= oldest { + sink(ServerMessage::TerminalOutputGap(TerminalOutputGap { + terminal_id: terminal_id.to_string(), + stream_id: s.stream_id.clone(), + attach_request_id: arid, + from_seq: oldest, + to_seq: back_seq, + reason: TerminalOutputGapReason::HandoffBoundaryReached, + head_seq: Some(back_seq), + oldest_retained_seq: Some(oldest), + })); + } + let sub = s + .subscribers + .get_mut(&conn_id) + .expect("subscriber checked above"); + sub.paced_deferred = false; + sub.paced_handoff_boundary = None; + // Finding F1: the retention-overrun bounded-baseline exit still + // sequences a staged exit AFTER the exact gaps — the gap, then + // exit, never exit-first. + deliver_staged_paced_exit(&mut s, conn_id); + return PacedTailCompletion::GapCompleted { + lost_from: from_seq + 1, + lost_to: oldest - 1, + end_seq: from_seq, + serialized_bytes: 0, + // THE plan:146 bounded-baseline exit: a GENUINE retention + // overrun (round-5 finding 3's diagnostics discriminator). + reason: PacedGapExitReason::RetentionOverrun, + }; + } + // Everything past `boundary` that the producer staged while the + // deferral held live admission back — the ring's retained tail past + // the boundary, WITHOUT reading the terminal's current head: the + // back frame's seq bounds, read from the ring exactly like + // `oldest_retained_seq` reads the front. + let back = s + .replay + .back() + .map(|f| (f.output.seq_start, f.output.seq_end)) + .unwrap_or((boundary, boundary)); + let built = paced_page_build( + &s, + conn_id, + from_seq, + boundary, + max_serialized_bytes, + OutputSource::Live, + ); + match built { + Some(build) => { + let sink = Arc::clone(&s.subscribers.get(&conn_id).expect("checked above").sink); + for message in build.messages { + sink(message); + } + if build.end_seq >= boundary { + // THE COMPLETING HOLD: the chunk covered the FIXED + // boundary B — the session's own window is fully + // delivered, and the deferral clears ATOMICALLY in + // this same hold. The frames the producer staged past + // B are NOT swept here (plan:145: a one-hold + // ring-sized admission is a pinned replay backlog + // outside every bound): the session COMPLETES AT B, + // and when a residual exists its EXACT interval is + // declared as the bounds-carrying + // `handoff_boundary_reached` delivery gap — the + // client's bounded baseline recovery (the + // checkpoint-cursor repair, the queue_overflow + // contract) fetches it as a fresh bounded paced + // session. A quiet terminal (no residual) completes + // cleanly: post-hold ingests fan out directly at + // their own ingest, so the boundary can neither lose + // nor duplicate a frame. + return complete_at_fixed_boundary( + &mut s, + conn_id, + boundary, + back, + build.end_seq, + build.serialized_bytes, + ); + } + PacedTailCompletion::Handoff { + end_seq: build.end_seq, + serialized_bytes: build.serialized_bytes, + } + } + // Defensive: the window is non-empty and retained (checked + // above), so an empty build means nothing the walk could + // select — treat the window as delivered and take the same + // completing-hold path rather than stalling the session. + None => complete_at_fixed_boundary(&mut s, conn_id, boundary, back, from_seq, 0), + } + } + + /// Resolve a terminal's shared handle under the registry lock, then drop + /// the registry lock (the page reads hold ONLY the per-terminal lock). + fn shared_for(&self, terminal_id: &str) -> Option>> { + let inner = self.inner.lock().expect("registry lock"); + inner + .terminals + .get(terminal_id) + .map(|h| Arc::clone(&h.shared)) + } + + /// On socket close: sweep `conn_id` out of EVERY terminal's subscriber set + /// and lifetime-claim set. All PTYs keep running (background sessions), + /// reattachable by a future socket. Transport loss is NOT release: the + /// sweep never restores `released_by_client` — a terminal left with no + /// subscribers and no claims by a socket drop stays wanted (24-hour hard + /// cap only), exactly like an attached-then-disconnected one. + pub fn remove_connection(&self, conn_id: u64) { + self.active_connections.fetch_sub(1, Ordering::Relaxed); + let shareds: Vec>> = { + let inner = self.inner.lock().expect("registry lock"); + inner + .terminals + .values() + .map(|h| Arc::clone(&h.shared)) + .collect() + }; + for shared in shareds { + let mut s = shared.lock().expect("terminal lock"); + let removed_subscriber = s.subscribers.remove(&conn_id).is_some(); + let removed_claim = s.claims.remove(&conn_id); + if (removed_subscriber || removed_claim) + && s.subscribers.is_empty() + && s.claims.is_empty() + && s.status == TerminalRunStatus::Running + { + // DEV-0009: a freshly-detached terminal gets a full idle + // threshold of grace — its meaningful clock may have expired + // while a watcher was attached (attached => reaper-exempt). + // The removal gate is essential here: this sweep visits + // EVERY terminal, and an unconditional bump would reset the + // countdown of unrelated, already-detached terminals on + // every socket close. As in `detach`, the wedge-backstop + // output clocks (`last_output_activity_at`, + // `last_meaningful_output_at`) are deliberately NOT + // bumped: a page refresh (socket drop + re-attach) must + // not reset stuck detection. + s.last_meaningful_activity_at = s.last_meaningful_activity_at.max(now_ms()); + } + } + } + + /// `terminal.input` write path (`terminal-registry.ts:3867-3894`): write bytes to + /// the PTY; bump `lastActivityAt` and the DEV-0009 meaningful-activity reap clock. + /// Unknown terminal => `InputOutcome { found: false }` (kata dtfn: previously a + /// silent no-op; the caller now replies on the wire). + pub fn input(&self, terminal_id: &str, data: &[u8]) -> InputOutcome { + let (found, tapped_mode) = { + let mut inner = self.inner.lock().expect("registry lock"); + match inner.terminals.get_mut(terminal_id) { + Some(handle) => { + if let Some(pty) = handle.pty.as_mut() { + let _ = pty.write_input(data); + } else { + // Headless rows have no PTY to receive bytes. Unreachable in + // production today (`register_headless` has no production // callers — ledger A16), but never let input vanish without // a trace (kata dtfn). tracing::warn!(terminal_id, "input_to_headless_terminal_dropped"); @@ -2480,11 +3922,44 @@ impl TerminalRegistry { exit_code, terminal_id: terminal_id.to_string(), }); - for sub in s.subscribers.values() { - (sub.sink)(exit.clone()); + // Round-2 finding F1 — the exit fan-out is PARTITIONED. A + // subscriber whose paced session is still DEFERRED (its final + // output is staged in the ring, pages undelivered) must NOT + // receive terminal.exit now: exit-first strands the deferred + // range (the client's exit handler clears the attach and rejects + // late frames). Stage the exit on the subscriber instead — the + // session's drain drives the deferred range to completion and + // the completing verdict delivers the staged exit in the SAME + // lock hold, ordered after the final output through the same + // sink (plan:190 sequenced exit delivery). The connection's + // notify hook (collected here, fired OUTSIDE the lock below) + // lets the ws layer move a still-CREDITED session into that + // drain. Every OTHER subscriber keeps the frozen behavior: + // exit now, subscription retired. + let mut staged_notify: Vec = Vec::new(); + let mut retire_now: Vec = Vec::new(); + for (conn_id, sub) in s.subscribers.iter_mut() { + if sub.paced_deferred { + sub.paced_exit_pending = Some(exit_code); + if let Some(notify) = sub.paced_exit_notify.clone() { + staged_notify.push(notify); + } + } else { + (sub.sink)(exit.clone()); + retire_now.push(*conn_id); + } + } + for conn_id in retire_now { + s.subscribers.remove(&conn_id); } - s.subscribers.clear(); drop(s); + // Fire the staged-exit notifications with NO lock held: the hook + // only enqueues on the owning connection's channel and must never + // re-enter this terminal's lock (it runs on the PTY reader + // thread). + for notify in staged_notify { + notify(terminal_id, exit_code); + } // Reconciliation §7.5: a generation that died inside the liveness // window counts toward the respawn cap; one that survived it resets // the counter (a healthy resume is not penalized). Natural exits only @@ -2851,6 +4326,7 @@ impl TerminalRegistry { resume_session_id: opts.resume_session_id, create_request_id, subscribers: HashMap::new(), + claims: BTreeSet::new(), released_by_client: true, name_ref: None, naming_handle: None, @@ -4000,6 +5476,13 @@ fn ingest(shared: &Arc>, msg: ServerMessage) { // (source stays 'live'). A single live frame is one small batch — the merge logic // is the same as replay's (proven byte-exact by the deterministic crate goldens). for sub in s.subscribers.values() { + // Restore contract (responsive-terminal-restore): a subscriber with a + // paced session in flight receives NOTHING inline — the retained + // ring is the staging and the session's pages deliver its range in + // seq order (see `Subscriber::paced_deferred`). + if sub.paced_deferred { + continue; + } match ( sub.terminal_output_batch_v1, sub.attach_request_id.as_deref(), @@ -4046,9 +5529,9 @@ fn deliver_batches( frames: &[RetainedFrame], attach_request_id: &str, source: &str, -) { +) -> u64 { if frames.is_empty() { - return; + return 0; } let batch_max = terminal_stream_batch_max_bytes() as i64; let inputs: Vec = frames.iter().map(|f| f.to_batch_input()).collect(); @@ -4060,6 +5543,7 @@ fn deliver_batches( attach_request_id: Some(attach_request_id.to_string()), source: Some(source.to_string()), }); + let mut serialized_bytes = 0u64; for batch in &batches { for payload in build_batch_wire_payloads(terminal_id, batch, attach_request_id, source, batch_max) @@ -4067,10 +5551,334 @@ fn deliver_batches( // The wire payload is exact JSON (camelCase, `type`-tagged); it round-trips // into the frozen `ServerMessage` variant it names. if let Ok(msg) = serde_json::from_value::(payload) { + serialized_bytes += + serde_json::to_string(&msg).map(|j| j.len()).unwrap_or(0) as u64; sink(msg); } } } + serialized_bytes +} + +// ── Paced replay page production (responsive-terminal-restore, W1) ────────── + +/// The completing hold's fixed-boundary finish (round-4, plan:146): the +/// session's own window — everything up to the FIXED completion boundary +/// B — is delivered (the chunk `build` just sank), and the deferral clears +/// ATOMICALLY in this same lock hold. The frames the producer staged past +/// B (`back` bounds the retained tail, read from the ring — never the +/// terminal's current head) are NOT swept: a one-hold ring-sized +/// admission is a pinned replay backlog outside every page budget +/// (plan:145). A quiet terminal (nothing staged past B) completes cleanly +/// — post-hold ingests fan out directly at their own ingest, so the +/// boundary can neither lose nor duplicate a frame. A producing terminal +/// completes AT B with the exact residual interval declared as the +/// bounds-carrying `handoff_boundary_reached` delivery gap (retained and +/// fetchable — the client's bounded baseline recovery, the queue_overflow +/// repair contract, fetches it as a fresh bounded paced session). +fn complete_at_fixed_boundary( + s: &mut TerminalShared, + conn_id: u64, + boundary: i64, + back: (i64, i64), + end_seq: i64, + serialized_bytes: u64, +) -> PacedTailCompletion { + let oldest = s.oldest_retained_seq(); + let sub = s.subscribers.get_mut(&conn_id).expect("subscriber present"); + if back.0 > boundary { + // A residual exists past the boundary: declare its EXACT interval. + let sink = Arc::clone(&sub.sink); + let arid = sub.attach_request_id.clone(); + sink(ServerMessage::TerminalOutputGap(TerminalOutputGap { + terminal_id: s.terminal_id.clone(), + stream_id: s.stream_id.clone(), + attach_request_id: arid, + from_seq: boundary + 1, + to_seq: back.1, + reason: TerminalOutputGapReason::HandoffBoundaryReached, + head_seq: Some(back.1), + oldest_retained_seq: Some(oldest), + })); + sub.paced_deferred = false; + sub.paced_handoff_boundary = None; + // Finding F1: the staged exit rides the completing hold — ordered + // after the residual gap frame, then the subscriber retires. + deliver_staged_paced_exit(s, conn_id); + return PacedTailCompletion::GapCompleted { + lost_from: boundary + 1, + lost_to: back.1, + end_seq, + serialized_bytes, + // The ordinary fixed-boundary residual exit (round-5 finding + // 3's diagnostics discriminator): RETAINED, fetchable loss. + reason: PacedGapExitReason::HandoffBoundaryResidual, + }; + } + sub.paced_deferred = false; + sub.paced_handoff_boundary = None; + deliver_staged_paced_exit(s, conn_id); + PacedTailCompletion::Completed { + end_seq, + serialized_bytes, + } +} + +/// E2R3 test-support: the subscriber-relevant subset of +/// `finish_pty_exit`'s STAGING, under a terminal lock the caller already +/// holds (the decision-path hook site) or has just acquired (the public +/// [`TerminalRegistry::stage_natural_exit_for_test`]). The deterministic +/// transition-race tests stage at a precise point through this twin +/// instead of a real PTY death (whose staging moment would be +/// uncontrolled); the real invariants are exercised by the +/// `finish_pty_exit` tests. Mirrors the production staging's shape: the +/// terminal flips to naturally `Exited` (monotone — a later real exit is +/// inert) and the exit code stages on the DEFERRED subscriber only. +fn stage_natural_exit_locked(s: &mut TerminalShared, conn_id: u64, exit_code: i64) -> bool { + if s.status == TerminalRunStatus::Exited { + return false; // monotone, once-only — the terminal is already dead + } + match s.subscribers.get_mut(&conn_id) { + Some(sub) if sub.paced_deferred => {} + _ => return false, // gone, or a non-deferred subscriber never stages + } + s.status = TerminalRunStatus::Exited; + s.exit_code = Some(exit_code); + s.subscribers + .get_mut(&conn_id) + .expect("subscriber present") + .paced_exit_pending = Some(exit_code); + true +} + +/// Round-2 finding F1: deliver a subscriber's STAGED natural exit, if +/// any, in the same lock hold that just cleared its deferral — the sink +/// is the connection queue (per-terminal FIFO), so the exit leases +/// strictly AFTER everything this hold sank (the final pages and any +/// gap frames), and the subscriber retires exactly like +/// `finish_pty_exit`'s immediate arm retired its non-paced peers. No-op +/// when nothing is staged (the ordinary completion paths stay +/// byte-identical). +fn deliver_staged_paced_exit(s: &mut TerminalShared, conn_id: u64) -> Option { + let code = s.subscribers.get(&conn_id)?.paced_exit_pending?; + let sink = Arc::clone(&s.subscribers.get(&conn_id)?.sink); + sink(ServerMessage::TerminalExit(TerminalExit { + exit_code: code, + terminal_id: s.terminal_id.clone(), + })); + s.subscribers.remove(&conn_id); + tracing::info!( + terminal_id = %s.terminal_id, + conn_id, + exit_code = code, + "terminal.paced_exit_delivered" + ); + Some(code) +} + +/// One built page: its ascending wire messages, the last seq it covers (the +/// session's new production cursor), and the page's total serialized bytes. +struct PacedPageBuild { + messages: Vec, + end_seq: i64, + serialized_bytes: u64, +} + +/// The last sequence covered by a page wire message. +fn page_last_seq(msg: &ServerMessage) -> Option { + match msg { + ServerMessage::TerminalOutput(o) => Some(o.seq_end), + ServerMessage::TerminalOutputBatch(b) => Some(b.seq_end), + _ => None, + } +} + +/// Conservative per-frame wire-segment estimate for the batch projection: +/// `{"seqStart":N,"seqEnd":M,"endOffset":E,"rawFrameCount":C}` plus the +/// `serializedBytes` field's share. The exact per-segment wire size for +/// realistic seq widths (≤11 digits — a frame per millisecond for a year) +/// stays under this, so a walk that stops at the budget never undercounts. +const SEGMENT_WIRE_OVERHEAD_ESTIMATE: i64 = 96; + +/// Fixed part of the once-per-batch envelope delta over the legacy +/// scaffold for the `terminal.output.batch` wire projection: the `.batch` +/// type suffix (7), the `,"serializedBytes":` key (20), and the +/// `,"segments":[` + `]` array wrapper (14). The variable part — the +/// digits of `serializedBytes` itself — is bounded by +/// [`crate::batch::digit_count`] of the page budget for any page that fits +/// it, so the walk charges `41 + digits(budget)` per batch-mode frame. +/// Charged on EVERY batch-mode frame because every produced batch contains +/// at least one frame (a multi-batch page — e.g. barrier-separated frames — +/// pays one envelope PER batch): merged frames were already over-charged +/// their full standalone envelope, so this only narrows their slack. +const BATCH_ENVELOPE_FIXED_BYTES: i64 = 41; + +/// Select and project ONE bounded, ascending page of the +/// `(from_seq, to_seq_inclusive]` window for `conn_id`'s subscriber. The +/// caller holds the terminal lock; frames are selected BEFORE cloning (no +/// full-ring snapshot per page). Packing accounts every frame at its +/// STANDALONE envelope cost (the batch builder's own accounting, reusing +/// `measure_serialized_json_bytes` via the incremental scaffold), plus the +/// batch-segment overhead and the once-per-batch envelope delta +/// (`serializedBytes` + the `segments[]` wrapper + the `.batch` type +/// suffix) on batch-capable subscribers — an overestimate of the merged +/// wire cost, so every produced page's real serialized bytes stay within +/// the budget. A single frame whose own envelope exceeds the budget forms +/// its own atomic single-frame page (guaranteed progress; the oversize +/// result is explicit, never silently coalesced). That arm is +/// UNREACHABLE under supported settings (round-2 finding F2): the +/// fragment splitter's cap is clamped to +/// [`crate::fragment::PACED_PAGE_BUDGET_FLOOR_BYTES`] — the smallest page +/// budget any supported queue setting can produce — so a production +/// frame can never exceed the page budget; the arm is retained as +/// defense-in-depth for the unsupported residue. +fn paced_page_build( + s: &TerminalShared, + conn_id: u64, + from_seq: i64, + to_seq_inclusive: i64, + budget: i64, + source: OutputSource, +) -> Option { + let sub = s.subscribers.get(&conn_id)?; + let batch_mode = sub.terminal_output_batch_v1 && sub.attach_request_id.is_some(); + let arid = sub.attach_request_id.clone(); + let source_str = match source { + OutputSource::Replay => "replay", + OutputSource::Live => "live", + }; + // The batch-mode per-frame charge adds the once-per-batch envelope + // delta: a page that fits the budget serializes every payload at or + // under it, so its `serializedBytes` digits never exceed + // `digit_count(budget)` (an over-budget page is the explicit atomic + // oversize result, which is allowed to exceed). + let batch_mode_charge = if batch_mode { + SEGMENT_WIRE_OVERHEAD_ESTIMATE + + BATCH_ENVELOPE_FIXED_BYTES + + crate::batch::digit_count(budget.max(0)) as i64 + } else { + 0 + }; + + // First ring index with seq_start > from_seq (the ring is seq-ascending; + // binary search avoids an O(ring) scan per page). + let (mut lo, mut hi) = (0usize, s.replay.len()); + while lo < hi { + let mid = (lo + hi) / 2; + if s.replay[mid].output.seq_start <= from_seq { + lo = mid + 1; + } else { + hi = mid; + } + } + let start = lo; + if start >= s.replay.len() || s.replay[start].output.seq_start > to_seq_inclusive { + return None; + } + let scaffold = crate::batch::legacy_envelope_scaffold_bytes( + &s.terminal_id, + &s.replay[start].output.stream_id, + arid.as_deref(), + Some(source_str), + ) as i64; + + let mut selected: Vec = Vec::new(); + let mut page_bytes: i64 = 0; + let mut end_seq = from_seq; + // `range` starts at the binary-searched index in O(1) (unlike + // `iter().skip`, which re-advances from the ring's front). + for f in s.replay.range(start..) { + if f.output.seq_start > to_seq_inclusive { + break; + } + let escaped = crate::batch::json_escaped_len(&f.output.data) as i64; + let digits = (crate::batch::digit_count(f.output.seq_start) + + crate::batch::digit_count(f.output.seq_end)) as i64; + let cost = scaffold + digits + escaped + batch_mode_charge; + if selected.is_empty() { + // Always include the first frame: an over-budget frame forms + // its own atomic single-frame page. + selected.push(f.clone()); + page_bytes = cost; + end_seq = f.output.seq_end; + if cost > budget { + break; + } + continue; + } + if page_bytes + cost > budget { + break; + } + selected.push(f.clone()); + page_bytes += cost; + end_seq = f.output.seq_end; + } + + // Project the selected frames the same way the inline paths deliver them. + let messages: Vec = if batch_mode { + build_batch_messages( + &s.terminal_id, + &selected, + arid.as_deref().unwrap_or(""), + source_str, + ) + } else { + selected + .iter() + .map(|f| { + let mut out = f.output.clone(); + out.attach_request_id = arid.clone(); + out.source = Some(source); + ServerMessage::TerminalOutput(out) + }) + .collect() + }; + let serialized_bytes = messages + .iter() + .map(|m| serde_json::to_string(m).map(|j| j.len()).unwrap_or(0)) + .sum::() as u64; + Some(PacedPageBuild { + messages, + end_seq, + serialized_bytes, + }) +} + +/// The page projection's batch arm: `terminal.output.batch` wire payloads for +/// a bounded page selection (same builder + repacking as +/// [`deliver_batches`], which keeps its per-payload streaming shape for the +/// legacy inline path; pages are budget-bounded, so materializing their +/// messages is bounded by the page budget). +fn build_batch_messages( + terminal_id: &str, + frames: &[RetainedFrame], + attach_request_id: &str, + source: &str, +) -> Vec { + if frames.is_empty() { + return Vec::new(); + } + let batch_max = terminal_stream_batch_max_bytes() as i64; + let inputs: Vec = frames.iter().map(|f| f.to_batch_input()).collect(); + let batches = build_terminal_output_batches(&BatchBuildInput { + frames: &inputs, + max_serialized_bytes: batch_max, + max_total_serialized_bytes: None, + terminal_id: terminal_id.to_string(), + attach_request_id: Some(attach_request_id.to_string()), + source: Some(source.to_string()), + }); + let mut messages = Vec::new(); + for batch in &batches { + for payload in + build_batch_wire_payloads(terminal_id, batch, attach_request_id, source, batch_max) + { + if let Ok(msg) = serde_json::from_value::(payload) { + messages.push(msg); + } + } + } + messages } /// b8ke ext r30 F2 (test-only): the rekey's deterministic INTERLOCK — @@ -4659,6 +6467,15 @@ mod tests { self.insert_headless_at(terminal_id, stream_id, now_ms()); } + /// Round-2 finding F1 test seam: this subscriber's staged exit + /// code, if a natural exit staged one behind an armed deferral + /// (None when the subscriber is gone or nothing is staged). + /// Delegates to the real accessor the ws layer sequences on + /// (E2R1 finding 1's [`TerminalRegistry::staged_paced_exit`]). + fn paced_exit_pending_of(&self, terminal_id: &str, conn_id: u64) -> Option { + self.staged_paced_exit(terminal_id, conn_id) + } + /// Same as [`insert_headless`](Self::insert_headless), but with an /// explicit `created_at` instead of the wall clock. Needed by tests that /// must pin two terminals to the SAME timestamp (e.g. exercising @@ -4752,8 +6569,11 @@ mod tests { Some("legacy".into()), 0, false, + false, None, None, + None, + PacedAttachOptions::default(), ); let legacy = outputs(&legacy_seen); assert!( @@ -4784,8 +6604,11 @@ mod tests { Some("batch".into()), 0, true, + false, None, None, + None, + PacedAttachOptions::default(), ); let bs = batches(&batch_seen); assert!( @@ -4833,7 +6656,19 @@ mod tests { reg.feed("T", frame(1, "a\u{1F600}b\r\n", "S")); // a😀b␍␊ let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("m".into()), 0, true, None, None); + let _ = reg.attach( + "T", + 1, + sink, + Some("m".into()), + 0, + true, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let bs = batches(&seen); assert_eq!(bs.len(), 1); let b = &bs[0]; @@ -4854,7 +6689,19 @@ mod tests { reg.feed("T", frame(3, "three\r\n", "S")); let (sink, seen) = collector(); - let out = reg.attach("T", 1, sink, Some("att-1".into()), 0, false, None, None); + let out = reg.attach( + "T", + 1, + sink, + Some("att-1".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(out.found); // attach.ready first, then the 3 replayed frames. @@ -4881,79 +6728,3965 @@ mod tests { } #[test] - fn detach_keeps_terminal_running_and_buffering_then_replays_on_reattach() { + fn paced_attach_ready_carries_oldest_retained_seq_from_the_ring_front() { + // Restore contract (responsive-terminal-restore): a NEGOTIATED attach + // (pacedTerminalReplayV1) reports the earliest sequence position still + // available for replay — the retained ring's front seqStart. let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "one\r\n", "S")); + reg.feed("T", frame(2, "two\r\n", "S")); - let (sink_a, seen_a) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("a".into()), 0, false, None, None); - reg.feed("T", frame(1, "before\r\n", "S")); - assert_eq!(outputs(&seen_a).len(), 1); - - // Detach: subscription gone, but the terminal keeps running + buffering. - reg.detach("T", 1); - assert!( - reg.is_running("T"), - "terminal survives detach (background session)" + let (sink, seen) = collector(); + let _ = reg.attach( + "T", + 1, + sink, + Some("fresh".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), ); - reg.feed("T", frame(2, "while-detached\r\n", "S")); - // The detached connection receives nothing more. - assert_eq!(outputs(&seen_a).len(), 1); - - // A fresh attach replays the FULL scrollback (both frames). - let (sink_b, seen_b) = collector(); - let _ = reg.attach("T", 2, sink_b, Some("b".into()), 0, false, None, None); - let replayed = outputs(&seen_b); + let ready = attach_ready(&seen).expect("attach.ready sent"); + assert_eq!(ready.head_seq, 2); assert_eq!( - replayed.iter().map(|f| f.data.as_str()).collect::>(), - vec!["before\r\n", "while-detached\r\n"] + ready.oldest_retained_seq, + Some(1), + "a negotiated fresh attach reports the ring front as the retention bound" ); } #[test] - fn two_attached_sockets_both_get_live_output_each_with_its_own_attach_id() { + fn paced_delta_attach_reports_ring_front_not_the_replay_slice() { + // A delta attach (sinceSeq > 0) replays only the newer frames, but the + // retention bound it reports is the RING's front — the earliest + // position still available for a future re-attach, not this attach's + // replay slice. let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "one\r\n", "S")); + reg.feed("T", frame(2, "two\r\n", "S")); + reg.feed("T", frame(3, "three\r\n", "S")); - let (sink_a, seen_a) = collector(); - let (sink_b, seen_b) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("aaa".into()), 0, false, None, None); - // Second attach: geometry authority flips to multi_client_unknown. - let _ = reg.attach("T", 2, sink_b, Some("bbb".into()), 0, false, None, None); - let ready_b = attach_ready(&seen_b).unwrap(); + let (sink, seen) = collector(); + let _ = reg.attach( + "T", + 1, + sink, + Some("delta".into()), + 2, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let ready = attach_ready(&seen).expect("attach.ready sent"); assert_eq!( - ready_b.geometry_authority, - Some(GeometryAuthority::MultiClientUnknown) + ready.replay_from_seq, 3, + "delta attach replays only frame 3" + ); + assert_eq!( + ready.oldest_retained_seq, + Some(1), + "the retention bound is the ring front even when the replay slice starts later" ); - - // One live frame fans out to BOTH sockets, each stamped with its own id. - reg.feed("T", frame(1, "shared\r\n", "S")); - let a = outputs(&seen_a); - let b = outputs(&seen_b); - assert_eq!(a.len(), 1); - assert_eq!(b.len(), 1); - assert_eq!(a[0].data, "shared\r\n"); - assert_eq!(b[0].data, "shared\r\n"); - assert_eq!(a[0].attach_request_id.as_deref(), Some("aaa")); - assert_eq!(b[0].attach_request_id.as_deref(), Some("bbb")); } #[test] - fn reconnect_catches_up_by_seq_only_replaying_newer_frames() { + fn paced_attach_ready_on_empty_ring_reports_head_plus_one() { + // Nothing older than the head is retained => the earliest replayable + // position is headSeq+1 (fresh headless terminal: head 0 => 1). let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); - let (sink_a, seen_a) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("a".into()), 0, false, None, None); - for i in 1..=5 { - reg.feed("T", frame(i, &format!("line-{i}\r\n"), "S")); - } - assert_eq!(outputs(&seen_a).len(), 5); - // Reconnect: the client already rendered through seq 3, so it re-attaches - // with sinceSeq=3. Only frames 4 and 5 are replayed (seqStart > 3). + let (sink, seen) = collector(); + let _ = reg.attach( + "T", + 1, + sink, + Some("empty".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let ready = attach_ready(&seen).expect("attach.ready sent"); + assert_eq!(ready.head_seq, 0); + assert_eq!( + ready.oldest_retained_seq, + Some(1), + "an empty ring retains nothing older than the head: the bound is headSeq+1" + ); + } + + #[test] + fn replay_bounds_reports_head_and_ring_front_without_payloads() { + // The writer-facing accessor: current head plus the earliest + // still-replayable position, in one brief per-terminal lock pass. + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "one\r\n", "S")); + reg.feed("T", frame(2, "two\r\n", "S")); + reg.feed("T", frame(3, "three\r\n", "S")); + assert_eq!( + reg.replay_bounds("T"), + Some(ReplayBounds { + head_seq: 3, + oldest_retained_seq: 1, + }) + ); + } + + #[test] + fn replay_bounds_on_empty_ring_reports_head_plus_one() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + assert_eq!( + reg.replay_bounds("T"), + Some(ReplayBounds { + head_seq: 0, + oldest_retained_seq: 1, + }) + ); + } + + #[test] + fn replay_bounds_for_unknown_terminal_is_none() { + let reg = TerminalRegistry::new(); + assert_eq!(reg.replay_bounds("nope"), None); + } + + // ── Paced replay core (responsive-terminal-restore, Workstream 1) ──────── + // + // The registry's page-read primitives + the attach-time session start. + // The ws pacing coordinator (credit gating, tail drain, events) drives + // these; its behavior is pinned by the `freshell-ws` integration suite. + + /// Flatten one page's wire messages into `(seq, data)` pairs (legacy + /// per-frame and batch pages both reassemble to these). + fn page_seq_data(messages: &[ServerMessage]) -> Vec<(i64, String)> { + let mut out = Vec::new(); + for msg in messages { + match msg { + ServerMessage::TerminalOutput(o) => out.push((o.seq_start, o.data.clone())), + ServerMessage::TerminalOutputBatch(b) => { + // A merged batch is one seq span over its concatenated data. + let mut prev = 0i64; + for seg in &b.segments { + let chunk = crate::batch::slice_utf16(&b.data, prev, seg.end_offset); + out.push((seg.seq_start, chunk)); + prev = seg.end_offset; + } + } + other => panic!("unexpected page message: {other:?}"), + } + } + out + } + + fn page_serialized_bytes(messages: &[ServerMessage]) -> usize { + messages + .iter() + .map(|m| serde_json::to_string(m).expect("page serializes").len()) + .sum() + } + + /// The subscriber-side view of one paced attach: sink messages seen so + /// far (should be ONLY the control prelude — ready/sync/gap) plus the + /// returned first page and session description. + #[test] + fn paced_attach_returns_the_first_page_instead_of_an_inline_replay() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(1024); + reg.insert_headless("T", "S"); + for seq in 1..=8 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-1".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!(out.found); + let start = out.paced.expect("negotiated attach starts a paced session"); + assert_eq!(start.session.terminal_id, "T"); + assert_eq!(start.session.stream_id, "S"); + assert_eq!(start.session.attach_request_id, "paced-1"); + assert_eq!(start.session.target, 8, "target is the head at attach time"); + assert_eq!(start.session.effective_since, 0); + + // The inline sink saw ONLY the control prelude — the replay frames + // travel as returned pages, never sunk under the attach lock. + for msg in seen.lock().unwrap().iter() { + assert!( + !matches!(msg, ServerMessage::TerminalOutput(_)), + "no replay output may be sunk inline: {msg:?}" + ); + } + + // The first page is a bounded ascending prefix of the replay window. + assert!(!start.first_page.is_empty(), "there is replay to page"); + assert!(page_serialized_bytes(&start.first_page) <= 1024); + let page1 = page_seq_data(&start.first_page); + assert_eq!( + page1.first().unwrap().0, + 1, + "the page starts at the baseline+1" + ); + let last_seq = page1.last().unwrap().0; + assert_eq!( + start.session.page_end, last_seq, + "the session cursor is the first page's last seq" + ); + assert!( + last_seq < 8, + "the first page is a bounded prefix, not the whole window" + ); + assert_eq!( + start.session.page_bytes, + page_serialized_bytes(&start.first_page) as u64 + ); + for msg in &start.first_page { + match msg { + ServerMessage::TerminalOutput(o) => { + assert_eq!(o.attach_request_id.as_deref(), Some("paced-1")); + assert_eq!( + o.source, + Some(OutputSource::Replay), + "replay pages are stamped source:'replay'" + ); + } + other => panic!("unexpected first-page message: {other:?}"), + } + } + } + + /// Round-2 finding F3: the negotiated `replayPageBytes` request is an + /// optional UPPER BOUND the server honors on the WHOLE session — + /// `min(requested, registry cap)` when present and positive, the + /// registry cap when absent or invalid — and the recorded session + /// budget is what credits and the tail drain page at later. + #[test] + fn paced_attach_honors_a_smaller_requested_page_budget() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(4096); + reg.insert_headless("T", "S"); + for seq in 1..=12 { + reg.feed("T", frame(seq, &format!("line-{seq:03}\r\n"), "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-budget".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions { + replay_page_bytes: Some(600), + ..PacedAttachOptions::default() + }, + ); + let start = out.paced.expect("paced session"); + assert_eq!( + start.session.page_budget, 600, + "the requested bound clamps to itself when under the cap" + ); + assert!( + page_serialized_bytes(&start.first_page) <= 600, + "the first page is sized to the request, not the cap" + ); + assert!( + start.first_page.len() < 12, + "a 600-byte bound pages the window instead of packing it whole" + ); + // The SAME bound pages the rest of the credited window: a + // registry-cap read would pack far more per page. + let mut cursor = start.session.page_end; + let mut pages = 1; + while cursor < start.session.target { + match reg.next_replay_page("T", 1, cursor, start.session.target, 600) { + PacedPage::Frames { + messages, end_seq, .. + } => { + assert!( + page_serialized_bytes(&messages) <= 600, + "every credited page honors the requested bound" + ); + assert!(end_seq > cursor); + cursor = end_seq; + pages += 1; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + assert!(pages > 2, "the smaller bound means more pages: {pages}"); + } + + #[test] + fn paced_attach_page_budget_clamps_at_the_server_cap() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(4096); + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("line-{seq}\r\n"), "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-cap-clamp".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions { + replay_page_bytes: Some(16 * 1024 * 1024), + ..PacedAttachOptions::default() + }, + ); + let start = out.paced.expect("paced session"); + assert_eq!( + start.session.page_budget, 4096, + "an oversized request clamps to the server's own cap" + ); + } + + #[test] + fn paced_attach_page_budget_absent_or_invalid_keeps_the_server_default() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(4096); + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("line-{seq}\r\n"), "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-default".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert_eq!( + out.paced.expect("paced session").session.page_budget, + 4096, + "an absent request keeps the server default" + ); + + // A non-positive request is invalid, not a bound: the registry + // double-checks positivity (the protocol layer's lossy + // deserializer already filters it on the wire, but the registry + // is also a direct embedder API). + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-invalid".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions { + replay_page_bytes: Some(0), + ..PacedAttachOptions::default() + }, + ); + assert_eq!( + out.paced.expect("paced session").session.page_budget, + 4096, + "an invalid request keeps the server default" + ); + } + + /// E2R1 finding 2: a request below the frame cap can still receive an + /// ATOMIC single-frame page — the page builder always includes the + /// window's first frame even when it alone exceeds the requested + /// budget (a single frame larger than the request forms its own page) + /// — and that page is bounded by the DOCUMENTED ceiling: the fragment + /// cap plus the page-envelope slack, never unbounded. Pinned in BOTH + /// wire forms (the plain per-frame arm and the batch arm), because + /// the drain-admission reservation accounts exactly this ceiling for + /// sub-cap budgets. + #[test] + fn the_atomic_single_frame_page_stays_within_the_documented_ceiling() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(4096); + reg.insert_headless("T", "S"); + // One maximal frame (the worst case production can stage: every + // PTY byte is ingested through the fragment splitter, so a + // frame's serialized payload never exceeds the fragment cap) plus + // a small successor, so the window provably has more to page + // after the atomic result. + let cap = crate::fragment::terminal_stream_batch_max_bytes(); + let mut len = cap; + while crate::fragment::measure_terminal_output_budget_payload_bytes( + "T", + "S", + &"A".repeat(len), + ) > cap + { + len -= 1; + } + reg.feed("T", frame(1, &"A".repeat(len), "S")); + reg.feed("T", frame(2, "after\r\n", "S")); + let ceiling = crate::fragment::paced_atomic_page_serialized_ceiling(); + + for (arid, batch_mode) in [("paced-atomic-plain", false), ("paced-atomic-batch", true)] { + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some(arid.into()), + 0, + batch_mode, + true, + None, + None, + None, + PacedAttachOptions { + replay_page_bytes: Some(2048), + ..PacedAttachOptions::default() + }, + ); + let start = out.paced.expect("paced session"); + assert_eq!( + start.session.page_budget, 2048, + "the sub-cap request is the session's budget ({arid})" + ); + assert_eq!( + start.first_page.len(), + 1, + "a frame larger than the request forms its own atomic single-frame page ({arid})" + ); + let page_bytes = page_serialized_bytes(&start.first_page); + assert!( + page_bytes > 2048, + "the atomic exception is REAL: the page provably exceeds the request ({arid}, {page_bytes}B)" + ); + assert!( + page_bytes <= ceiling, + "the atomic page reaches only the documented ceiling ({arid}: {page_bytes}B > {ceiling}B)" + ); + assert!( + start.session.page_end < start.session.target, + "the atomic page leaves the window's remainder for later pages ({arid})" + ); + } + } + + /// Round-2 finding F1 (Major): a natural exit while a paced + /// subscriber's deferral is armed must NOT sink terminal.exit ahead + /// of the still-deferred final output — the exit is STAGED on the + /// subscriber, the connection is notified (so its credited-phase + /// session can move to the exit-drain), and the completing verdict + /// delivers the staged exit in the SAME lock hold, ordered after the + /// final pages (plan:190 sequenced exit delivery). + #[test] + fn natural_exit_stages_the_paced_exit_until_the_drain_delivers_the_final_output() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(0); // per-frame pages: deterministic cursor control + reg.insert_headless("T", "S"); + for seq in 1..=3 { + reg.feed("T", frame(seq, "before-attach\r\n", "S")); + } + let (sink, seen) = collector(); + let notifies: Arc>> = Arc::new(StdMutex::new(Vec::new())); + let notify_sink = Arc::clone(¬ifies); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-exit".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions { + paced_exit_notify: Some(Arc::new(move |terminal_id, exit_code| { + notify_sink + .lock() + .unwrap() + .push((terminal_id.to_string(), exit_code)); + })), + ..PacedAttachOptions::default() + }, + ); + let start = out.paced.expect("paced session"); + assert_eq!(start.session.page_end, 1, "budget 0 => one frame per page"); + + // The terminal produces its FINAL output past the attach target + // and then exits naturally — the exit hook fires AFTER the last + // ingest (pty.rs runs the hook after the reader's final flush). + for seq in 4..=6 { + reg.feed("T", frame(seq, "FINAL-OUTPUT\r\n", "S")); + } + assert!(reg.finish_pty_exit("T", 0), "the first exit finishes"); + + // THE STAGING: no terminal.exit yet — the deferred range (2..=6) + // is undelivered, and exit-first would strand it (the client's + // exit handler clears the attach and rejects late frames). + assert!( + !seen + .lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalExit(_))), + "terminal.exit must NOT precede the deferred final output" + ); + // The connection was notified so its session can move to the + // exit-drain. + assert_eq!( + notifies.lock().unwrap().as_slice(), + &[("T".to_string(), 0)], + "the staged exit notifies the owning connection" + ); + // The subscriber survives the exit (the drain needs it); a + // LEGACY subscriber on another connection would have been retired + // immediately. + assert_eq!( + reg.paced_exit_pending_of("T", 1), + Some(0), + "the exit is staged on the paced subscriber" + ); + + // The exit-drain (the ws drain task calls complete_paced_tail with + // the head-at-exit target): every deferred page, then the staged + // exit LAST, and the subscriber retires with the exit. + let mut cursor = start.session.page_end; + loop { + match reg.complete_paced_tail("T", 1, "paced-exit", cursor, 6, 0) { + PacedTailCompletion::Handoff { end_seq, .. } => { + assert!(end_seq > cursor, "the exit-drain pages forward"); + cursor = end_seq; + } + PacedTailCompletion::Completed { end_seq, .. } => { + assert!(end_seq >= 6, "the deferred range is fully delivered"); + break; + } + other => panic!("unexpected verdict: {other:?}"), + } + } + let delivered: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some(o.data.clone()), + ServerMessage::TerminalExit(_) => Some("".to_string()), + _ => None, + }) + .collect(); + let exit_at = delivered + .iter() + .position(|d| d == "") + .expect("the staged exit is delivered at completion"); + assert_eq!( + delivered.len(), + exit_at + 1, + "terminal.exit is the LAST frame the client observes: {delivered:?}" + ); + assert!( + delivered + .iter() + .take(exit_at) + .any(|d| d.contains("FINAL-OUTPUT")), + "the deferred final output arrives before the exit: {delivered:?}" + ); + assert_eq!( + reg.paced_exit_pending_of("T", 1), + None, + "the subscriber retired with the delivered exit" + ); + } + + /// E2R1 finding 1: [`TerminalRegistry::staged_paced_exit`] is the + /// AUTHORITY the ws layer's exit-sequencing decisions query (the + /// staging happens under the terminal lock, before the connection's + /// notify hook fires — so a query anywhere after that point observes + /// it regardless of the notify's dispatch order). Pin its + /// truthfulness: None before any exit, None for unknown terminals + /// and connections, Some(code) once a deferred subscriber's terminal + /// exits naturally. + #[test] + fn staged_paced_exit_reports_the_authority_the_ws_layer_sequences_on() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(0); + reg.insert_headless("T", "S"); + for seq in 1..=3 { + reg.feed("T", frame(seq, "history\r\n", "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced-authority".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = out.paced.expect("paced session"); + assert_eq!( + reg.staged_paced_exit("T", 1), + None, + "no exit is staged while the terminal lives" + ); + assert_eq!( + reg.staged_paced_exit("T", 2), + None, + "an unknown connection has nothing staged" + ); + assert_eq!( + reg.staged_paced_exit("T-gone", 1), + None, + "an unknown terminal is None, not a panic" + ); + assert!(reg.finish_pty_exit("T", 7), "the first exit finishes"); + assert_eq!( + reg.staged_paced_exit("T", 1), + Some(7), + "the staged exit code is the ws layer's sequencing authority" + ); + } + + /// E2R3 (the atomic exit-transition decision) test-support seam: the + /// one-shot staging hook fires INSIDE a staged-exit read's + /// terminal-lock hold, immediately AFTER the read — the read itself + /// returns the PRE-staging state (the deterministic model of the PTY + /// reader staging its exit concurrently with the decision's use of + /// that read), the staging is durable, and the hook never fires + /// twice. + #[test] + fn paced_exit_stage_hook_stages_inside_the_reads_lock_scope_once() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + for seq in 1..=2 { + reg.feed("T", frame(seq, "history\r\n", "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("hook-site".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = out.paced.expect("paced session"); + + reg.set_paced_exit_stage_hook_for_tests(1, 9); + assert_eq!( + reg.staged_paced_exit("T", 1), + None, + "the firing read observes the PRE-staging state — the hook stages after it" + ); + assert_eq!( + reg.staged_paced_exit("T", 1), + Some(9), + "the hook's staging landed inside the firing read's lock scope (durable)" + ); + assert_eq!( + reg.staged_paced_exit("T", 1), + Some(9), + "the hook is one-shot: later reads never restage" + ); + // The staging mirrors finish_pty_exit's subscriber-relevant + // subset: the terminal is naturally Exited (monotone — a later + // real exit is inert) and the DEFERRED subscriber holds the code. + assert!( + reg.terminal_is_dead("T"), + "the staged exit marks the terminal dead" + ); + assert!( + !reg.finish_pty_exit("T", 77), + "the later real exit is inert (monotone, once-only)" + ); + assert_eq!(reg.paced_exit_pending_of("T", 1), Some(9)); + } + + /// E2R3 test-support: the public test staging mirrors + /// `finish_pty_exit`'s subscriber-relevant subset — it stages ONLY + /// on a DEFERRED (paced) subscriber, refuses unknown terminals and + /// connections without mutating anything, and is monotone. + #[test] + fn stage_natural_exit_for_test_stages_only_deferred_subscribers_monotonically() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + let (sink, _seen) = collector(); + // A LEGACY subscriber never stages (finish_pty_exit retires it + // instead); an unknown connection or terminal never stages. + let _ = reg.attach( + "T", + 7, + sink, + Some("legacy".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!( + !reg.stage_natural_exit_for_test("T", 7, 4), + "a non-deferred subscriber never stages" + ); + assert!( + !reg.stage_natural_exit_for_test("T", 8, 4), + "an unknown connection never stages" + ); + assert!( + !reg.stage_natural_exit_for_test("T-gone", 7, 4), + "an unknown terminal is false, not a panic" + ); + + let (paced_sink, _paced_seen) = collector(); + let out = reg.attach( + "T", + 1, + paced_sink, + Some("paced-stage".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = out.paced.expect("paced session"); + assert!(reg.stage_natural_exit_for_test("T", 1, 4)); + assert_eq!( + reg.paced_exit_pending_of("T", 1), + Some(4), + "the deferred subscriber holds the staged code" + ); + assert!( + !reg.stage_natural_exit_for_test("T", 1, 5), + "the staging is monotone (once-only)" + ); + assert_eq!( + reg.paced_exit_pending_of("T", 1), + Some(4), + "the refused restage changed nothing" + ); + } + + /// E2R3 (the atomic exit-transition decision): the decision reads + /// the staged exit and the terminal's FROZEN head together under + /// ONE lock hold — `None` before any exit (atomically confirmed), + /// the frozen final head once staged, `None` for unknown terminals + /// and connections. + #[test] + fn paced_exit_transition_reads_the_staging_and_frozen_head_together() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + for seq in 1..=2 { + reg.feed("T", frame(seq, "history\r\n", "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("transition".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = out.paced.expect("paced session"); + + assert_eq!( + reg.paced_exit_transition("T", 1), + PacedExitTransition { exit_head: None }, + "no exit staged: None, atomically confirmed under the decision's hold" + ); + assert_eq!( + reg.paced_exit_transition("T", 2), + PacedExitTransition { exit_head: None }, + "an unknown connection decides None" + ); + assert_eq!( + reg.paced_exit_transition("T-gone", 1), + PacedExitTransition { exit_head: None }, + "an unknown terminal decides None, not a panic" + ); + + // The exit stages with final output past the prior head; the + // decision carries the terminal's FROZEN final head (not the + // exit code) so the caller can extend the credited phase + // through it. + for seq in 3..=4 { + reg.feed("T", frame(seq, "FINAL\r\n", "S")); + } + assert!(reg.finish_pty_exit("T", 5)); + assert_eq!( + reg.paced_exit_transition("T", 1), + PacedExitTransition { exit_head: Some(4) }, + "the decision carries the terminal's frozen final head" + ); + } + + /// E2R3 test-support: the one-shot hook fires at the DECISION site + /// AFTER its read — the decision itself returns the pre-staging + /// state (the deterministic model of the PTY reader staging + /// concurrently with THIS decision's use of its read), and the + /// NEXT decision — with no further staging — reads the staged exit + /// under its own single hold. This is the interleave the ws + /// transition-race tests ride end-to-end. + #[test] + fn paced_exit_transition_hook_stages_after_the_decisions_read() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + for seq in 1..=2 { + reg.feed("T", frame(seq, "history\r\n", "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("decision-hook".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = out.paced.expect("paced session"); + + reg.set_paced_exit_stage_hook_for_tests(1, 6); + assert_eq!( + reg.paced_exit_transition("T", 1), + PacedExitTransition { exit_head: None }, + "the firing decision observes the PRE-staging state — the hook stages after its read" + ); + assert_eq!( + reg.paced_exit_transition("T", 1), + PacedExitTransition { exit_head: Some(2) }, + "the hook's staging landed inside the firing decision's lock scope; \ + the next decision reads the frozen head under its own hold" + ); + } + + /// Round-2 finding F1, the unchanged neighbors: a NON-paced subscriber + /// still receives terminal.exit immediately at the natural exit, and a + /// paced attach to an ALREADY-EXITED terminal keeps the frozen legacy + /// inline replay + synthesized exit (no session, no staging). + #[test] + fn natural_exit_delivers_immediately_to_non_paced_subscribers_and_exited_terminals_stay_legacy() + { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + for seq in 1..=2 { + reg.feed("T", frame(seq, "history\r\n", "S")); + } + // A LEGACY subscriber: exit delivery is immediate and final. + let (legacy_sink, legacy_seen) = collector(); + let _ = reg.attach( + "T", + 7, + legacy_sink, + Some("legacy".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!(reg.finish_pty_exit("T", 3)); + assert!( + legacy_seen + .lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalExit(e) if e.exit_code == 3)), + "the legacy subscriber gets terminal.exit immediately" + ); + assert!( + reg.paced_exit_pending_of("T", 7).is_none(), + "the legacy subscriber retired with its exit" + ); + + // A NEGOTIATED attach to the already-Exited terminal: the legacy + // inline replay + synthesized exit, exactly the frozen shape — no + // paced session, no staging (the attach-after-exit case is + // unchanged by the exit-sequencing fix). + let (paced_sink, paced_seen) = collector(); + let out = reg.attach( + "T", + 8, + paced_sink, + Some("after-exit".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!( + out.paced.is_none(), + "an already-exited terminal never starts a paced session" + ); + let seen = paced_seen.lock().unwrap(); + let exit_at = seen + .iter() + .position(|m| matches!(m, ServerMessage::TerminalExit(_))) + .expect("the synthesized exit arrives with the legacy attach"); + assert!( + seen.iter().take(exit_at).any(|m| matches!( + m, + ServerMessage::TerminalOutput(o) if o.seq_start == 1 + )), + "the frozen inline replay precedes the synthesized exit" + ); + } + + /// A batch-capable negotiated subscriber gets `terminal.output.batch` + /// pages that reassemble to the same bytes as the per-frame projection. + #[test] + fn paced_batch_pages_reassemble_to_the_frame_bytes() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(4096); + reg.insert_headless("T", "S"); + for seq in 1..=5 { + reg.feed("T", frame(seq, &format!("line-{seq}\r\n"), "S")); + } + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 2, + sink, + Some("batch-paced".into()), + 0, + true, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + assert!( + start + .first_page + .iter() + .all(|m| matches!(m, ServerMessage::TerminalOutputBatch(_))), + "a batch-capable subscriber gets batch pages" + ); + // Drive any remaining pages (the first page may be a bounded prefix) + // and reassemble EVERYTHING the session delivers. + let mut page_messages = start.first_page.clone(); + let mut cursor = start.session.page_end; + while cursor < start.session.target { + match reg.next_replay_page("T", 2, cursor, start.session.target, 4096) { + PacedPage::Frames { + messages, end_seq, .. + } => { + page_messages.extend(messages); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + let reassembled: String = { + let mut v: Vec<(i64, String)> = page_seq_data(&page_messages); + v.sort_by_key(|(s, _)| *s); + v.into_iter().map(|(_, d)| d).collect() + }; + assert_eq!( + reassembled, + (1..=5).map(|i| format!("line-{i}\r\n")).collect::() + ); + } + + /// Page-budget honesty at the boundary: the batch-mode charge must + /// cover the one-frame batch ENVELOPE delta — the `.batch` type suffix, + /// the `serializedBytes` field, and the `segments[]` wrapper, which the + /// legacy-scaffold + segment estimate alone never accounted. Two + /// BARRIER frames (BEL ⇒ `turn_complete`) each form their own + /// single-frame batch, so a page carrying both pays TWO batch + /// envelopes; the budget is tuned so the pre-delta accounting admits + /// both frames while their real wire bytes exceed it. With the delta + /// charged, the walk emits each frame as its own single-frame page — + /// every page the walk believes fits the budget really fits. + #[test] + fn paced_batch_page_accounting_covers_the_envelope_at_the_budget_boundary() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + let data_a = format!("{}{}", "a".repeat(120), '\u{0007}'); + let data_b = format!("{}{}", "b".repeat(120), '\u{0007}'); + reg.feed("T", frame(1, &data_a, "S")); + reg.feed("T", frame(2, &data_b, "S")); + + // Size the budget from the crate's own envelope accounting: the + // pre-delta per-frame charge (scaffold + seq digits + escaped data + + // the segment estimate) for BOTH frames plus a 5-byte margin — the + // exact boundary corner where an un-accounted envelope delta can + // push a page's real serialized bytes over the budget. + let scaffold = crate::batch::legacy_envelope_scaffold_bytes( + "T", + "S", + Some("batch-paced"), + Some("replay"), + ) as i64; + let accounted_pre_delta = |data: &str| { + scaffold + + crate::batch::digit_count(1) as i64 + + crate::batch::digit_count(2) as i64 + + crate::batch::json_escaped_len(data) as i64 + + SEGMENT_WIRE_OVERHEAD_ESTIMATE + }; + let budget = accounted_pre_delta(&data_a) + accounted_pre_delta(&data_b) + 5; + reg.set_paced_page_max_bytes(budget); + + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 2, + sink, + Some("batch-paced".into()), + 0, + true, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // The first page honors the budget with REAL serialized bytes. + assert!( + start.session.page_bytes as i64 <= budget, + "the first page's real serialized bytes ({}) must stay within the \ + budget ({budget}) — the batch envelope delta must be accounted", + start.session.page_bytes + ); + assert_eq!( + page_serialized_bytes(&start.first_page) as i64, + start.session.page_bytes as i64, + "the session's page_bytes is the page's real serialized size" + ); + // The envelope corner is real for this fixture: one single-frame + // batch page's wire cost exceeds the pre-delta accounted charge. + assert!( + page_serialized_bytes(&start.first_page) as i64 > accounted_pre_delta(&data_a), + "the fixture must actually exercise the envelope delta corner" + ); + // The walk stopped before frame 2: the first page is a single + // frame's own single-frame batch page. + let page1 = page_seq_data(&start.first_page); + assert_eq!( + page1.iter().map(|(s, _)| *s).collect::>(), + vec![1], + "the delta-inclusive charge leaves frame 2 for its own page" + ); + assert!( + start.first_page.len() == 1, + "one single-frame batch payload" + ); + assert!(matches!( + start.first_page[0], + ServerMessage::TerminalOutputBatch(_) + )); + + // Frame 2 arrives as its own bounded single-frame page. + match reg.next_replay_page("T", 2, start.session.page_end, start.session.target, budget) { + PacedPage::Frames { + messages, + end_seq, + serialized_bytes, + } => { + assert_eq!(end_seq, 2); + assert_eq!(messages.len(), 1, "one single-frame batch payload"); + assert!( + serialized_bytes as i64 <= budget, + "every page the walk emits within its accounting stays \ + within the budget: {serialized_bytes} > {budget}" + ); + assert!( + serialized_bytes as i64 > accounted_pre_delta(&data_b), + "frame 2's real page cost also exceeds the pre-delta charge" + ); + } + other => panic!("expected frame 2's page, got {other:?}"), + } + match reg.next_replay_page("T", 2, 2, start.session.target, budget) { + PacedPage::Done => {} + other => panic!("a drained window reads Done, got {other:?}"), + } + } + + /// Pages ascend within the serialized budget to the FIXED attach-time + /// target — output produced after the attach never extends the replay + /// range (it is tail-phase delivery instead). + #[test] + fn paced_replay_pages_stop_at_the_fixed_target_within_the_budget() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(1024); + reg.insert_headless("T", "S"); + for seq in 1..=8 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + let mut cursor = start.session.page_end; + let mut collected = page_seq_data(&start.first_page); + + // Concurrent output AFTER the attach must NOT extend the target. + for seq in 9..=12 { + reg.feed("T", frame(seq, &format!("post-{seq:03}\r\n"), "S")); + } + + while cursor < 8 { + match reg.next_replay_page("T", 1, cursor, 8, 1024) { + PacedPage::Frames { + messages, + end_seq, + serialized_bytes, + } => { + assert!( + serialized_bytes as usize <= 1024, + "every replay page honors the serialized budget" + ); + assert!(end_seq > cursor, "pages must make progress"); + assert!(end_seq <= 8, "pages never pass the fixed target"); + collected.extend(page_seq_data(&messages)); + cursor = end_seq; + } + PacedPage::Done => panic!("Done while cursor {cursor} < target 8"), + PacedPage::Expired { .. } => panic!("no retention loss in this fixture"), + PacedPage::Gone => panic!("terminal vanished mid-replay"), + } + } + assert_eq!(cursor, 8); + match reg.next_replay_page("T", 1, cursor, 8, 1024) { + PacedPage::Done => {} + other => panic!("a drained replay window reads Done, got {other:?}"), + } + let replayed: String = { + let mut v = collected.clone(); + v.sort_by_key(|(s, _)| *s); + v.into_iter().map(|(_, d)| d).collect() + }; + assert_eq!( + replayed, + (1..=8) + .map(|i| format!("data-{i:03}\r\n")) + .collect::(), + "the replay pages cover the window exactly, in order, no loss/dup" + ); + + // The post-attach frames are TAIL delivery: the drain pages from + // the target toward the FIXED drain target (the head at drain + // start — the terminal is quiet here, so the pages cover exactly + // the post-attach range), then the deferral clears at the + // completing verdict. + let drain_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!(drain_target, 12); + let mut tail_cursor = cursor; + let mut rounds = 0; + let mut handing_off = false; + loop { + rounds += 1; + assert!(rounds <= 32, "the quiet drain terminates in bounded rounds"); + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", tail_cursor, 1024) + } else { + reg.complete_paced_tail("T", 1, "paced", tail_cursor, drain_target, 1024) + }; + match verdict { + PacedTailCompletion::Handoff { + end_seq, + serialized_bytes, + } => { + assert!(serialized_bytes as usize <= 1024); + assert!(end_seq > tail_cursor, "pages must make progress"); + if !handing_off { + assert!(end_seq <= drain_target, "pages never pass the fixed target"); + } + tail_cursor = end_seq; + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // Fixed target covered with a staged remainder beyond: + // the bounded post-target handoff owns it. + tail_cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::CaughtUp => break, + PacedTailCompletion::Completed { end_seq, .. } => { + tail_cursor = tail_cursor.max(end_seq); + break; + } + other => panic!("a quiet terminal completes its drain, got {other:?}"), + } + } + assert_eq!(tail_cursor, 12, "the tail drains to the current head"); + let tail: String = { + let mut v: Vec<(i64, String)> = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) if (9..=12).contains(&o.seq_start) => { + Some((o.seq_start, o.data.clone())) + } + _ => None, + }) + .collect(); + v.sort_by_key(|(s, _)| *s); + v.into_iter().map(|(_, d)| d).collect() + }; + assert_eq!( + tail, + (9..=12) + .map(|i| format!("post-{i:03}\r\n")) + .collect::(), + "the tail delivers exactly the post-attach range" + ); + } + + /// The tail drain must COMPLETE against a continuously-producing + /// terminal within a bounded number of pages (plan W1: "Keep a fixed + /// initial catch-up target so ongoing live output cannot move the + /// completion condition indefinitely"; "Input and other panes' + /// traffic remain schedulable"). This fixture interleaves one new + /// frame with every tail page read — the deterministic model of a + /// producer outrunning the drain — so a drain that re-reads the + /// terminal's CURRENT head per page chases the moving head forever. + /// The fixed-target drain completes: pages up to the tail-start head, + /// then the one-shot handoff delivers whatever the producer staged + /// during the drain and clears the deferral atomically. + #[test] + fn paced_tail_drain_completes_in_bounded_pages_against_a_live_producer() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(1024); + reg.insert_headless("T", "S"); + for seq in 1..=8 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink.clone(), + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the fixed attach-time target (8). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 8 { + match reg.next_replay_page("T", 1, cursor, 8, 1024) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + PacedPage::Done => panic!("Done while cursor {cursor} < target 8"), + PacedPage::Expired { .. } => panic!("no retention loss in this fixture"), + PacedPage::Gone => panic!("terminal vanished mid-replay"), + } + } + + // A staged burst forms the tail, then the FIXED drain target is + // captured ONCE (exactly what the ws drain task does at drain + // start). It is NEVER reassigned below. + for seq in 9..=20 { + reg.feed("T", frame(seq, &format!("post-{seq:03}\r\n"), "S")); + } + let drain_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!( + drain_target, 20, + "the drain target is the head at drain start" + ); + + // The live producer: one new frame lands between EVERY page/chunk + // read. A drain chasing the current head can never catch up; the + // fixed-target drain must complete in bounded pages regardless — + // once the target is covered, the bounded post-target handoff + // delivers whatever the producer staged after the fixed target + // (one page budget per lock hold) and clears atomically when a + // chunk covers the current head. + let mut produced = 20usize; + let mut pages = 0usize; + let mut handing_off = false; + loop { + produced += 1; + reg.feed( + "T", + frame(produced as i64, &format!("live-{produced:03}\r\n"), "S"), + ); + pages += 1; + assert!( + pages <= 8, + "the drain must complete in bounded pages against a \ + live producer (still draining after {pages} pages, cursor \ + {cursor}, drain target {drain_target}, head {produced})", + ); + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", cursor, 1024) + } else { + reg.complete_paced_tail("T", 1, "paced", cursor, drain_target, 1024) + }; + match verdict { + PacedTailCompletion::Handoff { end_seq, .. } => { + assert!(end_seq > cursor, "pages must make progress"); + if !handing_off { + assert!( + end_seq <= drain_target, + "pages never pass the fixed drain target" + ); + } + cursor = end_seq; + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // Fixed target covered — never re-captured: the staged + // remainder flows through the bounded handoff. + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::CaughtUp => { + panic!("CaughtUp with a staged remainder is impossible here") + } + PacedTailCompletion::Completed { end_seq, .. } => { + // Quiet completing hold: everything staged was within + // the fixed boundary (this fixture's producer appends + // between EVERY call, so a residual past the boundary + // exists and the plan:146 exit below is the expected + // shape; this arm is the quiet-case contract). + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::GapCompleted { + lost_from, + lost_to, + end_seq, + .. + } => { + // THE plan:146 FIXED-BOUNDARY EXIT (round-4): the + // session completed AT the fixed boundary with the + // frames the producer staged past it declared as the + // EXACT bounds-carrying delivery gap — never swept + // (plan:145), never chased. The delivered stream tiles + // (.., end_seq] and the gap declares (end_seq, produced]. + assert_eq!( + lost_from, + end_seq + 1, + "the delivery gap starts just past the completing boundary" + ); + assert_eq!( + lost_to, produced as i64, + "the delivery gap declares everything staged past the boundary \ + through the producing head" + ); + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::Expired { .. } => panic!("no retention loss in this fixture"), + PacedTailCompletion::Gone => panic!("terminal vanished mid-drain"), + } + } + + // Everything up to the completion boundary was delivered exactly + // once, in order — the drain pages plus the handoff chunks tile + // the delivered range with no hole and no duplicate; the frames + // past the boundary are DECLARED (the delivery gap), never lost + // silently. + let handed_off: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some(o.seq_end), + _ => None, + }) + .collect(); + assert_eq!( + handed_off, + (9..=cursor).collect::>(), + "the drain + handoff deliver exactly the staged range, in seq order" + ); + let mut seqs: Vec = delivered.iter().map(|(s, _)| *s).collect(); + seqs.extend(handed_off); + seqs.sort_unstable(); + seqs.dedup(); + assert_eq!( + seqs.last(), + Some(&cursor), + "the paged drain + handoff cover everything up to the completion boundary" + ); + for expected in 1..=cursor { + assert!( + seqs.contains(&expected), + "the delivered stream must be contiguous across the handoff: \ + {expected} missing" + ); + } + assert_eq!( + seqs.len(), + cursor as usize, + "every seq is delivered exactly once across pages + handoff" + ); + + // Post-completion output flows through DIRECT fan-out (the + // deferral cleared at the completing hold): a new frame reaches + // the subscriber with no page read at all. + reg.feed("T", frame(produced as i64 + 1, "after-clear\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutput(o) + if o.seq_end == produced as i64 + 1)), + "post-clear output fans out directly to the subscriber" + ); + + // The declared delivery gap is FETCHABLE (the ring retained it): + // the client's bounded baseline recovery — the repair attach from + // the coverage cursor — is the fetch path (the queue_overflow + // repair contract). Modeled here exactly as the client drives it: + // a fresh attach from the completing boundary pages the declared + // interval in order, so the post-boundary frames ARRIVE via the + // normal paced page path, contiguously with the delivered prefix. + let head_after = reg.replay_bounds("T").expect("bounds").head_seq; + let out2 = reg.attach( + "T", + 1, + sink, + Some("paced-repair".into()), + cursor, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let repair = out2.paced.expect("the repair paced session"); + let mut repair_delivered = page_seq_data(&repair.first_page); + let mut repair_cursor = repair.session.page_end; + while repair_cursor < head_after { + match reg.next_replay_page("T", 1, repair_cursor, head_after, 1024) { + PacedPage::Frames { + messages, end_seq, .. + } => { + repair_delivered.extend(page_seq_data(&messages)); + repair_cursor = end_seq; + } + other => panic!("unexpected repair replay read: {other:?}"), + } + } + let repair_seqs: Vec = repair_delivered.iter().map(|(s, _)| *s).collect(); + assert_eq!( + repair_seqs, + (cursor + 1..=head_after).collect::>(), + "the repair session pages the declared interval in order — the \ + post-boundary frames arrive via the normal paced page path" + ); + } + + /// The tail handoff must PAGE the staged remainder through the SAME + /// budget-bounded paging as the replay (responsive-terminal-restore: + /// at most ONE unacknowledged page per (connection, terminal) — an + /// `i64::MAX` full-suffix batch cloned under one lock hold defeats the + /// page budget, blocks PTY ingestion, and monopolizes the dispatch + /// path). A staged tail larger than one page budget must arrive as + /// MULTIPLE handoff pages, each within the paced page budget. + #[test] + fn paced_tail_handoff_pages_the_staged_remainder_within_the_page_budget() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(1024); + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the attach-time target (4). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 4 { + match reg.next_replay_page("T", 1, cursor, 4, 1024) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + + // A staged tail far larger than one page budget (each frame ~120 + // serialized bytes → ~8 frames per 1024-byte page). + for seq in 5..=40 { + reg.feed("T", frame(seq, &format!("staged-{seq:03}\r\n"), "S")); + } + let completion_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!(completion_target, 40); + + // The completion drives the staged remainder through + // budget-bounded pages: more than one handoff page for this staged + // tail, each within the page budget, ending in EXACTLY ONE + // completing verdict that cleared the deferral. + let mut pages = 0usize; + let mut completions = 0usize; + let mut final_page_over_budget = false; + let mut handing_off = false; + loop { + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", cursor, 1024) + } else { + reg.complete_paced_tail("T", 1, "paced", cursor, completion_target, 1024) + }; + match verdict { + PacedTailCompletion::Handoff { + end_seq, + serialized_bytes, + } => { + assert!( + serialized_bytes as usize <= 1024, + "each handoff page must stay within the paced page budget \ + (got {serialized_bytes} bytes for the page ending at seq {end_seq})" + ); + assert!(end_seq > cursor); + if !handing_off { + assert!( + end_seq <= completion_target, + "pages never pass the fixed drain target" + ); + } + cursor = end_seq; + pages += 1; + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // Fixed target covered with a staged remainder beyond: + // the bounded post-target handoff owns it. + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::Completed { + end_seq, + serialized_bytes, + } => { + // The final page may honestly exceed the budget only as + // the explicit atomic oversize result (a single frame + // larger than the budget); this fixture's frames are + // tiny, so it must fit too. + if serialized_bytes as usize > 1024 { + final_page_over_budget = true; + } + assert_eq!( + end_seq, 40, + "the fixed-target boundary delivers the whole staged tail" + ); + cursor = end_seq; + completions += 1; + break; + } + PacedTailCompletion::CaughtUp => { + panic!("a staged remainder exists — the handoff must deliver it") + } + PacedTailCompletion::Expired { .. } => panic!("no retention loss in this fixture"), + PacedTailCompletion::GapCompleted { .. } => { + panic!("no mid-handoff retention loss in this fixture") + } + PacedTailCompletion::Gone => panic!("terminal vanished at completion"), + } + } + assert!( + pages > 1, + "a staged tail larger than one page budget must produce multiple \ + handoff pages, got {pages}" + ); + assert_eq!(completions, 1, "the handoff completes exactly once"); + assert!( + !final_page_over_budget, + "the final page fits the budget in this fixture" + ); + assert_eq!(cursor, 40, "the handoff drained everything staged"); + + // Everything staged was delivered through the sink, contiguously, + // exactly once (the pages + handoff partition the range). + let handed_off: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some(o.seq_end), + _ => None, + }) + .collect(); + assert_eq!( + handed_off, + (5..=cursor).collect::>(), + "the handoff delivers exactly the staged remainder, in seq order" + ); + let mut all: Vec = delivered.iter().map(|(s, _)| *s).collect(); + all.extend(handed_off); + all.sort_unstable(); + all.dedup(); + assert_eq!( + all, + (1..=cursor).collect::>(), + "the delivered stream is contiguous across the handoff" + ); + } + + /// A retention advance past the drain cursor must NEVER become a silent + /// forward jump in the handoff (responsive-terminal-restore: retention + /// loss mid-restore is an exact bounds-carrying gap, never a silent + /// truncation). The completion must emit the retention gap for the + /// omitted interval BEFORE any delivered frame, with bounds matching + /// the actual eviction. + #[test] + fn paced_tail_completion_reports_retention_loss_instead_of_a_silent_forward_jump() { + let reg = TerminalRegistry::new(); + // Tiny CHAR ring so feeding evicts the front deterministically. + reg.set_scrollback_max_bytes(60); + reg.insert_headless("T", "S"); + reg.set_paced_page_max_bytes(0); // per-frame pages: deterministic cursor control + for seq in 1..=3 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); // 10 chars each + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the attach-time target (3). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 3 { + match reg.next_replay_page("T", 1, cursor, 3, 0) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + assert_eq!(cursor, 3); + + // Evict frames 4.. well past the drain cursor: 10 more chunks (100 + // chars) push the front far past seq 3. + for seq in 4..=13 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + let bounds = reg.replay_bounds("T").expect("bounds"); + assert!( + bounds.oldest_retained_seq > 4, + "the fixture evicted the frames the handoff needs next" + ); + + // The handoff must declare the retention loss for the omitted + // interval with the EXACT bounds — never silently start at the + // ring front: a forward jump from seq 3 with no gap frame is + // silent corruption. The caller (the ws drive loop) sinks the + // bounds-carrying gap and resumes from the ring front; the same + // continuation is driven here. + let head_at_expiry = bounds.head_seq; + let gap_lo = cursor + 1; + let seen_len_at_expiry = seen.lock().unwrap().len(); + match reg.complete_paced_tail("T", 1, "paced", cursor, head_at_expiry, 0) { + PacedTailCompletion::Expired { + lost_from, + lost_to, + resume_from, + head_seq, + oldest_retained_seq, + } => { + assert_eq!( + lost_from, + cursor + 1, + "the lost interval starts at cursor+1" + ); + assert_eq!( + lost_to, + bounds.oldest_retained_seq - 1, + "the lost interval ends just before the new ring front" + ); + assert_eq!(resume_from, bounds.oldest_retained_seq - 1); + assert_eq!(head_seq, bounds.head_seq); + assert_eq!(oldest_retained_seq, bounds.oldest_retained_seq); + cursor = resume_from; + } + other => panic!( + "a retention advance past the tail boundary must report the exact \ + gap, never a silent forward jump, got {other:?}" + ), + } + assert_eq!( + seen.lock().unwrap().len(), + seen_len_at_expiry, + "the Expired verdict delivers NOTHING itself — the caller emits the gap" + ); + + // The caller's continuation: sink the exact gap (as the ws drain + // task does), then resume the paged drain from the ring front. + // The delivered stream is contiguous except for EXACTLY the + // asserted gap: 1..=3 delivered, 4..=oldest-1 declared lost, + // oldest..=head delivered in order. + let drain_target = head_at_expiry; + let mut handing_off = false; + loop { + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", cursor, 0) + } else { + reg.complete_paced_tail("T", 1, "paced", cursor, drain_target, 0) + }; + match verdict { + PacedTailCompletion::Handoff { end_seq, .. } => { + if !handing_off { + assert!(end_seq <= drain_target, "pages never pass the fixed target"); + } + cursor = end_seq; + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // Fixed target covered with a staged remainder beyond: + // the bounded post-target handoff owns it (mirrors the + // ws drain task). + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::Completed { end_seq, .. } => { + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::CaughtUp => break, + PacedTailCompletion::Expired { .. } => { + panic!("the eviction is fully declared — no second loss expected") + } + PacedTailCompletion::GapCompleted { .. } => { + panic!("the eviction is fully declared — no mid-handoff loss expected") + } + PacedTailCompletion::Gone => panic!("terminal vanished mid-handoff"), + } + } + assert_eq!( + cursor, bounds.head_seq, + "the resumed handoff drains to the head" + ); + let handed_off: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some(o.seq_end), + _ => None, + }) + .collect(); + assert_eq!( + handed_off, + (bounds.oldest_retained_seq..=bounds.head_seq).collect::>(), + "the resumed handoff delivers exactly the retained remainder, in seq order" + ); + // Contiguity modulo exactly the asserted gap: the delivered seqs + // plus the declared-lost interval tile 1..=head with no hole and + // no overlap. + let mut delivered_seqs: Vec = delivered.iter().map(|(s, _)| *s).collect(); + delivered_seqs.extend(handed_off); + delivered_seqs.sort_unstable(); + delivered_seqs.dedup(); + let mut expected: Vec = (1..=bounds.head_seq).collect(); + expected.retain(|seq| !(gap_lo..=bounds.oldest_retained_seq - 1).contains(seq)); + assert_eq!( + delivered_seqs, expected, + "the delivered stream is contiguous except exactly the asserted gap" + ); + + // Post-clear output flows through DIRECT fan-out (the deferral + // cleared at the completing verdict). + reg.feed("T", frame(bounds.head_seq + 1, "after-clear\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutput(o) + if o.seq_end == bounds.head_seq + 1)), + "post-clear output fans out directly to the subscriber" + ); + } + + /// The completion target is captured ONCE and never re-captured (the + /// round-2 recapture-loophole fix): when the FIXED target is covered, + /// the staged post-target remainder flows through the BOUNDED + /// post-target handoff — page-budget-sized chunks per lock hold, the + /// lock released between them (round-3 fix; never a bulk re-fan + /// under one hold) — until a chunk covers the terminal's current + /// head and the deferral clears atomically in that hold. Never by + /// re-reading the terminal's current head and re-targeting the + /// DRAIN: a drain that recaptures whenever a target is covered + /// chases a sustained producer one "bounded" page at a time, + /// indefinitely. + #[test] + fn paced_tail_completion_completes_at_the_fixed_target_never_recapturing() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(1024); + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the attach-time target (4). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 4 { + match reg.next_replay_page("T", 1, cursor, 4, 1024) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + assert_eq!(cursor, 4); + + // A staged tail (production during the replay), and the FIXED + // completion target captured ONCE at completion start — exactly + // what the ws drain task does. It is NEVER reassigned below. + for seq in 5..=24 { + reg.feed("T", frame(seq, &format!("staged-{seq:03}\r\n"), "S")); + } + let completion_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!(completion_target, 24); + + // The live producer: one new frame lands between EVERY page/chunk + // read. The drain pages toward the FIXED target — never the + // current head — and once the target is covered, the bounded + // handoff delivers the staged remainder toward the current head, + // one page budget per lock hold (the producer's single frame per + // round is cleared many frames per chunk, so it converges). + let mut produced = 24usize; + let mut rounds = 0usize; + let mut completed = false; + let mut handing_off = false; + while !completed { + produced += 1; + reg.feed( + "T", + frame(produced as i64, &format!("live-{produced:03}\r\n"), "S"), + ); + rounds += 1; + assert!( + rounds <= 64, + "the completion must terminate in bounded rounds (fixed target {completion_target}, \ + cursor {cursor}, produced {produced})" + ); + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", cursor, 1024) + } else { + reg.complete_paced_tail("T", 1, "paced", cursor, completion_target, 1024) + }; + match verdict { + PacedTailCompletion::Handoff { + end_seq, + serialized_bytes, + } => { + assert!( + serialized_bytes <= 1024, + "every page/chunk admits at most one page budget per lock hold" + ); + assert!(end_seq > cursor, "pages must make progress"); + if !handing_off { + assert!( + end_seq <= completion_target, + "drain pages never pass the fixed target" + ); + } + cursor = end_seq; + } + PacedTailCompletion::TargetCovered { + end_seq, + serialized_bytes, + } => { + // The FIXED target is covered — NEVER re-captured: the + // remainder now flows through the bounded handoff. + assert!( + serialized_bytes <= 1024, + "the target-covering page admits at most one page budget" + ); + assert!( + end_seq <= completion_target, + "the target-covering page never passes the fixed target" + ); + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::Completed { end_seq, .. } => { + // Quiet completing hold: everything staged was within + // the fixed boundary. The fixture's producer appends + // between EVERY round, so the expected shape is the + // plan:146 exit below; this arm pins the quiet case. + cursor = cursor.max(end_seq); + completed = true; + } + PacedTailCompletion::CaughtUp => { + panic!("a staged remainder exists — the completion must deliver it") + } + PacedTailCompletion::Expired { .. } => { + panic!("no retention loss in this fixture") + } + PacedTailCompletion::GapCompleted { + lost_from, + lost_to, + end_seq, + .. + } => { + // THE plan:146 FIXED-BOUNDARY EXIT (round-4): the + // session completed AT the fixed boundary — the frames + // the producer staged past it are DECLARED as the + // exact bounds-carrying delivery gap, never swept, + // never chased; the client's bounded baseline recovery + // fetches them (the repair attach below). + assert_eq!( + lost_from, + end_seq + 1, + "the delivery gap starts just past the completing boundary" + ); + assert_eq!( + lost_to, produced as i64, + "the delivery gap declares everything staged past the boundary" + ); + cursor = cursor.max(end_seq); + completed = true; + } + PacedTailCompletion::Gone => panic!("terminal vanished at completion"), + } + } + assert!( + cursor >= completion_target, + "the fixed target was covered before completing" + ); + assert!( + handing_off, + "the fixture's producer outran the fixed target, so the post-target \ + remainder must have flowed through the bounded handoff" + ); + + // The delivered stream tiles 1..=cursor contiguously — the pages + // (≤ target) and the handoff chunks (target, boundary] partition + // the range with no hole and no duplicate; everything past the + // boundary is DECLARED (the bounds-carrying delivery gap), never + // silently lost. + let handed_off: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some(o.seq_end), + _ => None, + }) + .collect(); + assert_eq!( + handed_off, + (5..=cursor).collect::>(), + "the completion delivers exactly the staged range, in seq order" + ); + let mut all: Vec = delivered.iter().map(|(s, _)| *s).collect(); + all.extend(handed_off); + all.sort_unstable(); + all.dedup(); + assert_eq!( + all, + (1..=cursor).collect::>(), + "the delivered stream is contiguous across the fixed-target boundary" + ); + + // Post-clear output flows through DIRECT fan-out (the deferral + // cleared at the completing verdict). + reg.feed("T", frame(cursor + 1, "after-clear\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutput(o) + if o.seq_end == cursor + 1)), + "post-clear output fans out directly to the subscriber" + ); + } + + /// The fixed-target completion must never admit more than ONE page + /// budget of data in a single lock hold (round-3 finding: the + /// completing call bulk-cloned the ENTIRE retained post-target + /// remainder under the terminal mutex — up to a full ring per + /// restoring pane — and admitted it to the connection queue outside + /// every bound). While a producer keeps producing past the fixed + /// target, the staged post-target remainder grows toward ring + /// capacity; the completion must still deliver it through the normal + /// live delivery path, in seq order, with EVERY verdict's admitted + /// serialized bytes within one page budget. + #[test] + fn paced_tail_completion_admits_at_most_one_page_budget_per_lock_hold() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(1024); + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the attach-time target (4). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 4 { + match reg.next_replay_page("T", 1, cursor, 4, 1024) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + + // The staged range below the FIXED completion target (5..=24), + // then a LARGE post-target remainder (25..=200) produced before + // the drain even starts — the completing call faces ~176 frames + // (a multi-page remainder) staged past the target. + for seq in 5..=24 { + reg.feed("T", frame(seq, &format!("staged-{seq:03}\r\n"), "S")); + } + let completion_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!(completion_target, 24); + for seq in 25..=200 { + reg.feed("T", frame(seq, &format!("remainder-{seq:03}\r\n"), "S")); + } + + // The completion target is FIXED and NEVER reassigned below. The + // drain pages toward it; once it is covered, the staged post-target + // remainder flows through the bounded post-target handoff — one + // page budget per lock hold, the "lock" held only per chunk. + let mut rounds = 0usize; + let mut per_hold_bytes: Vec = Vec::new(); + let mut handing_off = false; + loop { + rounds += 1; + assert!( + rounds <= 256, + "the completion must terminate in bounded rounds (cursor {cursor}, \ + target {completion_target})" + ); + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", cursor, 1024) + } else { + reg.complete_paced_tail("T", 1, "paced", cursor, completion_target, 1024) + }; + match verdict { + PacedTailCompletion::Handoff { + end_seq, + serialized_bytes, + } => { + assert!( + serialized_bytes <= 1024, + "each page/chunk admits at most one page budget per lock hold \ + (got {serialized_bytes})" + ); + assert!(end_seq > cursor, "pages must make progress"); + if !handing_off { + assert!( + end_seq <= completion_target, + "pages never pass the fixed target" + ); + } + per_hold_bytes.push(serialized_bytes); + cursor = end_seq; + } + PacedTailCompletion::TargetCovered { + end_seq, + serialized_bytes, + } => { + // THE ROUND-3 PER-HOLD BOUND at the target boundary: the + // target-covering page admits at most ONE page budget — + // the staged post-target remainder is NEVER bulk-cloned + // in this hold; it flows through the bounded handoff. + assert!( + serialized_bytes <= 1024, + "the completion admits at most one page budget per lock hold \ + (got {serialized_bytes} bytes — a bulk post-target re-fan)" + ); + assert!( + end_seq <= completion_target, + "the fixed target is the boundary" + ); + per_hold_bytes.push(serialized_bytes); + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::Completed { + end_seq, + serialized_bytes, + } => { + assert!( + serialized_bytes <= 1024, + "the clearing chunk admits at most one page budget per lock hold \ + (got {serialized_bytes})" + ); + assert!( + end_seq >= 200, + "everything staged through the producing head is delivered in \ + order (end {end_seq}, produced 200)" + ); + per_hold_bytes.push(serialized_bytes); + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::CaughtUp => { + panic!("a staged remainder exists — the completion must deliver it") + } + PacedTailCompletion::Expired { .. } => panic!("no retention loss in this fixture"), + PacedTailCompletion::GapCompleted { .. } => { + panic!("no mid-handoff retention loss in this fixture") + } + PacedTailCompletion::Gone => panic!("terminal vanished at completion"), + } + } + assert!( + per_hold_bytes.len() > 1, + "a multi-page remainder must never be delivered in one lock hold" + ); + assert!( + handing_off, + "the multi-page post-target remainder must flow through the bounded handoff" + ); + + // The post-target frames arrived through the normal live path, in + // seq order: the sink's delivered seqs tile (5..=cursor) exactly + // once, contiguously, across the whole completion. + let handed_off: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some(o.seq_end), + _ => None, + }) + .collect(); + assert_eq!( + handed_off, + (5..=cursor).collect::>(), + "the completion delivers exactly the staged range, in seq order" + ); + let mut all: Vec = delivered.iter().map(|(s, _)| *s).collect(); + all.extend(handed_off); + all.sort_unstable(); + all.dedup(); + assert_eq!( + all, + (1..=cursor).collect::>(), + "the delivered stream is contiguous across the fixed-target boundary" + ); + + // Post-clear output flows through DIRECT fan-out (the deferral + // cleared at the completing verdict). + reg.feed("T", frame(cursor + 1, "after-clear\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutput(o) + if o.seq_end == cursor + 1)), + "post-clear output fans out directly to the subscriber" + ); + } + + /// The handoff's completion boundary is the head captured ONCE at + /// completion start (the round-4 fix): the handoff delivers + /// budget-bounded pages up to that FIXED boundary and ONLY it — never + /// the terminal's CURRENT head. This fixture models production + /// AT-OR-ABOVE drain speed the deterministic way: the page budget sits + /// below one frame's serialized envelope, so EVERY page is the explicit + /// atomic single-frame page, and the producer appends at least one + /// frame per page consumed — a handoff that re-reads the head per call + /// (the pre-fix code) moves its completion boundary with every append + /// and can NEVER complete (the moving-head chase). The fixed-boundary + /// handoff completes within the B-DERIVED page bound, the deferral + /// clears atomically at the completing hold, and the frames the + /// producer staged past the boundary flow through the normal live + /// path (the completing hold's live sweep), in seq order. + #[test] + fn handoff_completes_at_the_fixed_boundary_against_at_or_above_production() { + let reg = TerminalRegistry::new(); + // Below every frame's serialized envelope: one atomic frame per + // page, so one producer frame per page consumed IS at-or-above + // drain speed (the gated-handoff model the round-4 brief requires). + reg.set_paced_page_max_bytes(64); + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink.clone(), + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the attach-time target (4). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 4 { + match reg.next_replay_page("T", 1, cursor, 4, 64) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + assert_eq!(cursor, 4); + + // The staged tail forms the drain window; the FIXED drain target + // is captured ONCE (exactly what the ws drain task does). + for seq in 5..=14 { + reg.feed("T", frame(seq, &format!("staged-{seq:03}\r\n"), "S")); + } + let drain_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!(drain_target, 14); + + // Phase 1 — the drain pages toward the FIXED target. The producer + // appends one frame per page consumed here too; the fixed target + // keeps this phase convergent by construction. + let mut produced = 14usize; + let mut pages = 0usize; + let boundary = loop { + produced += 1; + reg.feed( + "T", + frame(produced as i64, &format!("live-{produced:03}\r\n"), "S"), + ); + pages += 1; + assert!(pages <= 64, "the drain converges to its fixed target"); + match reg.complete_paced_tail("T", 1, "paced", cursor, drain_target, 64) { + PacedTailCompletion::Handoff { end_seq, .. } => { + assert!(end_seq > cursor, "pages must make progress"); + assert!(end_seq <= drain_target, "pages never pass the fixed target"); + cursor = end_seq; + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // COMPLETION START: the boundary B is the head captured + // ONCE under THIS hold (the fixture is single-threaded, + // so the head right after the verdict is exactly it). + assert!( + end_seq <= drain_target, + "the fixed target is the drain boundary" + ); + cursor = end_seq; + let boundary = reg.replay_bounds("T").expect("bounds").head_seq; + assert!( + boundary > drain_target, + "the producer staged past the target, so the handoff window is non-empty" + ); + break boundary; + } + PacedTailCompletion::Completed { .. } => { + panic!("a staged remainder exists past the fixed target") + } + PacedTailCompletion::CaughtUp => { + panic!("a staged remainder exists — the drain must page it") + } + PacedTailCompletion::Expired { .. } => panic!("no retention loss in this fixture"), + PacedTailCompletion::GapCompleted { .. } => { + panic!("the drain phase never takes the mid-handoff exit") + } + PacedTailCompletion::Gone => panic!("terminal vanished mid-drain"), + } + }; + + // THE B-DERIVED PAGE BOUND for the handoff: the window + // (cursor, boundary] at one frame per page, plus slack for the + // completing hold. A handoff that chases the per-call head exceeds + // this bound against the at-or-above producer and never completes. + let handoff_bound = (boundary - cursor) as usize + 4; + let mut handoff_pages = 0usize; + let mut declared_lost_to = 0i64; + loop { + // Production at-or-above drain speed: at least one appended + // frame per page consumed, exactly as the round-4 brief + // requires (the gated handoff's deterministic model). + produced += 1; + reg.feed( + "T", + frame(produced as i64, &format!("live-{produced:03}\r\n"), "S"), + ); + handoff_pages += 1; + assert!( + handoff_pages <= handoff_bound, + "the handoff completes within the B-derived page bound \ + (boundary {boundary}, cursor {cursor}, pages {handoff_pages}, \ + bound {handoff_bound}) — a per-call head read is the \ + moving-head chase" + ); + match reg.handoff_paced_tail("T", 1, "paced", cursor, 64) { + PacedTailCompletion::Handoff { end_seq, .. } => { + assert!(end_seq > cursor, "chunks must make progress"); + assert!( + end_seq <= boundary, + "chunks never pass the FIXED completion boundary" + ); + cursor = end_seq; + } + PacedTailCompletion::Completed { end_seq, .. } => { + // Quiet completing hold (nothing staged past the + // boundary — not this fixture's shape; the arm pins + // the quiet contract). + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::CaughtUp => break, + PacedTailCompletion::TargetCovered { .. } => { + panic!("the handoff never re-targets: a covered boundary completes") + } + PacedTailCompletion::Expired { .. } => { + panic!("no retention loss in this fixture") + } + PacedTailCompletion::GapCompleted { + lost_from, + lost_to, + end_seq, + reason, + .. + } => { + // THE plan:146 FIXED-BOUNDARY EXIT: the session + // completed AT B; the frames the producer staged past + // B are DECLARED as the exact bounds-carrying + // delivery gap — never swept (plan:145), never chased. + assert_eq!( + lost_from, + end_seq + 1, + "the delivery gap starts just past the fixed boundary" + ); + assert_eq!( + lost_to, produced as i64, + "the delivery gap declares everything staged past the boundary \ + through the producing head" + ); + // Round-5 finding 3: the diagnostics carry the + // mandatory gap-exit reason — the ORDINARY fetchable + // fixed-boundary residual exit, never a retention + // overrun (the declared interval is RETAINED and the + // client's checkpoint-cursor repair fetches it). + assert_eq!( + reason, + PacedGapExitReason::HandoffBoundaryResidual, + "the fixed-boundary residual's GapCompleted carries the \ + handoff_boundary_residual reason (structured diagnostics \ + consumers filter on)" + ); + declared_lost_to = lost_to; + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::Gone => panic!("terminal vanished mid-handoff"), + } + } + assert!( + cursor >= boundary, + "the handoff covered the fixed completion boundary" + ); + let boundary_at_completion = cursor; + + // The delivered stream tiles (5..=boundary_at_completion) exactly + // once, in seq order: the drain pages plus the handoff chunks; the + // frames past the boundary are DECLARED, never silently lost. + let handed_off: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) if o.seq_end > 4 => Some(o.seq_end), + _ => None, + }) + .collect(); + assert_eq!( + handed_off, + (5..=boundary_at_completion).collect::>(), + "drain pages + handoff chunks tile the session's window exactly \ + once, in seq order — the boundary exit declares (never \ + delivers) the residual past it" + ); + let mut all: Vec = delivered.iter().map(|(s, _)| *s).collect(); + all.extend(handed_off); + all.sort_unstable(); + all.dedup(); + assert_eq!( + all, + (1..=boundary_at_completion).collect::>(), + "the delivered stream is contiguous across the whole completion" + ); + // The exact bounds-carrying delivery gap was sunk through the + // subscriber's sink, ordered after the chunk that covered B. + let stream = seen.lock().unwrap().clone(); + let gap = stream + .iter() + .find(|m| { + matches!(m, ServerMessage::TerminalOutputGap(g) + if g.from_seq == boundary_at_completion + 1 + && g.to_seq == declared_lost_to) + }) + .expect("the exact bounds-carrying delivery gap for the residual"); + assert!( + matches!(gap, ServerMessage::TerminalOutputGap(g) + if g.reason == TerminalOutputGapReason::HandoffBoundaryReached), + "the residual exit declares the handoff_boundary_reached delivery gap: {gap:?}" + ); + + // Post-clear output flows through DIRECT fan-out (the deferral + // cleared at the completing hold): a frame appended after the + // completion reaches the subscriber with no handoff call at all. + let seen_len = seen.lock().unwrap().len(); + reg.feed( + "T", + frame(boundary_at_completion + 1, "after-clear\r\n", "S"), + ); + assert!( + seen.lock().unwrap().len() > seen_len, + "post-clear output fans out directly to the subscriber" + ); + + // The declared interval is FETCHABLE (the ring retained it): the + // client's bounded baseline recovery — the repair attach from the + // coverage cursor (the queue_overflow repair contract) — fetches + // it as a fresh bounded paced session, so the post-boundary + // frames ARRIVE via the normal paced page path, in order. + let head_after = reg.replay_bounds("T").expect("bounds").head_seq; + let out2 = reg.attach( + "T", + 1, + sink, + Some("paced-repair".into()), + boundary_at_completion, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let repair = out2.paced.expect("the repair paced session"); + let mut repair_delivered = page_seq_data(&repair.first_page); + let mut repair_cursor = repair.session.page_end; + while repair_cursor < head_after { + match reg.next_replay_page("T", 1, repair_cursor, head_after, 64) { + PacedPage::Frames { + messages, end_seq, .. + } => { + repair_delivered.extend(page_seq_data(&messages)); + repair_cursor = end_seq; + } + other => panic!("unexpected repair replay read: {other:?}"), + } + } + let repair_seqs: Vec = repair_delivered.iter().map(|(s, _)| *s).collect(); + assert_eq!( + repair_seqs, + (boundary_at_completion + 1..=head_after).collect::>(), + "the repair session pages the declared interval in order — the \ + post-B frames arrive via the normal path" + ); + } + + /// Retention overrun past the handoff cursor mid-handoff is the plan:146 + /// BOUNDED-BASELINE EXIT (round-4): the exact bounds-carrying gap for + /// the evicted interval, then the session COMPLETES AT THE RING FRONT + /// with the gap recorded — never a resumption of the paged handoff + /// toward an unreachable boundary (the pre-fix code resumed the chase + /// from the ring front: `Expired` + continue). The retained window + /// (ring front, head] flows through the normal live path in the same + /// completing hold, in seq order, so the delivered stream is contiguous + /// except exactly the declared gap; the deferral clears atomically and + /// live output resumes directly. + #[test] + fn handoff_retention_overrun_declares_the_gap_and_completes_at_the_ring_front() { + let reg = TerminalRegistry::new(); + // Tiny CHAR ring so feeding evicts the front deterministically. + reg.set_scrollback_max_bytes(60); + reg.set_paced_page_max_bytes(0); // per-frame pages: deterministic cursor control + reg.insert_headless("T", "S"); + for seq in 1..=3 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); // 10 chars each + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink.clone(), + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + + // Drive the replay to the attach-time target (3). + let mut cursor = start.session.page_end; + let mut delivered = page_seq_data(&start.first_page); + while cursor < 3 { + match reg.next_replay_page("T", 1, cursor, 3, 0) { + PacedPage::Frames { + messages, end_seq, .. + } => { + delivered.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + assert_eq!(cursor, 3); + + // Stage the drain window (4..=5) and then production past it + // (6..=7): 30 + 20 + 20 = 70 chars > the 60-char ring, so frame 1 + // evicts — the drain window stays retained. + for seq in 4..=5 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + let drain_target = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!(drain_target, 5); + for seq in 6..=7 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + + // Phase 1 — the drain pages its fixed target (per-frame pages) with + // frames staged past it: TargetCovered, and the boundary B is the + // head at that hold (7). + let verdict = loop { + match reg.complete_paced_tail("T", 1, "paced", cursor, drain_target, 0) { + PacedTailCompletion::Handoff { end_seq, .. } => { + assert!(end_seq > cursor, "pages must make progress"); + cursor = end_seq; + } + verdict @ PacedTailCompletion::TargetCovered { .. } => break verdict, + other => panic!("the drain must cover its fixed target, got {other:?}"), + } + }; + match verdict { + PacedTailCompletion::TargetCovered { end_seq, .. } => { + assert_eq!(end_seq, 5); + cursor = end_seq; + } + _ => unreachable!("the loop above only breaks on TargetCovered"), + } + let boundary = reg.replay_bounds("T").expect("bounds").head_seq; + assert_eq!( + boundary, 7, + "the completion boundary is the head at completion start" + ); + + // Retention OVERRUNS the handoff cursor mid-handoff: feed far past + // the 60-char ring so the ring front passes seq 5 (the cursor). + for seq in 8..=12 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + let bounds = reg.replay_bounds("T").expect("bounds"); + assert!( + bounds.oldest_retained_seq > cursor + 1, + "the fixture evicted the frames the handoff needs next \ + (front {} vs cursor {cursor})", + bounds.oldest_retained_seq + ); + + // THE plan:146 EXIT: the handoff call must COMPLETE the session — + // TWO exact bounds-carrying gaps sunk in the overrun's completing + // hold (the retention gap for the evicted interval, then the + // delivery gap for the retained window the session will not + // deliver), then the atomic clear. The pre-fix code returned + // `Expired` and resumed the paged chase from the ring front (its + // tests drove that continuation); this assertion is the round-4 + // RED. + let seen_len_at_overrun = seen.lock().unwrap().len(); + let verdict = reg.handoff_paced_tail("T", 1, "paced", cursor, 0); + match verdict { + PacedTailCompletion::Expired { .. } => { + panic!( + "the retention overrun must COMPLETE at the ring front with the \ + gaps recorded — resuming the paged handoff toward an unreachable \ + boundary is the plan:146 violation" + ) + } + PacedTailCompletion::TargetCovered { .. } => { + panic!("the handoff never re-targets: a covered boundary completes") + } + PacedTailCompletion::GapCompleted { + lost_from, + lost_to, + end_seq, + reason, + .. + } => { + assert_eq!( + lost_from, + cursor + 1, + "the retention gap starts at cursor+1" + ); + assert_eq!( + lost_to, + bounds.oldest_retained_seq - 1, + "the retention gap ends just before the new ring front" + ); + assert_eq!( + end_seq, cursor, + "the session's delivered-through boundary stays at the cursor — \ + nothing is swept in the exit hold (plan:145)" + ); + // Round-5 finding 3: the diagnostics carry the mandatory + // gap-exit reason — a genuine retention overrun (the lost + // interval is UNFETCHABLE), never the fetchable + // fixed-boundary residual exit. + assert_eq!( + reason, + PacedGapExitReason::RetentionOverrun, + "the mid-handoff retention overrun's GapCompleted carries the \ + retention_overrun reason (structured diagnostics consumers \ + filter on)" + ); + } + PacedTailCompletion::Completed { .. } | PacedTailCompletion::CaughtUp => { + panic!("the overrun must take the gap-completed exit") + } + other => panic!("the overrun must complete the session, got {other:?}"), + } + + // Both exact gaps reached the subscriber IN ORDER in the + // completing hold — the retention gap for the evicted interval + // first, then the handoff_boundary_reached delivery gap declaring + // the retained window the session will not deliver. + let stream = seen.lock().unwrap().clone(); + let retention_gap_index = stream + .iter() + .position(|m| { + matches!(m, ServerMessage::TerminalOutputGap(g) + if g.from_seq == cursor + 1 + && g.to_seq == bounds.oldest_retained_seq - 1 + && g.reason == TerminalOutputGapReason::ReplayWindowExceeded) + }) + .expect("the retention gap for the evicted interval"); + assert!( + retention_gap_index >= seen_len_at_overrun, + "the retention gap was sunk by the overrun's completing hold" + ); + let delivery_gap_index = stream + .iter() + .position(|m| { + matches!(m, ServerMessage::TerminalOutputGap(g) + if g.from_seq == bounds.oldest_retained_seq + && g.to_seq == bounds.head_seq + && g.reason == TerminalOutputGapReason::HandoffBoundaryReached) + }) + .expect("the delivery gap declaring the retained window"); + assert!( + delivery_gap_index > retention_gap_index, + "the delivery gap follows the retention gap in sink order" + ); + for msg in stream.iter().skip(delivery_gap_index + 1) { + if let ServerMessage::TerminalOutputGap(g) = msg { + assert!( + g.from_seq > cursor, + "no third loss was declared after the overrun exit: {g:?}" + ); + } + } + + // The session is OVER: the deferral cleared (post-clear output + // fans out directly), and a further handoff call is refused. + let seen_len_after = seen.lock().unwrap().len(); + reg.feed("T", frame(bounds.head_seq + 1, "after-clear\r\n", "S")); + assert!( + seen.lock().unwrap().len() > seen_len_after, + "post-clear output fans out directly — the deferral cleared at the \ + ring-front completion" + ); + match reg.handoff_paced_tail("T", 1, "paced", bounds.head_seq, 0) { + PacedTailCompletion::Gone => {} + other => panic!("a completed session refuses further handoff calls, got {other:?}"), + } + + // The declared retained window is FETCHABLE: the client's bounded + // baseline recovery (the repair attach from the coverage cursor) + // is retention-adjusted to the ring front and pages the retained + // window in order — the post-gap frames arrive via the normal + // paced page path, contiguously after the declared loss. + let repair_bounds = reg.replay_bounds("T").expect("bounds"); + let out2 = reg.attach( + "T", + 1, + sink, + Some("paced-repair".into()), + cursor, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let repair = out2.paced.expect("the repair paced session"); + assert_eq!( + repair.session.effective_since, + repair_bounds.oldest_retained_seq - 1, + "the repair's baseline is retention-adjusted to the ring front" + ); + let mut repair_delivered = page_seq_data(&repair.first_page); + let mut repair_cursor = repair.session.page_end; + while repair_cursor < repair.session.target { + match reg.next_replay_page("T", 1, repair_cursor, repair.session.target, 0) { + PacedPage::Frames { + messages, end_seq, .. + } => { + repair_delivered.extend(page_seq_data(&messages)); + repair_cursor = end_seq; + } + other => panic!("unexpected repair replay read: {other:?}"), + } + } + let repair_seqs: Vec = repair_delivered.iter().map(|(s, _)| *s).collect(); + assert_eq!( + repair_seqs, + (repair_bounds.oldest_retained_seq..=repair_bounds.head_seq).collect::>(), + "the repair session pages the declared retained window in order" + ); + } + + /// A superseded drain generation must never touch the new session's + /// deferral (the off-dispatcher drain's generation guard): the drain + /// task runs concurrently with the connection dispatcher, so a + /// re-attach mid-drain replaces the subscriber's attach generation + /// while the OLD drain is still paging. The old drain's page reads + /// must come back GONE for the superseded generation — an unguarded + /// read would page with the NEW generation's stamp and, worse, its + /// "drained" verdict would CLEAR the new session's deferral, + /// letting live output overtake the new session's pages. + #[test] + fn superseded_drain_generation_never_clears_the_new_sessions_deferral() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(0); // per-frame pages: deterministic control + reg.insert_headless("T", "S"); + for seq in 1..=4 { + reg.feed("T", frame(seq, &format!("data-{seq:03}\r\n"), "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink.clone(), + Some("arid-gen-1".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + assert_eq!(start.session.page_end, 1, "per-frame pages: one frame"); + + // The old generation's drain reaches the current head (a caught-up + // call would clear the deferral — for ITS OWN generation). + let old_cursor = 4; + // Production continues; then generation 2 re-attaches mid-drain + // (the registry replaces the subscriber: a NEW attach generation + // stamped on the SAME (connection, terminal), deferral re-armed, + // its own first page delivered). + for seq in 5..=8 { + reg.feed("T", frame(seq, &format!("more-{seq:03}\r\n"), "S")); + } + let out2 = reg.attach( + "T", + 1, + sink, + Some("arid-gen-2".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start2 = out2.paced.expect("paced session 2"); + assert_eq!( + start2.session.page_end, 1, + "gen-2's own first page restarts at the baseline" + ); + let seen_len_after_reattach = seen.lock().unwrap().len(); + + // The OLD drain's next call (still mid-flight off-dispatch): + // against the CURRENT 5-arg API this reads the subscriber with NO + // generation guard — the drain this test pins must refuse to + // touch the superseded subscriber. + let verdict = reg.complete_paced_tail("T", 1, "arid-gen-1", old_cursor, old_cursor, 1024); + match verdict { + PacedTailCompletion::Gone => { + // The guarded behavior: the superseded generation's drain + // is cancelled without delivering or clearing anything. + } + PacedTailCompletion::CaughtUp | PacedTailCompletion::Completed { .. } => { + panic!( + "the superseded drain's verdict cleared gen-2's deferral — \ + live output would overtake gen-2's pages" + ) + } + other => { + panic!("the superseded drain must be refused, got {other:?}"); + } + } + assert_eq!( + seen.lock().unwrap().len(), + seen_len_after_reattach, + "the superseded drain delivers NOTHING" + ); + + // Gen-2's deferral is still armed: new production STAGES (no + // direct fan-out) while gen-2's session is mid-replay. + reg.feed("T", frame(9, "while-deferred\r\n", "S")); + assert!( + !seen + .lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutput(o) if o.seq_end == 9)), + "gen-2's deferral stays armed — the frame stages, it does not fan out" + ); + + // Gen-2's own session pages the full retained window in order — + // including the staged frame 9 — and completes through the + // arid-guarded reads. + let mut cursor = start2.session.page_end; + let target2 = start2.session.target; + let mut guards = 0; + let mut handing_off = false; + loop { + guards += 1; + assert!(guards <= 64, "gen-2 drains in bounded rounds"); + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "arid-gen-2", cursor, 1024) + } else { + reg.complete_paced_tail("T", 1, "arid-gen-2", cursor, target2, 1024) + }; + match verdict { + PacedTailCompletion::Handoff { end_seq, .. } => cursor = end_seq, + PacedTailCompletion::GapCompleted { .. } => { + panic!("no mid-handoff retention loss in this fixture") + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // Fixed target covered with a staged remainder beyond: + // the bounded post-target handoff owns it. + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::Completed { end_seq, .. } => { + cursor = cursor.max(end_seq); + break; + } + PacedTailCompletion::CaughtUp => break, + PacedTailCompletion::Expired { .. } => panic!("no retention loss here"), + PacedTailCompletion::Gone => panic!("gen-2 is the live generation"), + } + } + assert_eq!(cursor, 9, "gen-2 drains through the staged frame"); + reg.feed("T", frame(10, "after-clear\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutput(o) if o.seq_end == 10)), + "post-clear output fans out directly after gen-2's completion" + ); + } + + /// Round-2 finding F2 — the DEFENSE-IN-DEPTH pin: a single frame whose + /// own envelope exceeds the page budget still forms its own atomic + /// single-frame page (guaranteed progress, never split, never silently + /// coalesced into a "budget" page). UNREACHABLE under supported + /// settings: every PTY byte is ingested through the fragment splitter + /// whose cap is clamped to [`crate::fragment::PACED_PAGE_BUDGET_FLOOR_BYTES`] + /// (the smallest page budget any supported queue setting can produce), + /// so a production frame can never exceed the page budget — this arm + /// exists exactly for the unsupported residue (a test-injected budget + /// below the floor, a future ingest path that bypasses the splitter) + /// and must never be removed. + #[test] + fn paced_replay_oversized_frame_atomic_page_is_defense_in_depth_beyond_the_boot_clamp() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(256); + reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "tiny-a\r\n", "S")); + reg.feed("T", frame(2, &"X".repeat(2048), "S")); + reg.feed("T", frame(3, "tiny-b\r\n", "S")); + + let (sink, _seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + assert_eq!( + start.session.page_end, 1, + "the first page stops before the oversize frame" + ); + + match reg.next_replay_page("T", 1, 1, 3, 256) { + PacedPage::Frames { + messages, + end_seq, + serialized_bytes, + } => { + assert_eq!(end_seq, 2); + assert_eq!( + messages.len(), + 1, + "the oversize frame is ONE atomic message" + ); + assert!( + serialized_bytes as usize > 256, + "the oversize frame honestly exceeds the budget as its own page" + ); + match &messages[0] { + ServerMessage::TerminalOutput(o) => { + assert_eq!(o.seq_start, 2); + assert_eq!(o.data.len(), 2048); + } + other => panic!("per-frame subscriber gets terminal.output: {other:?}"), + } + } + other => panic!("expected the atomic oversize page, got {other:?}"), + } + match reg.next_replay_page("T", 1, 2, 3, 256) { + PacedPage::Frames { + messages, end_seq, .. + } => { + assert_eq!(end_seq, 3); + assert_eq!( + page_seq_data(&messages), + vec![(3, "tiny-b\r\n".to_string())] + ); + } + other => panic!("expected the final page, got {other:?}"), + } + } + + /// Retention expiry mid-replay: the exact lost interval and the resume + /// position, both consistent with the ring's live bounds. + #[test] + fn paced_replay_expired_reports_the_exact_interval_and_resume() { + let reg = TerminalRegistry::new(); + // Tiny CHAR ring so feeding evicts the front deterministically. + reg.set_scrollback_max_bytes(60); + reg.insert_headless("T", "S"); + reg.set_paced_page_max_bytes(0); // per-frame pages: deterministic cursor control + for seq in 1..=3 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); // 10 chars each + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + assert_eq!(start.session.page_end, 1, "budget 0 => one frame per page"); + + // Evict frames 2..: 10 more chunks (100 chars) pushes the front well + // past the session cursor. + for seq in 4..=13 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + let bounds = reg.replay_bounds("T").expect("bounds"); + assert!( + bounds.oldest_retained_seq > 2, + "the fixture evicted the frames the session needs next" + ); + + match reg.next_replay_page("T", 1, 1, bounds.head_seq, 0) { + PacedPage::Expired { + lost_from, + lost_to, + resume_from, + head_seq, + oldest_retained_seq, + } => { + assert_eq!(lost_from, 2, "the lost interval starts at cursor+1"); + assert_eq!( + lost_to, + bounds.oldest_retained_seq - 1, + "the lost interval ends just before the new ring front" + ); + assert_eq!(resume_from, bounds.oldest_retained_seq - 1); + assert_eq!(head_seq, bounds.head_seq); + assert_eq!(oldest_retained_seq, bounds.oldest_retained_seq); + } + other => panic!("expected Expired, got {other:?}"), + } + + // Continuation from the new baseline: the drain pages everything + // the ring still holds toward the FIXED drain target (quiet + // terminal — the head), then the completing verdict clears the + // deferral. + let drain_target = reg.replay_bounds("T").expect("bounds").head_seq; + let mut tail_cursor = bounds.oldest_retained_seq - 1; + let mut rounds = 0; + loop { + rounds += 1; + assert!(rounds <= 64, "the quiet drain terminates in bounded rounds"); + match reg.complete_paced_tail("T", 1, "paced", tail_cursor, drain_target, 0) { + PacedTailCompletion::CaughtUp => break, + PacedTailCompletion::Completed { end_seq, .. } => { + tail_cursor = tail_cursor.max(end_seq); + break; + } + PacedTailCompletion::Handoff { end_seq, .. } => { + assert!(end_seq <= drain_target, "pages never pass the fixed target"); + tail_cursor = end_seq; + } + other => panic!("a quiet terminal completes its drain, got {other:?}"), + } + } + assert_eq!(tail_cursor, bounds.head_seq); + let drained_seqs: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) if o.seq_start >= bounds.oldest_retained_seq => { + Some(o.seq_start) + } + _ => None, + }) + .collect(); + assert_eq!( + drained_seqs, + (bounds.oldest_retained_seq..=bounds.head_seq).collect::>(), + "the continuation covers exactly the retained range" + ); + } + + /// While a paced session is active the subscriber's live output is NOT + /// sunk by ingest (the ring is the staging); after the session catches + /// up (deferral cleared under the tail read's lock), ingest resumes + /// direct delivery. Concurrent production across the flag-clear + /// boundary is delivered exactly once — paged or direct, never both, + /// never neither. + #[test] + fn paced_deferral_stages_live_output_and_resumes_direct_delivery_exactly_once() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(0); // per-frame pages + reg.insert_headless("T", "S"); + for seq in 1..=3 { + reg.feed("T", frame(seq, "early-1\r\n", "S")); + } + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out.paced.expect("paced session"); + assert_eq!(start.session.page_end, 1); + + // Live output while the session is active: staged in the ring, NOT + // sunk. + reg.feed("T", frame(4, "live-04\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .all(|m| !matches!(m, ServerMessage::TerminalOutput(_))), + "deferred subscriber receives nothing inline" + ); + + // Concurrent production racing the tail drain. + let feeder_reg = reg.clone(); + let feeder = std::thread::spawn(move || { + for seq in 5..=40 { + feeder_reg.feed("T", frame(seq, &format!("race-{seq:02}\r\n"), "S")); + std::thread::sleep(std::time::Duration::from_millis(1)); + } + }); + + // Drain: replay pages to the target (returned to this caller), then + // the paged drain toward the FIXED drain target (the head at drain + // start — the feeder races it), then the fixed-target atomic live + // handoff (drained clear or re-fan; the feeder's staged frames are + // delivered through the subscriber's sink, recorded in `seen` + // alongside the direct deliveries below). + let mut collected = page_seq_data(&start.first_page); + let mut cursor = start.session.page_end; + while cursor < start.session.target { + match reg.next_replay_page("T", 1, cursor, start.session.target, 0) { + PacedPage::Frames { + messages, end_seq, .. + } => { + collected.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + let drain_target = reg.replay_bounds("T").expect("bounds").head_seq; + let mut rounds = 0; + let mut handing_off = false; + loop { + rounds += 1; + assert!( + rounds <= 128, + "the racing drain terminates in bounded rounds" + ); + let verdict = if handing_off { + reg.handoff_paced_tail("T", 1, "paced", cursor, 0) + } else { + reg.complete_paced_tail("T", 1, "paced", cursor, drain_target, 0) + }; + match verdict { + PacedTailCompletion::CaughtUp | PacedTailCompletion::Completed { .. } => break, + PacedTailCompletion::Handoff { end_seq, .. } => { + if !handing_off { + assert!(end_seq <= drain_target, "pages never pass the fixed target"); + } + cursor = end_seq; + } + PacedTailCompletion::GapCompleted { .. } => { + panic!("no retention loss in this fixture (feeder max 41 frames)") + } + PacedTailCompletion::TargetCovered { end_seq, .. } => { + // Fixed target covered with the feeder still staging: + // the bounded post-target handoff delivers the racing + // remainder one frame-chunk per lock hold. + cursor = end_seq; + handing_off = true; + } + PacedTailCompletion::Expired { .. } => { + panic!("no retention loss in this fixture (feeder max 41 frames)") + } + PacedTailCompletion::Gone => panic!("terminal vanished at completion"), + } + } + feeder.join().expect("feeder joins"); + + // Post-clear production flows DIRECTLY through ingest again. The + // feeder's tail may have landed either side of the clear (any split + // is valid); frame 41 is fed strictly AFTER the drain, so it must be + // a DIRECT delivery. + reg.feed("T", frame(41, "after-41\r\n", "S")); + let direct: Vec<(i64, String)> = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) => Some((o.seq_start, o.data.clone())), + _ => None, + }) + .collect(); + assert_eq!( + direct.last(), + Some(&(41, "after-41\r\n".to_string())), + "post-clear output is delivered directly by ingest: {direct:?}" + ); + + // The no-loss/no-dup invariant across the flag-clear boundary: + // pages + direct deliveries together are exactly frames 1..=41, once. + let mut all: Vec<(i64, String)> = collected; + all.extend(direct); + all.sort_by_key(|(s, _)| *s); + let seqs: Vec = all.iter().map(|(s, _)| *s).collect(); + assert_eq!( + seqs, + (1..=41).collect::>(), + "every produced frame delivered exactly once, in seq order" + ); + assert_eq!(all.iter().map(|(_, d)| d.clone()).collect::(), { + let mut s = String::new(); + for _seq in 1..=3 { + s.push_str("early-1\r\n"); + } + s.push_str("live-04\r\n"); + for seq in 5..=40 { + s.push_str(&format!("race-{seq:02}\r\n")); + } + s.push_str("after-41\r\n"); + s + }); + } + + /// A re-attach replaces the subscriber — an armed paced deferral does not + /// survive into the new subscription unless the new attach arms its own. + #[test] + fn reattach_replaces_the_paced_deferral() { + let reg = TerminalRegistry::new(); + reg.set_paced_page_max_bytes(0); + reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "one\r\n", "S")); + + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink.clone(), + Some("paced-a".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!(out.paced.is_some(), "the paced attach arms the deferral"); + reg.feed("T", frame(2, "two\r\n", "S")); + assert!( + seen.lock() + .unwrap() + .iter() + .all(|m| !matches!(m, ServerMessage::TerminalOutput(_))), + "staged while the first session is active" + ); + + // A non-paced re-attach (the fallback shape) cancels the session's + // deferral: the legacy re-attach replays the window INLINE (frames 1 + // and 2 — including the one staged while deferred), and the NEXT + // produced frame flows directly by ingest, with no pages to drive. + let out2 = reg.attach( + "T", + 1, + sink.clone(), + Some("legacy-b".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!( + out2.paced.is_none(), + "a non-negotiated re-attach stays legacy" + ); + reg.feed("T", frame(3, "three\r\n", "S")); + let live: Vec = seen + .lock() + .unwrap() + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(o) if o.source == Some(OutputSource::Live) => { + Some(o.data.clone()) + } + _ => None, + }) + .collect(); + assert_eq!( + live, + vec!["three\r\n".to_string()], + "after the deferral clears, live output flows directly" + ); + + // A paced re-attach arms a FRESH session whose first page includes + // everything staged since its own baseline. + let out3 = reg.attach( + "T", + 1, + sink.clone(), + Some("paced-c".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let start = out3 + .paced + .expect("the paced re-attach starts a fresh session"); + assert_eq!(start.session.attach_request_id, "paced-c"); + assert_eq!(start.session.target, 3); + // Budget 0 => per-frame pages: drive the fresh session to completion + // and assert it pages the WHOLE window (including frame 2, staged + // while the first session was deferred). + let mut paged: Vec<(i64, String)> = page_seq_data(&start.first_page); + let mut cursor = start.session.page_end; + while cursor < start.session.target { + match reg.next_replay_page("T", 1, cursor, start.session.target, 0) { + PacedPage::Frames { + messages, end_seq, .. + } => { + paged.extend(page_seq_data(&messages)); + cursor = end_seq; + } + other => panic!("unexpected replay read: {other:?}"), + } + } + let mut paged_seqs: Vec = paged.iter().map(|(s, _)| *s).collect(); + paged_seqs.sort_unstable(); + assert_eq!( + paged_seqs, + vec![1, 2, 3], + "the fresh session pages the whole window" + ); + } + + /// Retention loss AT ATTACH (requested since predates the retained + /// ring): the negotiated connection gets the retention gap with the + /// task-2 bounds fields, `replayResetReason: retention_lost`, and an + /// effective baseline of `oldest-1` — the session continues from what + /// is retained (nothing is killed, nothing stalls). + #[test] + fn paced_attach_with_retention_loss_emits_the_negotiated_gap_and_resets_the_baseline() { + let reg = TerminalRegistry::new(); + reg.set_scrollback_max_bytes(60); + reg.insert_headless("T", "S"); + for seq in 1..=13 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + let bounds = reg.replay_bounds("T").expect("bounds"); + assert!( + bounds.oldest_retained_seq > 1, + "the front has evicted past seq 1" + ); + + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!(out.found); + let start = out + .paced + .expect("paced session continues from what is retained"); + assert_eq!( + start.session.effective_since, + bounds.oldest_retained_seq - 1, + "the baseline resets to oldest-1" + ); + assert_eq!(start.session.target, bounds.head_seq); + let first = page_seq_data(&start.first_page); + assert_eq!( + first.first().unwrap().0, + bounds.oldest_retained_seq, + "the first page starts at the retained front" + ); + + let ready = attach_ready(&seen).expect("attach.ready sent"); + assert_eq!( + ready.replay_reset_reason, + Some(TerminalReplayResetReason::RetentionLost), + "the ready frame names the retention reset" + ); + assert_eq!(ready.oldest_retained_seq, Some(bounds.oldest_retained_seq)); + assert_eq!( + ready.effective_since_seq, + Some(bounds.oldest_retained_seq - 1) + ); + assert_eq!(ready.requested_since_seq, Some(0)); + assert_eq!(ready.replay_from_seq, bounds.oldest_retained_seq); + assert_eq!(ready.replay_to_seq, bounds.head_seq); + + let gap = seen + .lock() + .unwrap() + .iter() + .find_map(|m| match m { + ServerMessage::TerminalOutputGap(g) => Some(g.clone()), + _ => None, + }) + .expect("the retention gap is sunk with the ready prelude"); + assert_eq!( + gap.reason, + freshell_protocol::TerminalOutputGapReason::ReplayWindowExceeded + ); + assert_eq!( + gap.from_seq, 1, + "the lost interval starts at the requested baseline+1" + ); + assert_eq!(gap.to_seq, bounds.oldest_retained_seq - 1); + assert_eq!(gap.attach_request_id.as_deref(), Some("paced")); + assert_eq!(gap.head_seq, Some(bounds.head_seq)); + assert_eq!(gap.oldest_retained_seq, Some(bounds.oldest_retained_seq)); + } + + /// The compatibility twin: a NON-negotiated attach with the same + /// retention loss keeps today's silent behavior — no gap frame, no reset + /// reason, no retention bounds, replay silently starting at the ring + /// front. + #[test] + fn plain_attach_with_retention_loss_stays_silent_and_byte_identical() { + let reg = TerminalRegistry::new(); + reg.set_scrollback_max_bytes(60); + reg.insert_headless("T", "S"); + for seq in 1..=13 { + reg.feed("T", frame(seq, "chunk123\r\n", "S")); + } + let bounds = reg.replay_bounds("T").expect("bounds"); + + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("plain".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!( + out.paced.is_none(), + "non-negotiated never starts a paced session" + ); + let ready = attach_ready(&seen).expect("attach.ready sent"); + assert_eq!(ready.replay_reset_reason, None); + assert_eq!(ready.oldest_retained_seq, None); + assert_eq!(ready.replay_from_seq, bounds.oldest_retained_seq); + let frames = outputs(&seen); + assert_eq!( + frames.first().map(|f| f.seq_start), + Some(bounds.oldest_retained_seq), + "legacy replay silently starts at the retained front" + ); + assert!( + !seen + .lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalOutputGap(_))), + "no gap frame for a non-negotiated connection" + ); + } + + /// A negotiated attach to an ALREADY-EXITED terminal keeps the legacy + /// inline path (frozen-tail replay + synthetic exit, in their legacy + /// order): pacing arms only for Running terminals this increment. + #[test] + fn paced_negotiation_on_an_exited_terminal_keeps_the_legacy_inline_path() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "tail\r\n", "S")); + assert!(reg.finish_pty_exit("T", 0)); + + let (sink, seen) = collector(); + let out = reg.attach( + "T", + 1, + sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!( + out.paced.is_none(), + "exited terminals stay on the legacy path" + ); + let frames = outputs(&seen); + assert_eq!(frames.len(), 1, "the frozen tail replays inline"); + assert_eq!(frames[0].source, Some(OutputSource::Replay)); + assert!( + seen.lock() + .unwrap() + .iter() + .any(|m| matches!(m, ServerMessage::TerminalExit(_))), + "the synthetic exit follows the inline replay" + ); + } + + /// TERM-07 seam: the attach-threaded `maxReplayBytes` is recorded on the + /// subscriber with NO delivery-behavior change (it stays unread for + /// delivery decisions this increment; the increment-3 snapshot work + /// consumes it). + #[test] + fn attach_records_max_replay_bytes_on_the_subscriber() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "one\r\n", "S")); + let (sink, _seen) = collector(); + let _ = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + Some(128 * 1024), + PacedAttachOptions::default(), + ); + let recorded = { + let inner = reg.inner.lock().unwrap(); + let handle = inner.terminals.get("T").unwrap(); + let s = handle.shared.lock().unwrap(); + s.subscribers + .get(&1) + .expect("subscriber installed") + .max_replay_bytes + }; + assert_eq!(recorded, Some(128 * 1024)); + + // Absent stays absent; the legacy path threads it identically. + let (sink2, _seen2) = collector(); + let _ = reg.attach( + "T", + 2, + sink2, + Some("b".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + let recorded2 = { + let inner = reg.inner.lock().unwrap(); + let handle = inner.terminals.get("T").unwrap(); + let s = handle.shared.lock().unwrap(); + s.subscribers.get(&2).expect("subscriber").max_replay_bytes + }; + assert_eq!(recorded2, None); + } + + #[test] + fn unpaced_attach_ready_pins_the_exact_wire_keys_for_both_negotiation_sides() { + // Compatibility invariant (load-bearing): a connection that did NOT + // negotiate sees a ready frame byte-identical to the pre-contract + // shape — `oldestRetainedSeq` is ABSENT from the wire (not null), and + // `replayResetReason` is still absent. The negotiated side adds + // exactly one key: `oldestRetainedSeq`. + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + reg.feed("T", frame(1, "one\r\n", "S")); + + let (plain_sink, plain_seen) = collector(); + let _ = reg.attach( + "T", + 1, + plain_sink, + Some("plain".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + let plain_ready = attach_ready(&plain_seen).expect("attach.ready sent"); + assert_eq!(plain_ready.oldest_retained_seq, None); + let json = serde_json::to_value(ServerMessage::TerminalAttachReady(plain_ready)).unwrap(); + let mut keys: Vec<&str> = json + .as_object() + .unwrap() + .keys() + .map(String::as_str) + .collect(); + keys.sort_unstable(); + assert_eq!( + keys, + vec![ + "attachRequestId", + "effectiveSinceSeq", + "geometryAuthority", + "geometryEpoch", + "headSeq", + "replayFromSeq", + "replayToSeq", + "requestedSinceSeq", + "streamId", + "terminalId", + "type", + ], + "the non-negotiated ready frame keeps the pre-contract key set: {json}" + ); + + let (paced_sink, paced_seen) = collector(); + let _ = reg.attach( + "T", + 2, + paced_sink, + Some("paced".into()), + 0, + false, + true, + None, + None, + None, + PacedAttachOptions::default(), + ); + let paced_ready = attach_ready(&paced_seen).expect("attach.ready sent"); + assert_eq!(paced_ready.oldest_retained_seq, Some(1)); + let json = serde_json::to_value(ServerMessage::TerminalAttachReady(paced_ready)).unwrap(); + let mut keys: Vec<&str> = json + .as_object() + .unwrap() + .keys() + .map(String::as_str) + .collect(); + keys.sort_unstable(); + assert_eq!( + keys, + vec![ + "attachRequestId", + "effectiveSinceSeq", + "geometryAuthority", + "geometryEpoch", + "headSeq", + "oldestRetainedSeq", + "replayFromSeq", + "replayToSeq", + "requestedSinceSeq", + "streamId", + "terminalId", + "type", + ], + "the negotiated ready frame adds exactly oldestRetainedSeq: {json}" + ); + } + + #[test] + fn detach_keeps_terminal_running_and_buffering_then_replays_on_reattach() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + + let (sink_a, seen_a) = collector(); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + reg.feed("T", frame(1, "before\r\n", "S")); + assert_eq!(outputs(&seen_a).len(), 1); + + // Detach: subscription gone, but the terminal keeps running + buffering. + reg.detach("T", 1); + assert!( + reg.is_running("T"), + "terminal survives detach (background session)" + ); + reg.feed("T", frame(2, "while-detached\r\n", "S")); + // The detached connection receives nothing more. + assert_eq!(outputs(&seen_a).len(), 1); + + // A fresh attach replays the FULL scrollback (both frames). + let (sink_b, seen_b) = collector(); + let _ = reg.attach( + "T", + 2, + sink_b, + Some("b".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + let replayed = outputs(&seen_b); + assert_eq!( + replayed.iter().map(|f| f.data.as_str()).collect::>(), + vec!["before\r\n", "while-detached\r\n"] + ); + } + + #[test] + fn two_attached_sockets_both_get_live_output_each_with_its_own_attach_id() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + + let (sink_a, seen_a) = collector(); + let (sink_b, seen_b) = collector(); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("aaa".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + // Second attach: geometry authority flips to multi_client_unknown. + let _ = reg.attach( + "T", + 2, + sink_b, + Some("bbb".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + let ready_b = attach_ready(&seen_b).unwrap(); + assert_eq!( + ready_b.geometry_authority, + Some(GeometryAuthority::MultiClientUnknown) + ); + + // One live frame fans out to BOTH sockets, each stamped with its own id. + reg.feed("T", frame(1, "shared\r\n", "S")); + let a = outputs(&seen_a); + let b = outputs(&seen_b); + assert_eq!(a.len(), 1); + assert_eq!(b.len(), 1); + assert_eq!(a[0].data, "shared\r\n"); + assert_eq!(b[0].data, "shared\r\n"); + assert_eq!(a[0].attach_request_id.as_deref(), Some("aaa")); + assert_eq!(b[0].attach_request_id.as_deref(), Some("bbb")); + } + + #[test] + fn reconnect_catches_up_by_seq_only_replaying_newer_frames() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + let (sink_a, seen_a) = collector(); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + for i in 1..=5 { + reg.feed("T", frame(i, &format!("line-{i}\r\n"), "S")); + } + assert_eq!(outputs(&seen_a).len(), 5); + + // Reconnect: the client already rendered through seq 3, so it re-attaches + // with sinceSeq=3. Only frames 4 and 5 are replayed (seqStart > 3). reg.detach("T", 1); let (sink_r, seen_r) = collector(); - let _ = reg.attach("T", 2, sink_r, Some("a2".into()), 3, false, None, None); + let _ = reg.attach( + "T", + 2, + sink_r, + Some("a2".into()), + 3, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let ready = attach_ready(&seen_r).unwrap(); assert_eq!(ready.effective_since_seq, Some(3)); assert_eq!(ready.replay_from_seq, 4); @@ -4972,7 +10705,19 @@ mod tests { reg.feed("T", frame(1, "old\r\n", "S")); let (sink, seen) = collector(); - let _ = reg.attach("T", 7, sink, Some("z".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 7, + sink, + Some("z".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); // A live frame produced AFTER attach must arrive after the replayed one. reg.feed("T", frame(2, "new\r\n", "S")); @@ -4994,7 +10739,19 @@ mod tests { fn attach_to_unknown_terminal_reports_not_found() { let reg = TerminalRegistry::new(); let (sink, seen) = collector(); - let out = reg.attach("nope", 1, sink, None, 0, false, None, None); + let out = reg.attach( + "nope", + 1, + sink, + None, + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(!out.found); assert!(seen.lock().unwrap().is_empty()); } @@ -5024,7 +10781,19 @@ mod tests { reg.insert_headless("T", "S"); let rev_before = reg.revision(); let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(reg.kill("T")); assert!(!reg.is_running("T"), "killed terminal is removed"); @@ -5052,8 +10821,32 @@ mod tests { reg.insert_headless("T-b", "S2"); let (sink_a, seen_a) = collector(); let (sink_b, seen_b) = collector(); - let _ = reg.attach("T-a", 1, sink_a, None, 0, false, None, None); - let _ = reg.attach("T-b", 2, sink_b, None, 0, false, None, None); + let _ = reg.attach( + "T-a", + 1, + sink_a, + None, + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = reg.attach( + "T-b", + 2, + sink_b, + None, + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let rev_before = reg.revision(); let killed = reg.kill_all(); @@ -5339,7 +11132,19 @@ mod tests { assert!(!dir[0].has_clients); let (sink, _seen) = collector(); - let _ = reg.attach("T", 9, sink, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 9, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(reg.directory()[0].has_clients); reg.detach("T", 9); assert!(!reg.directory()[0].has_clients); @@ -5457,11 +11262,17 @@ mod tests { Some("a-1".into()), 0, false, + false, None, None, + // TERM-07 seam: the geometry-authorized path threads the + // client's replay-budget request just like the plain + // path (pinned by the subscriber-recording assert below). + Some(256 * 1024), TerminalAttachIntent::ViewportHydrate, 131, 48, + PacedAttachOptions::default(), ) }) }; @@ -5478,11 +11289,14 @@ mod tests { Some("b-1".into()), 0, false, + false, + None, None, None, TerminalAttachIntent::TransportReconnect, 67, 30, + PacedAttachOptions::default(), ) }) }; @@ -5509,6 +11323,31 @@ mod tests { matches!(geometry, (131, 48, 1) | (67, 30, 1)), "the final dimensions must belong to the one successful first claim: {geometry:?}" ); + // TERM-07 seam on the GEOMETRY-AUTHORIZED path (attach_with_geometry + // threads max_replay_bytes into the same attach_to_shared as the + // plain path): each subscriber records its own attach's value — + // a's request verbatim, b's absence as None — with no + // delivery-behavior change. + let (recorded_a, recorded_b) = { + let inner = reg.inner.lock().unwrap(); + let handle = inner.terminals.get("T").unwrap(); + let s = handle.shared.lock().unwrap(); + ( + s.subscribers + .get(&1) + .expect("conn 1 subscriber") + .max_replay_bytes, + s.subscribers + .get(&2) + .expect("conn 2 subscriber") + .max_replay_bytes, + ) + }; + assert_eq!(recorded_a, Some(256 * 1024)); + assert_eq!( + recorded_b, None, + "an attach without maxReplayBytes records absence" + ); } #[test] @@ -5522,7 +11361,19 @@ mod tests { assert_eq!(out, AttachResizeStatus::Resized); assert_eq!(reg.geometry("T"), Some((131, 48, 1))); let (sink_a, _seen_a) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("a-1".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("a-1".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); // A secondary viewer must be able to attach without silently taking // over the shared terminal's geometry. @@ -5530,7 +11381,19 @@ mod tests { assert_eq!(out, AttachResizeStatus::Skipped); assert_eq!(reg.geometry("T"), Some((131, 48, 1))); let (sink_b, _seen_b) = collector(); - let _ = reg.attach("T", 2, sink_b, Some("b-1".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 2, + sink_b, + Some("b-1".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); // Later attach generations from that same second socket are still // replay operations, not implicit geometry transfers. @@ -5593,8 +11456,20 @@ mod tests { let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); let (sink, _seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); // conn 1 is attached - // conn 2 reconnects with another socket attached and no prior attachment of its own. + let _ = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); // conn 1 is attached + // conn 2 reconnects with another socket attached and no prior attachment of its own. let out = reg.resize_for_attach("T", 2, TerminalAttachIntent::TransportReconnect, 95, 41); assert_eq!(out, AttachResizeStatus::Skipped); assert_eq!(reg.geometry("T"), Some((120, 30, 1))); @@ -5605,7 +11480,19 @@ mod tests { let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); let (sink_a, _seen_a) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); // The first transport reconnect from B is replay-only while A views // the terminal. Register B so its second generation exercises the @@ -5613,7 +11500,19 @@ mod tests { let out = reg.resize_for_attach("T", 2, TerminalAttachIntent::TransportReconnect, 95, 41); assert_eq!(out, AttachResizeStatus::Skipped); let (sink_b, _seen_b) = collector(); - let _ = reg.attach("T", 2, sink_b, Some("b".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 2, + sink_b, + Some("b".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let out = reg.resize_for_attach("T", 2, TerminalAttachIntent::TransportReconnect, 95, 41); assert_eq!(out, AttachResizeStatus::Skipped); @@ -5652,8 +11551,32 @@ mod tests { reg.insert_headless("T2", "S2"); let (sink1, seen1) = collector(); let (sink2, seen2) = collector(); - let _ = reg.attach("T1", 42, sink1, Some("a".into()), 0, false, None, None); - let _ = reg.attach("T2", 42, sink2, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T1", + 42, + sink1, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + let _ = reg.attach( + "T2", + 42, + sink2, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); reg.remove_connection(42); // Both terminals survive; the swept connection receives no further output. @@ -5679,7 +11602,19 @@ mod tests { assert!(reg.finish_pty_exit("T", 7)); let (sink, seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); + let outcome = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); let exit = seen.lock().unwrap().iter().find_map(|m| match m { @@ -5730,8 +11665,11 @@ mod tests { Some("att-1".into()), 0, false, + false, None, Some(true), + None, + PacedAttachOptions::default(), ); assert!(outcome.found); @@ -5772,7 +11710,19 @@ mod tests { reg.feed("T", frame(2, "banner\r\n", "S")); let (sink, seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("b1".into()), 0, true, None, Some(true)); + let outcome = reg.attach( + "T", + 1, + sink, + Some("b1".into()), + 0, + true, + false, + None, + Some(true), + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); let msgs = seen.lock().unwrap().clone(); @@ -5802,7 +11752,19 @@ mod tests { // Flag absent (None) … let (sink_a, seen_a) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); // … and explicitly false: no sync either way (fixture f14's gating). let (sink_b, seen_b) = collector(); let _ = reg.attach( @@ -5812,8 +11774,11 @@ mod tests { Some("b".into()), 0, false, + false, None, Some(false), + None, + PacedAttachOptions::default(), ); assert!(modes_syncs(&seen_a).is_empty(), "flag absent => no sync"); @@ -5829,7 +11794,19 @@ mod tests { reg.feed("T", frame(1, "just plain text\r\n", "S")); let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, Some(true)); + let _ = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + Some(true), + None, + PacedAttachOptions::default(), + ); assert!(modes_syncs(&seen).is_empty(), "empty synthesis => no sync"); } @@ -5842,7 +11819,19 @@ mod tests { // The client fails closed on a sync lacking attachRequestId // (`missing_attach_request_id`), so the server never builds one. let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, None, 0, false, None, Some(true)); + let _ = reg.attach( + "T", + 1, + sink, + None, + 0, + false, + false, + None, + Some(true), + None, + PacedAttachOptions::default(), + ); assert!( modes_syncs(&seen).is_empty(), "no attachRequestId => no sync" @@ -5860,7 +11849,19 @@ mod tests { assert!(reg.finish_pty_exit("T", 3)); let (sink, seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, Some(true)); + let outcome = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + Some(true), + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); let msgs = seen.lock().unwrap().clone(); @@ -5935,7 +11936,19 @@ mod tests { let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); let (sink, _seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); + let outcome = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); reg.set_auto_kill_idle_minutes(1); // Far past any threshold, but a client is attached -- legacy: @@ -6111,99 +12124,354 @@ mod tests { // terminal-core.md A13: "terminal stays running"). let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); - let (sink, _seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); - assert!(outcome.found); + let (sink, _seen) = collector(); + let outcome = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!(outcome.found); + reg.set_auto_kill_idle_minutes(1); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + reg.detach("T", 1); + + // Freshly detached: the transition bump spares it a full threshold. + assert!(reg.enforce_idle_kills().is_empty()); + assert_eq!(reg.inventory().len(), 1); + + // Once it goes stale again AFTER the detach, it is reaped normally. + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + assert_eq!(reg.enforce_idle_kills(), vec!["T".to_string()]); + } + + #[test] + fn disconnect_grants_full_idle_threshold_of_grace() { + // Socket-close cleanup (remove_connection) is the other live + // transition-to-detached path and must grant the same grace as an + // explicit detach. The bump is gated on "this connection actually + // subscribed here AND the set became empty AND status is Running" — + // remove_connection iterates EVERY terminal, and an unconditional + // bump would reset unrelated detached terminals' countdowns on + // every socket close. + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + let (sink, _seen) = collector(); + let outcome = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); + assert!(outcome.found); + // A second, already-detached terminal whose countdown must NOT be + // disturbed by conn 1's disconnect. + reg.insert_headless("U", "S"); + reg.set_auto_kill_idle_minutes(1); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + reg.backdate_last_activity("U", now_ms() - 10 * 60_000); + reg.remove_connection(1); + + let killed = reg.enforce_idle_kills(); + + // T was freshly detached by the disconnect => spared one threshold. + // U never had a subscriber => its stale countdown stands => reaped. + assert_eq!(killed, vec!["U".to_string()]); + assert_eq!(reg.inventory().len(), 1); + } + + #[test] + fn enforce_idle_kills_spares_disconnect_detached_terminal_past_threshold() { + // The background-session promise (§1.3, "what if tmux and Claude fell + // in love"): a user who closes the browser / sleeps the laptop with a + // quiet shell in a pane must find it ALIVE when they come back — the + // client never sent terminal.detach, so the terminal is detached but + // still wanted. Before the fix, 20 idle minutes past the default + // 15-minute autoKillIdleMinutes threshold silently SIGKILLed it + // (scrollback gone). Now only the 24h hard cap applies. + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + let (sink, _seen) = collector(); + assert!( + reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found + ); + reg.set_auto_kill_idle_minutes(15); // the shipped default + reg.remove_connection(1); // socket drop, NOT an explicit detach + // 20 minutes idle — past the old ~15-minute threshold, far under 24h. + reg.backdate_last_activity("T", now_ms() - 20 * 60_000); + + let killed = reg.enforce_idle_kills(); + + assert!( + killed.is_empty(), + "a detached-but-wanted terminal (connection lost, never released) \ + must survive past the configured idle threshold, got {killed:?}" + ); + assert_eq!(reg.inventory().len(), 1); + } + + #[test] + fn enforce_idle_kills_reaps_disconnect_detached_terminal_past_the_hard_cap() { + // Counterweight: the sweep stays the cleanup backstop for sessions + // whose client never comes back — same 24h hard cap agents get. + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + let (sink, _seen) = collector(); + assert!( + reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found + ); + reg.set_auto_kill_idle_minutes(15); + reg.remove_connection(1); + // 25 hours stale — past the 24-hour hard cap. + reg.backdate_last_activity("T", now_ms() - 25 * 60 * 60_000); + + assert_eq!(reg.enforce_idle_kills(), vec!["T".to_string()]); + } + + // ── Hidden-pane lifetime claims (responsive-terminal-restore WS1) ── + + /// A claim marks the terminal wanted for a connection WITHOUT attaching: + /// it clears `released_by_client` under the terminal lock, adds no + /// subscriber, and delivers no replay (there is no sink to deliver to). + #[test] + fn claim_clears_released_by_client_without_attaching() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + // A never-attached row starts fast-reap eligible (released == true). + assert_eq!( + reg.claim_state("T"), + Some(ClaimState { + claimers: 0, + released_by_client: true + }) + ); + + assert!(reg.claim_terminal("T", 7)); + + let state = reg.claim_state("T").expect("terminal exists"); + assert_eq!(state.claimers, 1); + assert!( + !state.released_by_client, + "a claim must clear released_by_client (the terminal is wanted)" + ); + // No attach happened: the row still has zero subscribers (hasClients + // stays false) — the claim never granted replay or output delivery. + assert!(!reg.directory()[0].has_clients); + // Unknown ids are a no-op (stale client layouts may claim dead rows). + assert!(!reg.claim_terminal("nope", 7)); + } + + /// Explicit withdrawal of the last claim (no subscribers left) restores + /// configured-threshold reapability with the same fresh-idle-grace bump + /// detach uses (DEV-0009): spared immediately, reaped once stale again. + #[test] + fn explicit_withdrawal_of_last_claim_restores_threshold_eligibility_with_grace() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + reg.claim_terminal("T", 7); reg.set_auto_kill_idle_minutes(1); reg.backdate_last_activity("T", now_ms() - 10 * 60_000); - reg.detach("T", 1); - // Freshly detached: the transition bump spares it a full threshold. + reg.withdraw_claim("T", 7); + let state = reg.claim_state("T").expect("terminal exists"); + assert_eq!(state.claimers, 0); + assert!( + state.released_by_client, + "withdrawal of the last claim must restore fast-reap eligibility" + ); + // DEV-0009: the withdrawal just happened — one full threshold of + // grace, so the stale pre-withdrawal clock does not reap immediately. assert!(reg.enforce_idle_kills().is_empty()); - assert_eq!(reg.inventory().len(), 1); - // Once it goes stale again AFTER the detach, it is reaped normally. + // Once stale again AFTER the withdrawal, the configured threshold + // applies (the claim is gone; this is the explicit release path). reg.backdate_last_activity("T", now_ms() - 10 * 60_000); assert_eq!(reg.enforce_idle_kills(), vec!["T".to_string()]); } + /// Claim membership is per-connection and sweeps away with its socket + /// (`remove_connection`), but transport loss is NOT release: a dropped + /// claiming connection keeps the terminal wanted (24h hard cap only — + /// identical to attached-then-disconnected today). A reconnected client + /// re-claims and the terminal stays wanted. #[test] - fn disconnect_grants_full_idle_threshold_of_grace() { - // Socket-close cleanup (remove_connection) is the other live - // transition-to-detached path and must grant the same grace as an - // explicit detach. The bump is gated on "this connection actually - // subscribed here AND the set became empty AND status is Running" — - // remove_connection iterates EVERY terminal, and an unconditional - // bump would reset unrelated detached terminals' countdowns on - // every socket close. + fn claim_connection_drop_keeps_terminal_wanted_and_reconnect_reclaims() { let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); - let (sink, _seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); - assert!(outcome.found); - // A second, already-detached terminal whose countdown must NOT be - // disturbed by conn 1's disconnect. - reg.insert_headless("U", "S"); reg.set_auto_kill_idle_minutes(1); + reg.claim_terminal("T", 1); + + reg.remove_connection(1); // transport loss, NOT an explicit release + let state = reg.claim_state("T").expect("terminal exists"); + assert_eq!(state.claimers, 0); + assert!( + !state.released_by_client, + "a dropped socket must not restore fast-reap eligibility" + ); + + // Past the configured threshold (but far under 24h): still wanted. reg.backdate_last_activity("T", now_ms() - 10 * 60_000); - reg.backdate_last_activity("U", now_ms() - 10 * 60_000); - reg.remove_connection(1); + assert!(reg.enforce_idle_kills().is_empty()); - let killed = reg.enforce_idle_kills(); + // The reconnected client re-claims the hidden terminal (reconnect- + // hidden re-claims) and it survives again. + assert!(reg.claim_terminal("T", 2)); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + assert!(reg.enforce_idle_kills().is_empty()); - // T was freshly detached by the disconnect => spared one threshold. - // U never had a subscriber => its stale countdown stands => reaped. - assert_eq!(killed, vec!["U".to_string()]); - assert_eq!(reg.inventory().len(), 1); + // The 24h hard cap stays the cleanup backstop for an abandoned claim. + reg.remove_connection(2); + reg.backdate_last_activity("T", now_ms() - 25 * 60 * 60_000); + assert_eq!(reg.enforce_idle_kills(), vec!["T".to_string()]); } + /// A claimed terminal that is then ATTACHED and whose socket drops stays + /// wanted — no regression of the attached-then-disconnected contract + /// (transport loss is not release) when claims coexist with subscriptions. #[test] - fn enforce_idle_kills_spares_disconnect_detached_terminal_past_threshold() { - // The background-session promise (§1.3, "what if tmux and Claude fell - // in love"): a user who closes the browser / sleeps the laptop with a - // quiet shell in a pane must find it ALIVE when they come back — the - // client never sent terminal.detach, so the terminal is detached but - // still wanted. Before the fix, 20 idle minutes past the default - // 15-minute autoKillIdleMinutes threshold silently SIGKILLed it - // (scrollback gone). Now only the 24h hard cap applies. + fn claimed_then_attached_terminal_survives_socket_drop() { let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); + reg.set_auto_kill_idle_minutes(1); + reg.claim_terminal("T", 2); let (sink, _seen) = collector(); assert!( - reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None) - .found + reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); - reg.set_auto_kill_idle_minutes(15); // the shipped default - reg.remove_connection(1); // socket drop, NOT an explicit detach - // 20 minutes idle — past the old ~15-minute threshold, far under 24h. - reg.backdate_last_activity("T", now_ms() - 20 * 60_000); - let killed = reg.enforce_idle_kills(); + reg.remove_connection(1); // the ATTACHED connection drops + let state = reg.claim_state("T").expect("terminal exists"); + assert_eq!(state.claimers, 1, "the claiming connection still holds it"); + assert!(!state.released_by_client); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + assert!(reg.enforce_idle_kills().is_empty()); + } + /// Detach reconciler interplay: a claimed+attached terminal leaving every + /// pane layout is released — but only once BOTH the last subscriber and + /// the last claim are gone. The claim is the explicit withdrawal path. + #[test] + fn detach_releases_only_when_the_last_claim_is_also_gone() { + let reg = TerminalRegistry::new(); + reg.insert_headless("T", "S"); + reg.set_auto_kill_idle_minutes(1); + reg.claim_terminal("T", 1); + let (sink, _seen) = collector(); assert!( - killed.is_empty(), - "a detached-but-wanted terminal (connection lost, never released) \ - must survive past the configured idle threshold, got {killed:?}" + reg.attach( + "T", + 2, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); - assert_eq!(reg.inventory().len(), 1); + + // The detach reconciler acts (terminal left every pane layout) while + // the claim is still recorded: still wanted — the other connection's + // claim keeps it alive. + reg.detach("T", 2); + let state = reg.claim_state("T").expect("terminal exists"); + assert_eq!(state.claimers, 1); + assert!( + !state.released_by_client, + "detach must not release a terminal another connection still claims" + ); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + assert!(reg.enforce_idle_kills().is_empty()); + + // The interest snapshot withdraws the claim (the pane is gone): NOW + // the terminal is genuinely orphaned and reap-eligible at the + // threshold, with the DEV-0009 grace bump on the release transition. + reg.withdraw_claim("T", 1); + assert!(reg.enforce_idle_kills().is_empty()); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); + assert_eq!(reg.enforce_idle_kills(), vec!["T".to_string()]); } + /// Reaper lifecycle for a created-hidden terminal (never attached, + /// released starts true): held only by the negotiated claim it survives + /// the configured threshold; explicit withdrawal (no subscribers) reaps + /// it at the threshold. #[test] - fn enforce_idle_kills_reaps_disconnect_detached_terminal_past_the_hard_cap() { - // Counterweight: the sweep stays the cleanup backstop for sessions - // whose client never comes back — same 24h hard cap agents get. + fn created_hidden_terminal_held_only_by_claim_survives_then_withdrawal_reaps() { let reg = TerminalRegistry::new(); reg.insert_headless("T", "S"); - let (sink, _seen) = collector(); + reg.set_auto_kill_idle_minutes(1); + + // Created-hidden: the client claims it in its interest snapshot + // instead of attaching. + assert!(reg.claim_terminal("T", 1)); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); assert!( - reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None) - .found + reg.enforce_idle_kills().is_empty(), + "a claim-only terminal must survive the configured idle threshold" ); - reg.set_auto_kill_idle_minutes(15); - reg.remove_connection(1); - // 25 hours stale — past the 24-hour hard cap. - reg.backdate_last_activity("T", now_ms() - 25 * 60 * 60_000); + // Explicit withdrawal with no subscribers: back to fast-reap + // eligibility, reaped at the threshold once past the grace bump. + reg.withdraw_claim("T", 1); + reg.backdate_last_activity("T", now_ms() - 10 * 60_000); assert_eq!(reg.enforce_idle_kills(), vec!["T".to_string()]); } @@ -6216,14 +12484,38 @@ mod tests { reg.insert_headless("T", "S"); let (sink, _seen) = collector(); assert!( - reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None) - .found + reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); reg.detach("T", 1); // explicitly released — fast-reap eligible let (sink2, _seen2) = collector(); assert!( - reg.attach("T", 2, sink2, Some("b".into()), 0, false, None, None) - .found + reg.attach( + "T", + 2, + sink2, + Some("b".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); reg.set_auto_kill_idle_minutes(15); reg.remove_connection(2); // wanted again, then socket drop @@ -6302,7 +12594,19 @@ mod tests { /// `enforce_idle_kills_never_kills_an_attached_terminal`'s attach shape. fn attach_test_subscriber(reg: &TerminalRegistry) { let (sink, _seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); + let outcome = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); } @@ -6385,8 +12689,20 @@ mod tests { // shape). let (sink, seen) = collector(); assert!( - reg.attach("T", 1, sink, Some("att-surv".into()), 0, false, None, None) - .found + reg.attach( + "T", + 1, + sink, + Some("att-surv".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); let stuck = stuck_frames(&seen); assert_eq!(stuck.len(), 1); @@ -6663,8 +12979,20 @@ mod tests { // truth still reports stuck:true — the flag survived the teardown. let (sink, seen) = collector(); assert!( - reg.attach("T", 2, sink, Some("a2".into()), 0, false, None, None) - .found + reg.attach( + "T", + 2, + sink, + Some("a2".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); let stuck = stuck_frames(&seen); assert_eq!(stuck.len(), 1); @@ -6685,8 +13013,20 @@ mod tests { reg.remove_connection(1); // last-subscriber teardown: grace bump fires let (sink, seen) = collector(); assert!( - reg.attach("T", 2, sink, Some("a2".into()), 0, false, None, None) - .found + reg.attach( + "T", + 2, + sink, + Some("a2".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default() + ) + .found ); let stuck = stuck_frames(&seen); assert_eq!(stuck.len(), 1); @@ -6727,7 +13067,19 @@ mod tests { let reg = stuck_test_registry("opencode"); flag_stuck_row(®); let (sink, seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("att-1".into()), 0, false, None, None); + let outcome = reg.attach( + "T", + 1, + sink, + Some("att-1".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); let msgs = seen.lock().unwrap().clone(); let ready_pos = msgs @@ -6759,7 +13111,19 @@ mod tests { let reg = stuck_test_registry("opencode"); reg.feed("T", frame(1, "meaningful boot text\r\n", "S")); let (sink, seen) = collector(); - let outcome = reg.attach("T", 1, sink, Some("att-2".into()), 0, false, None, None); + let outcome = reg.attach( + "T", + 1, + sink, + Some("att-2".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!(outcome.found); let stuck = stuck_frames(&seen); assert_eq!( @@ -6783,7 +13147,19 @@ mod tests { created_at: Some(now_ms()), }); let (sink, seen) = collector(); - let _ = reg.attach("T-shell", 1, sink, None, 0, false, None, None); + let _ = reg.attach( + "T-shell", + 1, + sink, + None, + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!( stuck_frames(&seen).is_empty(), "shell-mode rows emit no stuck truth on attach" @@ -6792,7 +13168,19 @@ mod tests { reg.feed("T", frame(1, "\r\x1b[2K⠋", "S")); assert!(reg.finish_pty_exit("T", 3)); let (sink, seen) = collector(); - let _ = reg.attach("T", 2, sink, None, 0, false, None, None); + let _ = reg.attach( + "T", + 2, + sink, + None, + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); assert!( stuck_frames(&seen).is_empty(), "exited rows emit no stuck truth on attach (the synthetic exit answers)" @@ -6808,7 +13196,19 @@ mod tests { let reg = stuck_test_registry("opencode"); flag_stuck_row(®); let (sink_a, seen_a) = collector(); - let _ = reg.attach("T", 1, sink_a, Some("att-a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink_a, + Some("att-a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let stuck_a = stuck_frames(&seen_a); assert_eq!(stuck_a.len(), 1); assert!(stuck_a[0].stuck, "precondition: sink A saw the flag"); @@ -6824,7 +13224,19 @@ mod tests { // The reconnect: a NEW subscriber (fresh conn id) learns stuck:false. let (sink_b, seen_b) = collector(); - let _ = reg.attach("T", 2, sink_b, Some("att-b".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 2, + sink_b, + Some("att-b".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let stuck_b = stuck_frames(&seen_b); assert_eq!(stuck_b.len(), 1); assert!( @@ -6885,7 +13297,19 @@ mod tests { reg.feed("T", frame(2, "abcdefghij", "S")); // another 10 bytes -> over cap let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let replayed = outputs(&seen); // Whole-frame FIFO eviction keeps at least one frame; the FIRST frame // must have been evicted once the second pushed bytes over the cap. @@ -6903,7 +13327,19 @@ mod tests { reg.feed("T", frame(2, "abcdefghij", "S")); let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("a".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink, + Some("a".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let replayed = outputs(&seen); assert_eq!( replayed.len(), @@ -6935,7 +13371,19 @@ mod tests { reg_ascii.feed("A", frame(1, "abcdef", "S")); // 6 chars, 6 bytes reg_ascii.feed("A", frame(2, "ghijkl", "S")); // 6 chars, 6 bytes -> 12 total, at cap let (sink_a, seen_a) = collector(); - let _ = reg_ascii.attach("A", 1, sink_a, Some("r".into()), 0, false, None, None); + let _ = reg_ascii.attach( + "A", + 1, + sink_a, + Some("r".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let ascii_chars: usize = outputs(&seen_a) .iter() .map(|f| f.data.chars().count()) @@ -6954,7 +13402,19 @@ mod tests { frame(2, "\u{2500}\u{2500}\u{2500}\u{2500}\u{2500}\u{2500}", "S"), ); let (sink_b, seen_b) = collector(); - let _ = reg_box.attach("B", 1, sink_b, Some("r".into()), 0, false, None, None); + let _ = reg_box.attach( + "B", + 1, + sink_b, + Some("r".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let box_chars: usize = outputs(&seen_b) .iter() .map(|f| f.data.chars().count()) @@ -7214,7 +13674,19 @@ mod tests { } let (sink, seen) = collector(); - let _ = reg.attach("T", 1, sink, Some("r".into()), 0, false, None, None); + let _ = reg.attach( + "T", + 1, + sink, + Some("r".into()), + 0, + false, + false, + None, + None, + None, + PacedAttachOptions::default(), + ); let retained_chars: usize = outputs(&seen).iter().map(|f| f.data.chars().count()).sum(); assert!( retained_chars as i64 <= cap, diff --git a/crates/freshell-terminal/tests/stuck_real_pty.rs b/crates/freshell-terminal/tests/stuck_real_pty.rs index 5330e75a3..5c7ef24dc 100644 --- a/crates/freshell-terminal/tests/stuck_real_pty.rs +++ b/crates/freshell-terminal/tests/stuck_real_pty.rs @@ -370,8 +370,11 @@ fn real_pty_capture_stream_flags_stuck_and_clears_on_meaningful() { Some("att-real".into()), 0, false, + false, + None, + None, None, - None + freshell_terminal::PacedAttachOptions::default() ) .found ); diff --git a/crates/freshell-ws/src/backpressure.rs b/crates/freshell-ws/src/backpressure.rs index 7e18e56f1..7843138a3 100644 --- a/crates/freshell-ws/src/backpressure.rs +++ b/crates/freshell-ws/src/backpressure.rs @@ -69,12 +69,109 @@ impl Default for Term09Config { fn default() -> Self { Self { queue_max_bytes: DEFAULT_TERMINAL_CLIENT_QUEUE_MAX_BYTES, - catastrophic_buffered_bytes: 16 * 1024 * 1024, + // Responsive-terminal-restore Workstream 3: the pressure-related + // disconnect threshold sits strictly ABOVE the spill bound (4x + // it). Legacy shipped 16 MiB — BELOW its own 32 MiB spill bound, + // so the disconnect fired before eviction could relieve the same + // pressure (the production incident). Because eviction holds + // pending bytes at or below `queue_max_bytes` (plus one + // indivisible in-flight frame), a threshold above the spill bound + // is unreachable by ordinary output pressure: the monitor is a + // last-resort guard for accounting drift and an oversize + // wedged in-flight frame, both independently bounded by the + // per-send write timeout. + catastrophic_buffered_bytes: 64 * 1024 * 1024, catastrophic_stall_ms: 10_000, } } } +/// Per-field sanity floor for `queue_max_bytes` (env +/// `TERMINAL_CLIENT_QUEUE_MAX_BYTES`): the connection loop already widens the +/// control budget to at least 64 KiB regardless, and a queue bound below one +/// large frame's scale degenerates the metadata-limit derivation +/// (`(limit / 64).clamp(64, ..)`) rather than tuning pressure. +pub const TERM09_QUEUE_MAX_BYTES_FLOOR: usize = 64 * 1024; + +/// Responsive-terminal-restore round-5 (finding 1, degenerate settings): +/// the paced-replay page-budget ceiling implied by a TERM-09 queue cap. +/// The drain-admission watermark is `queue_max_bytes / 2` +/// ([`crate::connection_writer`]'s reserve-then-admit gate), and the +/// queue's own byte cap evicts past `queue_max_bytes` — so a page larger +/// than the watermark can only admit into a fully drained queue (slow +/// but live), while a page larger than the whole cap self-spills on +/// admission. The server boot therefore CLAMPS the registry's page +/// budget to this ceiling at the one place both knobs are known +/// (`freshell-server`'s TERM-09 resolution; the ws test harness mirrors +/// the same relationship), documenting the queue-cap >= page-budget +/// relationship instead of leaving it to chance. With the defaults this +/// is a no-op (the 128 KiB page budget sits far below the 8 MiB +/// watermark); it only bites the small-queue settings — the supported +/// 64 KiB floor caps pages at 32 KiB. +/// +/// The clamped budget still exceeds one realtime frame's envelope by a +/// wide margin at every valid queue setting (the 64 KiB floor's 32 KiB +/// ceiling vs the 16 KiB `MAX_REALTIME_MESSAGE_BYTES` chunk cap), so the +/// page builder's single-frame atomic page never exceeds its budget. +pub const fn paced_page_budget_ceiling(queue_max_bytes: usize) -> i64 { + (queue_max_bytes / 2) as i64 +} + +/// Per-field sanity floor for `catastrophic_stall_ms` (env +/// `TERMINAL_WS_CATASTROPHIC_STALL_MS`): the monitor samples at +/// `stall / 4` (10 ms minimum), so a window below 100 ms would make the +/// last-resort disconnect decision under realistic scheduling/RTT jitter. +pub const TERM09_CATASTROPHIC_STALL_MS_FLOOR: u64 = 100; + +/// Fail-fast validation error for [`Term09Config`] +/// (responsive-terminal-restore Workstream 3): a configuration that would +/// restore the spill≥disconnect inversion must refuse to boot. Each variant +/// names the offending env var(s) so the operator knows exactly what to fix. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub enum Term09ConfigError { + /// `catastrophic_buffered_bytes <= queue_max_bytes`: the + /// pressure-related disconnect would fire at or before the bounded + /// spill (eviction + gap) could relieve the same pressure. + DisconnectNotAboveSpill { + queue_bytes: usize, + catastrophic_bytes: usize, + }, + /// `queue_max_bytes` below [`TERM09_QUEUE_MAX_BYTES_FLOOR`]. + QueueMaxBytesFloor { bytes: usize }, + /// `catastrophic_stall_ms` below [`TERM09_CATASTROPHIC_STALL_MS_FLOOR`]. + CatastrophicStallMsFloor { ms: u64 }, +} + +impl std::fmt::Display for Term09ConfigError { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + match *self { + Self::DisconnectNotAboveSpill { + queue_bytes, + catastrophic_bytes, + } => write!( + f, + "invalid TERM-09 backpressure config: TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES \ + ({catastrophic_bytes}) must be strictly greater than \ + TERMINAL_CLIENT_QUEUE_MAX_BYTES ({queue_bytes}); a disconnect threshold at or \ + below the spill bound disconnects slow clients before bounded spill (eviction \ + + generation-scoped gap) can relieve pressure" + ), + Self::QueueMaxBytesFloor { bytes } => write!( + f, + "invalid TERM-09 backpressure config: TERMINAL_CLIENT_QUEUE_MAX_BYTES ({bytes}) \ + is below the {TERM09_QUEUE_MAX_BYTES_FLOOR}-byte floor" + ), + Self::CatastrophicStallMsFloor { ms } => write!( + f, + "invalid TERM-09 backpressure config: TERMINAL_WS_CATASTROPHIC_STALL_MS ({ms}) \ + is below the {TERM09_CATASTROPHIC_STALL_MS_FLOOR}-ms floor" + ), + } + } +} + +impl std::error::Error for Term09ConfigError {} + use crate::env_parse; impl Term09Config { @@ -94,17 +191,71 @@ impl Term09Config { ), } } + + /// Cross-field validation, enforced fail-fast at boot wiring + /// (`freshell-server` resolves this before constructing `WsState`): + /// normal output pressure must reach bounded admission/spill (eviction + + /// generation-scoped gap) STRICTLY before any pressure-related + /// disconnect, so `catastrophic_buffered_bytes` must sit strictly above + /// `queue_max_bytes` (equal also refuses: the disconnect would fire the + /// moment the queue is full). Per-field sanity floors reject degenerate + /// tunings. Test harnesses may still inject arbitrary `Term09Config` + /// values directly into `WsState`; only the env/boot path is guarded. + pub fn validate(&self) -> Result<(), Term09ConfigError> { + if self.queue_max_bytes < TERM09_QUEUE_MAX_BYTES_FLOOR { + return Err(Term09ConfigError::QueueMaxBytesFloor { + bytes: self.queue_max_bytes, + }); + } + if self.catastrophic_stall_ms < TERM09_CATASTROPHIC_STALL_MS_FLOOR { + return Err(Term09ConfigError::CatastrophicStallMsFloor { + ms: self.catastrophic_stall_ms, + }); + } + if self.catastrophic_buffered_bytes <= self.queue_max_bytes { + return Err(Term09ConfigError::DisconnectNotAboveSpill { + queue_bytes: self.queue_max_bytes, + catastrophic_bytes: self.catastrophic_buffered_bytes, + }); + } + Ok(()) + } } -/// Tracks how long the connection writer's pending output bytes (queued plus -/// in-flight frame) have been continuously over `catastrophic_buffered_bytes`. -/// Mirrors `catastrophicBlocked` (`broker.ts:1087-1109`): the threshold must -/// be exceeded for the FULL stall duration, uninterrupted, before firing; any -/// tick that observes recovery resets the clock. +/// Fire-time evidence for ONE sustained catastrophic-backpressure +/// occurrence (task-007 review M3, landed by task-010): what the +/// `ws.terminal_stream.catastrophic_close` event needs to be diagnosable +/// from the log line alone. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct CatastrophicFire { + /// Completed socket sends DURING the deciding window for THIS + /// occurrence — a per-occurrence delta, not a lifetime counter. + /// Structurally zero (any send resets the window), so a nonzero value + /// is accounting drift the log line exposes directly; a lifetime + /// total cannot answer this question (a wedge-after-progress episode + /// reads large lifetime sends while the window was send-silent). + pub sends_in_window: u64, +} + +/// Tracks whether the connection writer's pending output bytes (queued plus +/// in-flight frame) have been continuously over `catastrophic_buffered_bytes` +/// WITH ZERO successful socket sends, for the full `stall` duration +/// (responsive-terminal-restore Workstream 3). The byte threshold alone is +/// NOT a dead-socket signal — a slow-but-draining client holds a large +/// backlog while making steady send progress — so the window resets on +/// EITHER recovery below the threshold OR send progress, and firing requires +/// both conditions to hold for the whole window. Byte reductions from +/// eviction or superseded attachments do not count: they are not sends. pub struct CatastrophicMonitor { threshold_bytes: usize, stall: Duration, since: Option, + /// Completed sends accumulated inside the CURRENT sustained window + /// (see [`CatastrophicFire::sends_in_window`]): every tick's delta adds + /// here while the window stays open, and any window reset (recovery or + /// send progress) zeroes it — so at fire time it is exactly the sends + /// that happened during the deciding window. + window_sends: u64, } impl CatastrophicMonitor { @@ -113,20 +264,45 @@ impl CatastrophicMonitor { threshold_bytes: threshold_bytes.max(1), stall: Duration::from_millis(stall_ms.max(1)), since: None, + window_sends: 0, } } - /// Call on each periodic check with the CURRENT pending-byte count. - /// Returns `true` the moment sustained overflow has crossed the stall - /// duration (fires exactly once per sustained episode; the caller is - /// expected to close the connection immediately on `true`). - pub fn tick(&mut self, pending_bytes: usize) -> bool { - if pending_bytes <= self.threshold_bytes { + /// Call on each periodic check with the CURRENT pending-byte count and + /// the number of SUCCESSFUL SOCKET SENDS completed since the previous + /// tick (drain-progress liveness, responsive-terminal-restore Workstream + /// 3). The sustained window resets when EITHER pending bytes fall below + /// the threshold OR sends progressed: a slow-but-draining client is not + /// a dead socket, so disconnect requires sustained bytes over threshold + /// AND zero successful sends for the whole stall window. Byte + /// reductions from eviction or superseded attachments do NOT count as + /// progress — only completed sends (the caller feeds the connection + /// writer's completed-send counter; queue-size deltas are invisible + /// here). Fires exactly once per sustained episode (the caller closes + /// the connection immediately on `Some`); the returned evidence carries + /// the per-occurrence `sends_in_window` delta for the close event. + pub fn tick( + &mut self, + pending_bytes: usize, + sends_since_last_tick: u64, + ) -> Option { + if pending_bytes <= self.threshold_bytes || sends_since_last_tick > 0 { self.since = None; - return false; + self.window_sends = 0; + return None; } + self.window_sends = self.window_sends.saturating_add(sends_since_last_tick); let since = *self.since.get_or_insert_with(Instant::now); - since.elapsed() >= self.stall + if since.elapsed() < self.stall { + return None; + } + // One fire per sustained episode: a caller that (against the + // contract) kept ticking would need a full fresh window to fire + // again, with fresh per-occurrence evidence. + self.since = None; + let sends_in_window = self.window_sends; + self.window_sends = 0; + Some(CatastrophicFire { sends_in_window }) } } @@ -135,18 +311,198 @@ mod tests { use super::*; #[test] - fn term09_config_defaults_match_legacy_constants() { + fn paced_page_budget_ceiling_tracks_the_admission_watermark() { + // Round-5 finding 1 (degenerate settings): the page-budget ceiling + // is the queue's drain-admission watermark (queue cap / 2), so the + // boot clamp keeps every paced page inside the reserve-then-admit + // gate's normal grant arm — never larger than the queue itself + // (self-spill) and never deadlocking the gate. + assert_eq!( + paced_page_budget_ceiling(16 * 1024 * 1024), + 8 * 1024 * 1024, + "the default 16 MiB queue leaves the 128 KiB default budget untouched" + ); + assert_eq!( + paced_page_budget_ceiling(TERM09_QUEUE_MAX_BYTES_FLOOR), + 32 * 1024, + "the supported 64 KiB queue floor caps pages at 32 KiB — the degenerate \ + default-128-KiB-page case must not self-spill a 64 KiB queue" + ); + // The clamped budget always dwarfs one realtime frame's envelope + // (the 16 KiB MAX_REALTIME_MESSAGE_BYTES chunk cap), so the page + // builder's single-frame atomic page never exceeds the budget at + // any valid queue setting. + assert!(paced_page_budget_ceiling(TERM09_QUEUE_MAX_BYTES_FLOOR) > 16 * 1024); + } + + #[test] + fn the_page_floor_equals_the_terminal_crates_fragment_cap_floor() { + // Round-2 finding F2 (cross-crate consistency pin): the terminal + // crate clamps its fragment cap to "the smallest page budget any + // supported queue setting can produce", computed on THIS side as + // the ceiling at the queue floor. The two constants must agree — + // if either drifts, the frame-fits-page invariant silently breaks + // (a fragment cap above the floor could mint frames the page + // builder cannot pack; a floor above the fragment clamp would + // re-open the env-override gap). + assert_eq!( + paced_page_budget_ceiling(TERM09_QUEUE_MAX_BYTES_FLOOR), + freshell_terminal::PACED_PAGE_BUDGET_FLOOR_BYTES as i64, + "the fragment-cap floor must equal the paced page budget floor at the queue floor" + ); + } + + #[test] + fn term09_config_defaults_spill_before_disconnect() { + // Responsive-terminal-restore Workstream 3: the defaults must place + // the spill bound (eviction + gap) STRICTLY below the + // pressure-related disconnect, sized so the production incident's + // ~21-25 MB backlog spills gracefully instead of disconnecting. let cfg = Term09Config::default(); - assert_eq!(cfg.queue_max_bytes, 32 * 1024 * 1024); - assert_eq!(cfg.catastrophic_buffered_bytes, 16 * 1024 * 1024); + assert_eq!(cfg.queue_max_bytes, 16 * 1024 * 1024, "spill bound: 16 MiB"); + assert_eq!( + cfg.catastrophic_buffered_bytes, + 64 * 1024 * 1024, + "disconnect bound: 64 MiB" + ); assert_eq!(cfg.catastrophic_stall_ms, 10_000); + // The incident backlog (~21-25 MB) exceeds the spill bound (so it + // spills) but stays far below the disconnect bound (so it survives). + let incident_low = 21 * 1024 * 1024; + let incident_high = 25 * 1024 * 1024; + assert!(incident_low > cfg.queue_max_bytes); + assert!(incident_high < cfg.catastrophic_buffered_bytes); + cfg.validate().expect("defaults must satisfy the ordering"); + } + + #[test] + fn validate_rejects_disconnect_at_or_below_spill() { + // Equal refuses (strict ordering): the disconnect would fire the + // moment the queue sits exactly full. + let equal = Term09Config { + queue_max_bytes: 8 * 1024 * 1024, + catastrophic_buffered_bytes: 8 * 1024 * 1024, + catastrophic_stall_ms: 10_000, + }; + let err = equal.validate().expect_err("equalized pair must refuse"); + assert_eq!( + err, + Term09ConfigError::DisconnectNotAboveSpill { + queue_bytes: 8 * 1024 * 1024, + catastrophic_bytes: 8 * 1024 * 1024, + } + ); + // Below refuses — the restored legacy inversion (32 MB spill / 16 MB + // disconnect) must never boot again. + let inverted = Term09Config { + queue_max_bytes: 32 * 1024 * 1024, + catastrophic_buffered_bytes: 16 * 1024 * 1024, + catastrophic_stall_ms: 10_000, + }; + let err = inverted + .validate() + .expect_err("inverted pair must refuse (the incident's config)"); + let message = err.to_string(); + assert!( + message.contains("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES") + && message.contains("TERMINAL_CLIENT_QUEUE_MAX_BYTES"), + "the error must name the offending env vars: {message}" + ); + // Strictly above passes. + Term09Config { + queue_max_bytes: 8 * 1024 * 1024, + catastrophic_buffered_bytes: 8 * 1024 * 1024 + 1, + catastrophic_stall_ms: 10_000, + } + .validate() + .expect("strictly-above ordering must validate"); + } + + #[test] + fn validate_rejects_degenerate_floors() { + let queue_floor = Term09Config { + queue_max_bytes: TERM09_QUEUE_MAX_BYTES_FLOOR - 1, + catastrophic_buffered_bytes: 64 * 1024 * 1024, + catastrophic_stall_ms: 10_000, + }; + let err = queue_floor.validate().expect_err("queue floor"); + let message = err.to_string(); + assert!( + message.contains("TERMINAL_CLIENT_QUEUE_MAX_BYTES"), + "the error must name the offending env var: {message}" + ); + let stall_floor = Term09Config { + queue_max_bytes: 16 * 1024 * 1024, + catastrophic_buffered_bytes: 64 * 1024 * 1024, + catastrophic_stall_ms: TERM09_CATASTROPHIC_STALL_MS_FLOOR - 1, + }; + let err = stall_floor.validate().expect_err("stall floor"); + let message = err.to_string(); + assert!( + message.contains("TERMINAL_WS_CATASTROPHIC_STALL_MS"), + "the error must name the offending env var: {message}" + ); + // The floors themselves validate. + Term09Config { + queue_max_bytes: TERM09_QUEUE_MAX_BYTES_FLOOR, + catastrophic_buffered_bytes: TERM09_QUEUE_MAX_BYTES_FLOOR + 1, + catastrophic_stall_ms: TERM09_CATASTROPHIC_STALL_MS_FLOOR, + } + .validate() + .expect("floor values must validate"); + } + + /// Env-dependent cases live in ONE test fn: `std::env::set_var` mutates + /// whole-process state, so parallel sibling tests must not race these + /// vars. The vars are removed on exit; no other test in this crate reads + /// them. + #[test] + fn from_env_overrides_validate_at_boot_shape() { + std::env::remove_var("TERMINAL_CLIENT_QUEUE_MAX_BYTES"); + std::env::remove_var("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES"); + std::env::remove_var("TERMINAL_WS_CATASTROPHIC_STALL_MS"); + + // Unset -> defaults, which must satisfy the ordering. + let defaults = Term09Config::from_env(); + defaults + .validate() + .expect("unset env must yield valid defaults"); + + // An override that inverts the ordering must fail validation. + std::env::set_var("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES", "1048576"); + let inverted = Term09Config::from_env(); + let err = inverted + .validate() + .expect_err("an inverted env override must fail validation"); + assert_eq!( + err, + Term09ConfigError::DisconnectNotAboveSpill { + queue_bytes: defaults.queue_max_bytes, + catastrophic_bytes: 1024 * 1024, + } + ); + + // Valid overrides pass. + std::env::remove_var("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES"); + std::env::set_var("TERMINAL_CLIENT_QUEUE_MAX_BYTES", "2097152"); + std::env::set_var("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES", "8388608"); + let tuned = Term09Config::from_env(); + assert_eq!(tuned.queue_max_bytes, 2 * 1024 * 1024); + assert_eq!(tuned.catastrophic_buffered_bytes, 8 * 1024 * 1024); + tuned + .validate() + .expect("ordered env overrides must validate"); + + std::env::remove_var("TERMINAL_CLIENT_QUEUE_MAX_BYTES"); + std::env::remove_var("TERMINAL_WS_CATASTROPHIC_BUFFERED_BYTES"); + std::env::remove_var("TERMINAL_WS_CATASTROPHIC_STALL_MS"); } #[test] fn catastrophic_monitor_never_fires_under_threshold() { let mut m = CatastrophicMonitor::new(100, 10); for _ in 0..5 { - assert!(!m.tick(50)); + assert!(m.tick(50, 0).is_none()); std::thread::sleep(Duration::from_millis(15)); } } @@ -154,22 +510,107 @@ mod tests { #[test] fn catastrophic_monitor_resets_on_recovery_before_stall_elapses() { let mut m = CatastrophicMonitor::new(100, 1000); - assert!(!m.tick(200)); // starts the clock - assert!(!m.tick(50)); // recovers immediately -> resets + assert!(m.tick(200, 0).is_none()); // starts the clock + assert!(m.tick(50, 0).is_none()); // recovers immediately -> resets std::thread::sleep(Duration::from_millis(5)); // Overflow again: a FRESH clock, so it must not have carried over // elapsed time from the first (reset) episode. - assert!(!m.tick(200)); + assert!(m.tick(200, 0).is_none()); } #[test] fn catastrophic_monitor_fires_after_sustained_overflow() { let mut m = CatastrophicMonitor::new(100, 20); - assert!(!m.tick(200)); + assert!(m.tick(200, 0).is_none()); std::thread::sleep(Duration::from_millis(35)); assert!( - m.tick(200), + m.tick(200, 0).is_some(), "sustained overflow past the stall duration must fire" ); } + + /// Drain-progress liveness (responsive-terminal-restore Workstream 3): + /// successful sends reset the sustained window even while pending bytes + /// stay over the threshold — a slow-but-progressing client is not a dead + /// socket. + #[test] + fn send_progress_resets_the_stall_window() { + let mut m = CatastrophicMonitor::new(100, 40); + assert!(m.tick(200, 0).is_none()); // starts the clock + std::thread::sleep(Duration::from_millis(25)); + assert!(m.tick(200, 1).is_none()); // a send completed -> resets the window + std::thread::sleep(Duration::from_millis(30)); + // 55 ms since the FIRST over-threshold tick — past the 40 ms window — + // but only 30 ms since the progress reset: must NOT fire. + assert!( + m.tick(200, 0).is_none(), + "the window must restart from the last progress, not the first tick" + ); + // With no further progress it DOES fire after the full window. + std::thread::sleep(Duration::from_millis(45)); + assert!(m.tick(200, 0).is_some()); + } + + /// Eviction and supersede reduce queue bytes WITHOUT a send; those byte + /// reductions must never masquerade as drain progress. The monitor only + /// sees completed sends, so over-threshold bytes with zero sends close on + /// schedule no matter how the byte count wiggles. + #[test] + fn byte_reductions_without_sends_do_not_reset_the_window() { + let mut m = CatastrophicMonitor::new(100, 30); + assert!(m.tick(180, 0).is_none()); // starts the clock + std::thread::sleep(Duration::from_millis(10)); + assert!(m.tick(150, 0).is_none()); // "eviction" shrank the count; still over, no sends + std::thread::sleep(Duration::from_millis(10)); + assert!(m.tick(190, 0).is_none()); // refilled; still no sends + std::thread::sleep(Duration::from_millis(15)); + assert!( + m.tick(160, 0).is_some(), + "over-threshold bytes with zero sends across the whole window must close" + ); + } + + /// Task-007 review M3 (landed by task-010): the fire evidence carries + /// `sends_in_window` — completed sends DURING the deciding window for + /// THIS occurrence, not the caller's lifetime counter. A + /// progress-then-wedge episode must report ZERO window sends even though + /// sends happened before the window opened; structurally the field is + /// always 0 at fire time (any send resets the window), so a nonzero + /// value is accounting drift the log line exposes directly. + #[test] + fn fire_evidence_reports_sends_inside_the_deciding_window_not_the_lifetime() { + let mut m = CatastrophicMonitor::new(100, 30); + // Wedge-after-progress: sends complete while over threshold (each + // resets the window), then a sustained zero-send window fires. + assert!(m.tick(200, 7).is_none(), "send progress resets the window"); + std::thread::sleep(Duration::from_millis(10)); + assert!( + m.tick(200, 4).is_none(), + "more progress, window keeps resetting" + ); + // The next over-threshold zero-send tick OPENS the deciding window; + // it cannot fire yet. + assert!(m.tick(200, 0).is_none(), "the window opens, not fires"); + std::thread::sleep(Duration::from_millis(35)); + let Some(fire) = m.tick(200, 0) else { + panic!("the sustained zero-send window must fire"); + }; + assert_eq!( + fire.sends_in_window, 0, + "no sends completed inside the deciding window — the per-occurrence \ + evidence must say so directly (a lifetime counter would read 11 here)" + ); + // A second sustained episode after the fire starts a FRESH window + // with fresh evidence (the monitor reports one occurrence per + // sustained episode; the caller closes the connection on fire). + assert!( + m.tick(200, 0).is_none(), + "post-fire ticks open a fresh window" + ); + std::thread::sleep(Duration::from_millis(35)); + let Some(second) = m.tick(200, 0) else { + panic!("the second sustained window must fire too"); + }; + assert_eq!(second.sends_in_window, 0); + } } diff --git a/crates/freshell-ws/src/connection_writer.rs b/crates/freshell-ws/src/connection_writer.rs index 9a848d48d..2514d407a 100644 --- a/crates/freshell-ws/src/connection_writer.rs +++ b/crates/freshell-ws/src/connection_writer.rs @@ -25,7 +25,7 @@ use freshell_protocol::ServerMessage; mod delivery; #[path = "terminal_interest.rs"] mod terminal_interest; -use delivery::{Delivery, DeliveryQueue, Range}; +use delivery::{Delivery, DeliveryQueue, EvictedOutput, Range}; use freshell_terminal::output_queue::output_frame_meta; use futures_util::{Sink, SinkExt}; use terminal_interest::InterestState; @@ -92,6 +92,25 @@ struct Control { /// replay. const CONTROL_STREAK_LIMIT: usize = 8; +/// Minimum spacing between `ws.terminal_stream.queue_overflow_spill` events +/// per connection (responsive-terminal-restore Workstream 3 observability). +/// Under sustained eviction a single admission can evict many frames — +/// hundreds of evictions per second under incident-scale pressure — so +/// per-eviction events would flood the log; evictions inside the window are +/// folded into the next event's `suppressed` count. Identifiers and +/// measurements only. +const SPILL_EVENT_MIN_INTERVAL: Duration = Duration::from_secs(5); + +/// One rate-limited spill (queue-overflow eviction) observability event. +struct SpillEvent { + terminal_id: String, + stream_id: String, + from_seq: i64, + to_seq: i64, + suppressed: u64, + pending_bytes: usize, +} + struct Queues { output: DeliveryQueue, interest: InterestState, @@ -106,14 +125,107 @@ struct Queues { /// are strictly increasing in true admission order. next_seq: u64, closed: bool, + /// Successful socket sends completed on this connection (drain-progress + /// liveness, responsive-terminal-restore Workstream 3): incremented by + /// `finish_frame` for every frame whose send resolved Ok — output and + /// control alike, since the pump serializes all sends. Eviction and + /// superseded attachments reduce queued bytes WITHOUT touching this. + completed_sends: u64, + /// Rate-limited spill-event bookkeeping (see + /// [`SPILL_EVENT_MIN_INTERVAL`]): when the last + /// `ws.terminal_stream.queue_overflow_spill` event was emitted, and how + /// many evictions have been folded into the next one since. Evictions + /// are counted at ADMISSION time (task-007 review M2); leasing the + /// coalesced gap later is delivery, not a spill occurrence. + spill_last_logged: Option, + spill_suppressed: u64, + /// Total bytes reserved by in-flight drain admissions + /// (responsive-terminal-restore W1, round-5 finding 1): the paced + /// drain task reserves its page's budget BEFORE building and sinking + /// the page, so concurrent pane drains can never double-book the + /// admission watermark (the reserve-then-admit gate). Released by the + /// [`DrainAdmission`] guard's Drop — after the page's bytes are real + /// queued backlog, or unused when the admission produced no page. + drain_reserved: usize, +} + +impl Queues { + /// Rate-limited spill-event bookkeeping (task-007 review M2, landed by + /// task-010): returns `Some` when an event should be emitted NOW (this + /// eviction's range plus the count of evictions suppressed since the + /// previous event), `None` when this eviction is folded into a future + /// event. Called under the admission lock at EVICTION time — the moment + /// the spill happens — so a connection that dies while backlogged (its + /// coalesced gap never leased to the socket) still leaves spill evidence + /// in the live log. + fn note_spill(&mut self, terminal_id: &str, range: &Range) -> Option { + let now = std::time::Instant::now(); + let due = self + .spill_last_logged + .is_none_or(|last| now.duration_since(last) >= SPILL_EVENT_MIN_INTERVAL); + if !due { + self.spill_suppressed = self.spill_suppressed.saturating_add(1); + return None; + } + let suppressed = self.spill_suppressed; + self.spill_suppressed = 0; + self.spill_last_logged = Some(now); + Some(SpillEvent { + terminal_id: terminal_id.to_string(), + stream_id: range.stream_id.clone(), + from_seq: range.from_seq, + to_seq: range.to_seq, + suppressed, + pending_bytes: self + .output + .pending_bytes() + .saturating_add(self.in_flight_output_bytes), + }) + } + + /// Map the delivery queue's admission-time eviction records through the + /// per-connection rate limiter (task-007 review M2). Call under the + /// admission lock immediately after `push`; emit the returned events only + /// AFTER the lock is dropped (never hold the admission lock across a log + /// write). + fn take_admission_spills(&mut self, evicted: Vec) -> Vec { + evicted + .into_iter() + .filter_map(|evicted| self.note_spill(&evicted.terminal_id, &evicted.range)) + .collect() + } } +/// Emission-time resolver for one terminal's current restore-contract +/// replay-retention bounds (responsive-terminal-restore): installed ONLY on +/// connections whose `hello` negotiated `pacedTerminalReplayV1`, backed by +/// the registry's [`freshell_terminal::TerminalRegistry::replay_bounds`]. +/// Generic over the registry so the writer's unit tests can drive the pump +/// deterministically without a PTY. +type GapBoundsSource = Arc Option + Send + Sync>; + struct Shared { queues: Mutex, output_limit: usize, control_limit: usize, ready: Notify, stop: watch::Sender>, + /// Drain-side backpressure signal (responsive-terminal-restore W1): + /// the connection's current output backlog (`pending + in-flight`), + /// updated by the writer pump on every completed frame send. The + /// paced drain task waits on it between pages so the un-credited + /// drain is bounded by the connection queue's REAL backpressure — + /// the sink itself (`push_server`) admits without yielding (it + /// evicts the oldest queued output past the byte limit), so without + /// this gate a drain could self-spill its own unconsumed pages. + backlog: watch::Sender, + /// Restore-contract bounds for materialized `terminal.output.gap` + /// frames: set ONLY on negotiated connections, ONCE, by the connection + /// setup BEFORE the pump is spawned (the same pre-spawn setup rule as + /// `enable_terminal_interest`) — a gap can never be leased before the + /// source exists. Unset keeps gap frames byte-identical to the + /// pre-capability wire shape. + gap_bounds: std::sync::OnceLock, } /// A nonblocking, bounded outbox. Its Sink flush means "accepted by this @@ -139,6 +251,53 @@ struct NextFrame { flushed: Option>, } +/// One granted drain-admission reservation (round-5 finding 1): the +/// holder may build and sink one budget-bounded paced page — its bytes +/// were accounted atomically under the queue's lock BEFORE the page was +/// produced, alongside every other in-flight drain reservation. The +/// guard is the reservation's lifetime: dropping it releases the +/// capacity (after the page's bytes became real queued backlog, or unused +/// when the admission produced no page — a retention gap, a Gone +/// verdict, a cancelled session). +pub(crate) struct DrainAdmission { + shared: Arc, + bytes: usize, +} + +impl Drop for DrainAdmission { + fn drop(&mut self) { + // Also runs after the pump's Drop reset the queue state: the + // saturating subtraction keeps a stale guard's release inert. + let mut backlog_publish: Option = None; + if let Ok(mut queues) = self.shared.queues.lock() { + queues.drain_reserved = queues.drain_reserved.saturating_sub(self.bytes); + // Round-2 finding F5: the release FREES admission capacity — + // publish the backlog so tasks waiting on the gate + // (`reserve_drain_admission`) re-evaluate NOW instead of + // sleeping until an unrelated socket send or keepalive + // publishes it. A drain that completed without admitting (a + // CaughtUp/Gone verdict: no page, no send) otherwise strands + // concurrent drains behind capacity that already came back. + // `watch::send` wakes every waiter even for an equal value + // (the change mark is versioned, not value-compared), and the + // post-pump-Drop reset path publishes the same way the pump's + // own Drop does. + backlog_publish = Some( + queues + .output + .pending_bytes() + .saturating_add(queues.in_flight_output_bytes), + ); + } + if let Some(backlog) = backlog_publish { + // Best-effort like every other publish (a closed channel means + // no drain is waiting — nothing to wake). Never held under the + // queue lock. + let _ = self.shared.backlog.send(backlog); + } + } +} + impl WriterSender { pub(super) fn new( output_limit: usize, @@ -146,6 +305,8 @@ impl WriterSender { write_timeout: Duration, ) -> (Self, WriterPump) { let (stop_tx, stop_rx) = watch::channel(None); + let (backlog_tx, backlog_rx) = watch::channel(0usize); + let _ = backlog_rx; // receivers subscribe per wait; the sender owns the channel let shared = Arc::new(Shared { queues: Mutex::new(Queues { output: DeliveryQueue::new(output_limit, metadata_limit(output_limit)), @@ -156,11 +317,17 @@ impl WriterSender { controls_since_last_output: 0, next_seq: 0, closed: false, + completed_sends: 0, + spill_last_logged: None, + spill_suppressed: 0, + drain_reserved: 0, }), output_limit: output_limit.max(1), control_limit: control_limit.max(1), ready: Notify::new(), stop: stop_tx, + backlog: backlog_tx, + gap_bounds: std::sync::OnceLock::new(), }); ( Self { @@ -276,7 +443,11 @@ impl WriterSender { return false; } }; - if meta.is_none() && !exit { + // Restore contract: a directly pushed `terminal.output.gap` (the paced + // replay core's retention gaps) joins the OUTPUT queue as a sequenced + // control like `terminal.exit` — see the gap arm below. + let sequenced_gap = matches!(&msg, ServerMessage::TerminalOutputGap(_)); + if meta.is_none() && !exit && !sequenced_gap { return self .push_control(Message::Text(json.into()), None, supersedes.as_deref()) .is_ok(); @@ -288,7 +459,14 @@ impl WriterSender { let seq = queues.next_seq; queues.next_seq += 1; let bytes = json.len(); - if let Some(meta) = meta { + // All three queued shapes below share one admission tail: the push, + // the admission-time spill evidence, and the notify/fail mapping. + // Spill observability (task-007 review M2, landed by task-010): + // evictions surface HERE — the moment they happen — including on the + // error path (a dying connection's evictions are exactly the + // undercounted spills the review found), and the events are emitted + // only after the admission lock is dropped. + let pushed = if let Some(meta) = meta { let range = Range { stream_id: meta.stream_id, attach_request_id: meta.attach_request_id, @@ -296,54 +474,87 @@ impl WriterSender { to_seq: meta.seq_end, }; let priority = queues.interest.priority(&meta.terminal_id); - if queues - .output - .push( - &meta.terminal_id, - priority, - Message::Text(json.into()), - bytes, - Some(range), - seq, - ) - .is_err() - { - drop(queues); - self.fail(WriterExit::OutputCapacityExceeded); - return false; - } - } else { + queues.output.push( + &meta.terminal_id, + priority, + Message::Text(json.into()), + bytes, + Some(range), + seq, + ) + } else if let ServerMessage::TerminalExit(exit) = &msg { // Preserve final-output -> exit. It must not use the control lane. - let ServerMessage::TerminalExit(exit) = msg else { - unreachable!("sequenced exit only") - }; let priority = queues.interest.priority(&exit.terminal_id); // Sequenced exits are zero-weight, exactly as legacy queued them: // they can never force an eviction nor close the connection, and // they still cost one service unit per frame (count-bounded by // the metadata limit). - if queues - .output - .push( - &exit.terminal_id, - priority, - Message::Text(json.into()), - 0, - None, - seq, - ) - .is_err() - { - drop(queues); + let pushed = queues.output.push( + &exit.terminal_id, + priority, + Message::Text(json.into()), + 0, + None, + seq, + ); + // A dead terminal never needs its attach fallback again. + if pushed.is_ok() { + queues.interest.detach(&exit.terminal_id); + } + pushed + } else { + // Restore contract (responsive-terminal-restore): a + // `terminal.output.gap` pushed DIRECTLY by the paced replay core + // (retention loss at attach / mid-replay expiry) is sequenced + // WITH the terminal's output — exactly the `terminal.exit` + // zero-weight non-evictable control treatment. The control lane + // would preempt it AHEAD of already-admitted pages, breaking + // per-terminal sequence order (the queue's own gap markers, by + // contrast, materialize at lease time from eviction and never + // pass through here). + let ServerMessage::TerminalOutputGap(gap) = &msg else { + unreachable!("meta-less output frames are exit or gap only") + }; + let priority = queues.interest.priority(&gap.terminal_id); + queues.output.push( + &gap.terminal_id, + priority, + Message::Text(json.into()), + 0, + None, + seq, + ) + }; + let evicted = queues.output.take_evictions(); + let spills = queues.take_admission_spills(evicted); + drop(queues); + Self::emit_spill_events(spills); + match pushed { + Ok(()) => { + self.shared.ready.notify_one(); + true + } + Err(_) => { self.fail(WriterExit::OutputCapacityExceeded); - return false; + false } - // A dead terminal never needs its attach fallback again. - queues.interest.detach(&exit.terminal_id); } - drop(queues); - self.shared.ready.notify_one(); - true + } + + /// Log the rate-limited spill events collected at admission time. Must + /// be called with the admission lock NOT held. + fn emit_spill_events(spills: Vec) { + for spill in spills { + tracing::warn!( + terminal_id = %spill.terminal_id, + stream_id = %spill.stream_id, + from_seq = spill.from_seq, + to_seq = spill.to_seq, + suppressed = spill.suppressed, + pending_bytes = spill.pending_bytes, + "ws.terminal_stream.queue_overflow_spill" + ); + } } pub(super) fn enable_terminal_interest(&self) { @@ -355,26 +566,58 @@ impl WriterSender { .enable(); } + /// Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1): + /// arm the connection's `terminal.interest.claimedTerminalIds` handling. + /// Called ONCE by the connection setup when the hello negotiated + /// `terminalLifetimeClaimV1`; a connection that never negotiated keeps + /// its snapshots' claim fields ignored server-side. + pub(super) fn enable_terminal_lifetime_claims(&self) { + self.shared + .queues + .lock() + .expect("writer queue lock") + .interest + .enable_claims(); + } + + /// Restore contract (responsive-terminal-restore): install the + /// negotiated-connection gap-bounds source. Called ONCE by the + /// connection setup, BEFORE the writer pump is spawned (a gap can never + /// be leased before the source exists). The source resolves a + /// terminal's current `head_seq`/earliest-replayable position at + /// gap-emission time; connections that never negotiated leave the + /// source unset and their gap frames stay byte-identical to the + /// pre-capability wire shape. + pub(super) fn set_paced_replay_gap_bounds(&self, source: GapBoundsSource) { + let _ = self.shared.gap_bounds.set(source); + } + /// Apply one full presentation-interest snapshot. A rejected snapshot is /// returned without replacing the last accepted state; scheduling changes - /// are queued-data-only (no attach, resize, spawn, or kill). + /// are queued-data-only (no attach, resize, spawn, or kill). On + /// acceptance, the negotiated claim-set diff is handed back to the + /// dispatcher, which applies it to the terminal registry (the writer owns + /// only the connection-local interest state). pub(super) fn set_terminal_interest( &self, snapshot: &freshell_protocol::client_messages::TerminalInterest, - ) -> Result<(), &'static str> { + ) -> Result, &'static str> { let mut queues = self.shared.queues.lock().expect("writer queue lock"); if queues.closed { return Err("Connection writer is closed"); } - if queues.interest.apply(snapshot)? { + if let Some(change) = queues.interest.apply(snapshot)? { let Queues { output, interest, .. } = &mut *queues; output.update_priorities(|id| interest.priority(id)); + drop(queues); + self.shared.ready.notify_one(); + Ok(Some(change)) + } else { + drop(queues); + Ok(None) } - drop(queues); - self.shared.ready.notify_one(); - Ok(()) } /// Pre-snapshot fallback: a client that never negotiated terminalInterestV1 @@ -413,6 +656,102 @@ impl WriterSender { .pending_bytes() .saturating_add(queues.in_flight_output_bytes) } + + /// The drain-side backpressure watermark (responsive-terminal-restore + /// W1): half the connection's output-queue budget. A producer that + /// has pushed the backlog to (or past) this mark must wait for the + /// writer pump to drain real frames before pushing more — the gate + /// that keeps the un-credited paced drain bounded by the connection + /// queue's actual consumption instead of its eviction behavior. + pub(crate) fn backlog_watermark(&self) -> usize { + (self.shared.output_limit / 2).max(1) + } + + /// Try to reserve `bytes` of DRAIN admission capacity under the + /// queue's own lock (round-5 finding 1, the atomic reserve-then-admit + /// gate): the grant accounts for the page the caller is about to + /// admit AND for every other in-flight drain reservation, so + /// concurrent pane drains can never double-book the watermark the way + /// the old check-then-act gate did (each woken drain observed the + /// same pre-admission backlog and every one admitted a full page on + /// top of it). `None` means not grantable now, or the writer is gone + /// (the caller's cancel path owns the exit). + /// + /// The grant limit is `max(watermark, bytes)`: a page that FITS the + /// watermark admits while the aggregate — backlog, every other + /// reservation, and this page — stays at-or-below the watermark, + /// leaving the cap's upper half as headroom for live traffic, so a + /// gated drain can never push the queue into evicting its own pages. + /// A page LARGER than the watermark (a degenerate queue/page + /// relationship the server boot's page-budget clamp exists to + /// prevent) still admits into a fully drained queue, so the gate can + /// never DEADLOCK, whatever the budget. + fn try_reserve_drain_admission(&self, bytes: usize) -> Option { + let mut queues = self.shared.queues.lock().expect("writer queue lock"); + if queues.closed { + return None; + } + let backlog = queues + .output + .pending_bytes() + .saturating_add(queues.in_flight_output_bytes); + let limit = self.backlog_watermark().max(bytes); + if backlog + .saturating_add(queues.drain_reserved) + .saturating_add(bytes) + > limit + { + return None; + } + queues.drain_reserved = queues.drain_reserved.saturating_add(bytes); + Some(DrainAdmission { + shared: Arc::clone(&self.shared), + bytes, + }) + } + + /// Wait until the connection's output backlog can absorb a drain page + /// of `bytes` (reserve-then-admit, round-5 finding 1): the reservation + /// is granted atomically against the backlog, every other in-flight + /// drain reservation, and the page itself. The `watch` channel is fed + /// by the writer pump on every completed frame send, and a `watch` + /// receiver retains unseen-change marks, so subscribe-then-check-then- + /// await can never miss a wakeup: this is REAL backpressure (the + /// sink's `push_server` admits without yielding; the reservation is + /// what bounds a producing drain by the connection queue's actual + /// consumption). `None` = the writer pump is gone; the caller's + /// cancel path owns the exit. + pub(crate) async fn reserve_drain_admission(&self, bytes: usize) -> Option { + loop { + if let Some(permit) = self.try_reserve_drain_admission(bytes) { + return Some(permit); + } + let mut rx = self.shared.backlog.subscribe(); + // Re-check after subscribing: a drain between the first check + // and the subscription is covered by the retained change mark. + if let Some(permit) = self.try_reserve_drain_admission(bytes) { + return Some(permit); + } + if rx.changed().await.is_err() { + // The writer pump is gone; the caller's cancel path owns + // the exit. + return None; + } + } + } + + /// Total successful socket sends completed on this connection (drain- + /// progress liveness, responsive-terminal-restore Workstream 3). The + /// catastrophic-backpressure monitor feeds its per-tick delta into its + /// window decision: sends are the ONLY progress signal (eviction and + /// supersede reduce queued bytes without being sends). + pub(super) fn completed_sends(&self) -> u64 { + self.shared + .queues + .lock() + .expect("writer queue lock") + .completed_sends + } } impl Sink for WriterSender { @@ -488,6 +827,35 @@ impl WriterPump { let (frame, bytes) = match delivery { Delivery::Frame { payload, bytes } => (payload, bytes), Delivery::Gap { terminal_id, range } => { + // Restore contract (responsive-terminal-restore): a + // negotiated connection's gap carries the terminal's + // CURRENT bounds, resolved at emission time. The source + // (registry-backed) takes the per-terminal lock, whose + // holders — subscriber fan-out, attach replays — acquire + // THIS admission lock under theirs, so resolving under + // the admission lock would invert the established lock + // order and can deadlock: release, resolve, re-acquire. + let bounds = match self.shared.gap_bounds.get() { + Some(source) => { + drop(queues); + let bounds = source(&terminal_id); + queues = self.shared.queues.lock().expect("writer queue lock"); + if queues.closed { + // A concurrent stop won the race while the + // admission lock was released. The queue is + // dead (Drop clears it) and the pump returns + // via the stop watch — never lease output + // past a stop. + return Ok(None); + } + bounds + } + None => None, + }; + // Spill observability (task-007 review M2) now fires at + // ADMISSION time, when the eviction happens — see + // `Queues::take_admission_spills`. Leasing the coalesced + // gap here is delivery, not a spill occurrence. let message = ServerMessage::TerminalOutputGap(freshell_protocol::TerminalOutputGap { terminal_id, @@ -496,6 +864,8 @@ impl WriterPump { from_seq: range.from_seq, to_seq: range.to_seq, reason: freshell_protocol::TerminalOutputGapReason::QueueOverflow, + head_seq: bounds.map(|b| b.head_seq), + oldest_retained_seq: bounds.map(|b| b.oldest_retained_seq), }); let json = serde_json::to_string(&message) .map_err(|_| WriterExit::SerializationFailed)?; @@ -509,12 +879,14 @@ impl WriterPump { queues.in_flight_output_bytes = bytes; queues.output.set_reserved_bytes(bytes); queues.controls_since_last_output = 0; - return Ok(Some(NextFrame { + let next = NextFrame { frame, control_bytes: 0, output_bytes: bytes, flushed: None, - })); + }; + drop(queues); + return Ok(Some(next)); } if let Some(control) = queues.controls.pop_front() { queues.controls_since_last_output += 1; @@ -534,6 +906,16 @@ impl WriterPump { queues.in_flight_output_bytes = queues.in_flight_output_bytes.saturating_sub(output_bytes); let reserved = queues.in_flight_output_bytes; queues.output.set_reserved_bytes(reserved); + // Drain-progress liveness: one successful socket send just completed + // (output or control — the pump serializes all sends, so either + // proves the socket accepted bytes). + queues.completed_sends = queues.completed_sends.saturating_add(1); + let backlog = queues.output.pending_bytes().saturating_add(reserved); + drop(queues); + // Drain-side backpressure signal (responsive-terminal-restore W1): + // publish the new backlog so gated producers (the paced drain + // task) wake as the queue drains. Sent with no queue lock held. + let _ = self.shared.backlog.send(backlog); } /// Generic over the real transport so tests can stop a flush at a precise @@ -644,12 +1026,16 @@ impl Drop for WriterPump { queues.control_bytes = 0; queues.in_flight_output_bytes = 0; queues.controls_since_last_output = 0; + queues.drain_reserved = 0; queues.output = DeliveryQueue::new( self.shared.output_limit, metadata_limit(self.shared.output_limit), ); queues.interest = InterestState::default(); } + // Wake any gated producers: the backlog is gone with the writer, + // and their own cancel paths own the exit. + let _ = self.shared.backlog.send(0); } } diff --git a/crates/freshell-ws/src/connection_writer_tests.rs b/crates/freshell-ws/src/connection_writer_tests.rs index 9db4c7e3b..3f71a4f9e 100644 --- a/crates/freshell-ws/src/connection_writer_tests.rs +++ b/crates/freshell-ws/src/connection_writer_tests.rs @@ -1,6 +1,6 @@ use super::*; use futures_util::task::AtomicWaker; -use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering}; #[derive(Default)] struct Capture { @@ -89,6 +89,483 @@ async fn join(task: tokio::task::JoinHandle) -> WriterExit { .unwrap() } +/// The paced drain's admission reservation (responsive-terminal-restore +/// W1, round-5): while the connection's output backlog cannot absorb a +/// page of `bytes` at/under the watermark, the reservation PENDS — the +/// sink itself (`push_server`) admits without yielding, so this gate is +/// what bounds a producing drain by the connection queue's REAL +/// consumption. The reservation grants only when the writer pump has +/// completed actual frame sends and the published backlog leaves room +/// for the page (backlog + reservation + page <= watermark). +#[tokio::test] +async fn drain_admission_waits_for_real_queue_consumption() { + let (sender, pump) = WriterSender::new(4096, 4096, Duration::from_secs(10)); + let capture = Arc::new(Capture::default()); + capture.block_flush.store(true, Ordering::SeqCst); + // Fill past the watermark (output_limit/2 = 2048 bytes): a blocked + // in-flight frame plus enough queued pages. + for seq in 1..=24 { + assert!(sender.push_server(output(seq))); + } + let task = tokio::spawn(pump.run(TestSink(Arc::clone(&capture)))); + started(&capture).await; + assert!( + sender.pending_output_bytes() >= sender.backlog_watermark(), + "the fixture holds the backlog at/above the watermark (pending {})", + sender.pending_output_bytes() + ); + let permit = sender.reserve_drain_admission(1024); + tokio::pin!(permit); + // The reservation is CLOSED: it must not resolve while the backlog + // cannot absorb the page. + assert!( + tokio::time::timeout(Duration::from_millis(100), &mut permit) + .await + .is_err(), + "the reservation stays closed while the backlog cannot absorb the page" + ); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!(queues.drain_reserved, 0, "nothing was reserved yet"); + } + // REAL consumption releases it: the in-flight frame's flush completes + // and `finish_frame` publishes the drained backlog. + unblock(&capture); + let permit = tokio::time::timeout(Duration::from_secs(2), permit) + .await + .expect("the reservation grants when the pump drains real frames"); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, 1024, + "the granted permit holds its bytes" + ); + } + drop(permit); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, 0, + "dropping the permit releases the reservation" + ); + } + sender.stop_without_close(); + let _ = join(task).await; +} + +/// Round-5 finding 1 (degenerate settings): a drain page LARGER than the +/// admission watermark (a queue/page relationship the boot clamp exists +/// to prevent — injected directly here) must still admit once the queue +/// fully drains: the reservation gate can never DEADLOCK, whatever the +/// budget. The oversize page admits into a fully drained queue only. +#[tokio::test] +async fn an_oversize_drain_page_admits_into_a_fully_drained_queue() { + let (sender, pump) = WriterSender::new(64 * 1024, 4096, Duration::from_secs(10)); + // The degenerate shape: the 64 KiB minimum queue (watermark 32 KiB) + // against the default-sized 128 KiB page budget. + assert_eq!(sender.backlog_watermark(), 32 * 1024); + let capture = Arc::new(Capture::default()); + capture.block_flush.store(true, Ordering::SeqCst); + assert!(sender.push_server(output(1))); + let task = tokio::spawn(pump.run(TestSink(Arc::clone(&capture)))); + started(&capture).await; + // The oversize reservation pends while ANY backlog stands (the + // in-flight frame alone blocks it): grantable only into a fully + // drained queue. + let permit = sender.reserve_drain_admission(128 * 1024); + tokio::pin!(permit); + assert!( + tokio::time::timeout(Duration::from_millis(100), &mut permit) + .await + .is_err(), + "the oversize reservation waits for a fully drained queue" + ); + unblock(&capture); + let permit = tokio::time::timeout(Duration::from_secs(2), permit) + .await + .expect("the oversize page admits once the queue is fully drained — no deadlock"); + drop(permit); + sender.stop_without_close(); + let _ = join(task).await; +} + +/// Round-2 finding F5: releasing a [`DrainAdmission`] must WAKE the tasks +/// waiting on the backlog watch channel — a drain that consumed the +/// available reservation and completed WITHOUT admitting (a `CaughtUp` or +/// `Gone` verdict: a retention gap instead of a page, a cancelled session) +/// frees admission capacity, and a concurrently gated drain must +/// re-evaluate immediately rather than sleeping until an unrelated socket +/// send or keepalive publishes the backlog. At the supported 64 KiB queue +/// floor two 32 KiB drains are exactly this shape (watermark 32 KiB, the +/// floor's clamped page budget 32 KiB): drain A holds the whole watermark; +/// drain B is gated; A completes without admitting. +/// +/// Bounded-assert discipline: the wake is observed VIA THE CHANNEL/state — +/// the reservation resolves without a single completed socket send (the +/// pump is never run), so the only possible waker is the release itself. +#[tokio::test] +async fn drain_admission_release_wakes_a_gated_drain_without_a_socket_send() { + let (sender, _pump) = WriterSender::new(64 * 1024, 4096, Duration::from_secs(10)); + assert_eq!( + sender.backlog_watermark(), + 32 * 1024, + "the reviewer's 64 KiB queue-floor shape" + ); + // Drain A consumes the whole watermark's reservation. + let permit_a = sender + .reserve_drain_admission(32 * 1024) + .await + .expect("drain A reserves against an empty queue"); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!(queues.drain_reserved, 32 * 1024); + } + // Drain B gates: A's reservation leaves no room for a second page. + let permit_b = sender.reserve_drain_admission(32 * 1024); + tokio::pin!(permit_b); + assert!( + tokio::time::timeout(Duration::from_millis(100), &mut permit_b) + .await + .is_err(), + "drain B stays gated while drain A holds the reservation" + ); + // Drain A completes WITHOUT admitting anything (the CaughtUp/Gone + // shape: no page was built, no frame was sent): its release alone must + // free and PUBLISH the capacity. + drop(permit_a); + let permit_b = tokio::time::timeout(Duration::from_secs(2), permit_b) + .await + .expect("the release alone wakes the gated drain — no socket send required") + .expect("the writer is alive"); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, + 32 * 1024, + "drain B now holds the released capacity" + ); + } + assert_eq!( + sender.completed_sends(), + 0, + "no socket send ever happened: the wake came from the reservation release" + ); + drop(permit_b); +} + +/// Round-5 finding 1 (Major), the reviewer's exact scenario: MULTIPLE pane +/// drains awakened together against a JUST-UNDER-WATERMARK backlog with +/// FULL-SIZE pages. The drain admission gate must account for the page it +/// is about to admit AND reserve admission capacity atomically, so +/// concurrent pane drains can never double-book the watermark. Observed +/// end state: ZERO queue_overflow evictions and the admitted aggregate +/// never exceeding the watermark + one page. +/// +/// The reviewer's numbers: the supported 256 KiB queue cap, the default +/// 128 KiB paced page budget (watermark = 128 KiB). +#[tokio::test] +async fn concurrent_full_size_drain_admissions_never_self_spill_the_queue() { + let events = crate::invariants::capture::capture(); + let (sender, pump) = WriterSender::new(256 * 1024, 4096, Duration::from_secs(10)); + let capture = Arc::new(Capture::default()); + capture.block_flush.store(true, Ordering::SeqCst); + let watermark = sender.backlog_watermark(); + assert_eq!(watermark, 128 * 1024, "the reviewer's 256 KiB queue shape"); + + // One FULL-SIZE paced page: a single output frame whose serialized + // size sits just under the default 128 KiB page budget. + let page = |terminal_id: &'static str, seq: i64| { + let mut message = output(seq); + if let ServerMessage::TerminalOutput(frame) = &mut message { + frame.terminal_id = terminal_id.to_string(); + frame.data = "P".repeat(128 * 1024 - 256); + } + message + }; + let page_bytes = serde_json::to_string(&page("drain-admit-probe", 1)) + .unwrap() + .len(); + assert!( + page_bytes < 128 * 1024, + "the fixture's page is a realistic full-size page (serialized {page_bytes})" + ); + + // Fill the backlog to JUST UNDER the watermark (the reviewer's + // "just-under-watermark" wake state) with the socket blocked. + let mut seq = 0; + while sender.pending_output_bytes() < watermark - 2048 { + seq += 1; + assert!(sender.push_server(named_output("drain-admit-fill", seq))); + } + let task = tokio::spawn(pump.run(TestSink(Arc::clone(&capture)))); + started(&capture).await; + let backlog_at_wake = sender.pending_output_bytes(); + assert!( + backlog_at_wake < watermark, + "the fixture wakes the drains just under the watermark ({backlog_at_wake})" + ); + let spill_count = || { + events + .lock() + .unwrap() + .iter() + .filter(|e| { + e.message.contains("queue_overflow_spill") + && e.fields + .get("terminal_id") + .is_some_and(|id| id.starts_with("drain-admit")) + }) + .count() + }; + assert_eq!(spill_count(), 0, "no spill before the drains wake"); + + // THE CONCURRENT WAKE: three pane drains reserve their full-size page + // admissions together against the just-under-watermark backlog. The + // reserve-then-admit gate must grant NONE of them while the backlog + // stands (each reservation accounts for its own page — the page can + // no longer ride on top of the backlog unaccounted). + let max_observed = Arc::new(AtomicUsize::new(0)); + let mut drains = Vec::new(); + for (drain, terminal_id) in ["drain-admit-a", "drain-admit-b", "drain-admit-c"] + .into_iter() + .enumerate() + { + let sender = sender.clone(); + let max_observed = Arc::clone(&max_observed); + drains.push(tokio::spawn(async move { + let permit = sender + .reserve_drain_admission(page_bytes) + .await + .expect("the writer is alive"); + assert!(sender.push_server(page(terminal_id, 1 + drain as i64))); + let observed = sender.pending_output_bytes(); + max_observed.fetch_max(observed, Ordering::SeqCst); + drop(permit); + })); + } + tokio::time::sleep(Duration::from_millis(100)).await; + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, 0, + "no reservation is granted against a backlog the page cannot join \ + without crossing the watermark" + ); + } + assert_eq!( + spill_count(), + 0, + "no page was admitted while the backlog stood (the socket is blocked)" + ); + assert_eq!( + max_observed.load(Ordering::SeqCst), + 0, + "no drain admitted anything before the queue drained" + ); + + // REAL consumption releases the admissions: the pump drains the + // backlog, the reservations grant, and every drain completes its page. + unblock(&capture); + for drain in drains { + tokio::time::timeout(Duration::from_secs(5), drain) + .await + .expect("each reserved drain completes") + .unwrap(); + } + + // THE BOUND: zero drain-induced queue_overflow evictions, and the + // admitted aggregate never exceeded the watermark + one page. + let observed = max_observed.load(Ordering::SeqCst); + assert_eq!( + spill_count(), + 0, + "THE BOUND: concurrent pane drains must never evict the connection's own pages \ + (observed aggregate {observed}B vs watermark {watermark}B + one page {page_bytes}B)" + ); + assert!( + observed <= watermark + page_bytes, + "THE BOUND: the admitted aggregate ({observed}) must never exceed the watermark \ + ({watermark}) + one page ({page_bytes})", + ); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, 0, + "every reservation was released after its page became real backlog" + ); + } + + sender.stop_without_close(); + let _ = join(task).await; +} + +/// E2R1 finding 2(b) — the SUB-CAP twin of +/// [`concurrent_full_size_drain_admissions_never_self_spill_the_queue`]: +/// a client's `replayPageBytes` request far below the production frame +/// size makes every page the builder's ATOMIC single-frame result, so +/// the drain's reservation must be max(requested budget, the atomic +/// page ceiling) — reserving only the requested bytes lets all three +/// concurrent pane drains be granted together against the same +/// just-under-watermark backlog, each admitting a full atomic page on +/// top of it, and the queue evicts the drains' own pages. The numbers +/// mirror the full-size twin's shape at the supported 64 KiB queue +/// floor: watermark 32 KiB, requested budget 2048 bytes, atomic pages +/// ~8.4 KiB (a maximal PTY read under the fragment cap). +#[tokio::test] +async fn concurrent_sub_cap_drain_admissions_never_under_reserve_the_atomic_page() { + let events = crate::invariants::capture::capture(); + let (sender, pump) = WriterSender::new(64 * 1024, 4096, Duration::from_secs(10)); + let capture = Arc::new(Capture::default()); + capture.block_flush.store(true, Ordering::SeqCst); + let watermark = sender.backlog_watermark(); + assert_eq!( + watermark, + 32 * 1024, + "the supported 64 KiB queue floor's watermark" + ); + + // One ATOMIC page: a single output frame whose serialized size is a + // full PTY read (~8 KiB data) — far above the 2048-byte requested + // page budget, and within the fragment cap (frames are + // pre-fragmented, so this is the worst case production can stage). + let page = |terminal_id: &'static str, seq: i64| { + let mut message = output(seq); + if let ServerMessage::TerminalOutput(frame) = &mut message { + frame.terminal_id = terminal_id.to_string(); + frame.data = "P".repeat(8 * 1024); + } + message + }; + let page_bytes = serde_json::to_string(&page("drain-subcap-probe", 1)) + .unwrap() + .len(); + let ceiling = freshell_terminal::paced_atomic_page_serialized_ceiling(); + assert!( + page_bytes <= ceiling, + "the fixture's atomic page is a real single frame under the ceiling \ + ({page_bytes} <= {ceiling})" + ); + assert!( + page_bytes > 2048, + "the atomic page provably exceeds the sub-cap request (the under-booked page)" + ); + + // Fill the backlog to JUST UNDER the grant threshold for three + // 2048-byte reservations (the wake state where the pre-fix gate + // granted all of them together) with the socket blocked. + let admission_bytes = crate::paced_replay::drain_admission_bytes(2048); + let grant_backlog_cap = watermark.saturating_sub(3 * 2048); + let mut seq = 0; + while sender.pending_output_bytes() < grant_backlog_cap.saturating_sub(4096) { + seq += 1; + assert!(sender.push_server(named_output("drain-subcap-fill", seq))); + } + let task = tokio::spawn(pump.run(TestSink(Arc::clone(&capture)))); + started(&capture).await; + let backlog_at_wake = sender.pending_output_bytes(); + assert!( + backlog_at_wake < grant_backlog_cap, + "the fixture wakes the drains just under the {grant_backlog_cap}-byte grant \ + threshold ({backlog_at_wake})" + ); + let spill_count = || { + events + .lock() + .unwrap() + .iter() + .filter(|e| { + e.message.contains("queue_overflow_spill") + && e.fields + .get("terminal_id") + .is_some_and(|id| id.starts_with("drain-subcap")) + }) + .count() + }; + assert_eq!(spill_count(), 0, "no spill before the drains wake"); + + // THE CONCURRENT WAKE: three pane drains reserve their sub-cap + // admissions together against the just-under-watermark backlog. The + // gate must grant NONE of them while the backlog stands — a + // reservation of only the requested 2048 bytes would admit all + // three, and their ~8.4 KiB atomic pages (+ the standing backlog) + // would blow past the 64 KiB queue and evict the drains' own pages. + let max_observed = Arc::new(AtomicUsize::new(0)); + let mut drains = Vec::new(); + for (drain, terminal_id) in ["drain-subcap-a", "drain-subcap-b", "drain-subcap-c"] + .into_iter() + .enumerate() + { + let sender = sender.clone(); + let max_observed = Arc::clone(&max_observed); + drains.push(tokio::spawn(async move { + let permit = sender + .reserve_drain_admission(admission_bytes) + .await + .expect("the writer is alive"); + assert!(sender.push_server(page(terminal_id, 1 + drain as i64))); + let observed = sender.pending_output_bytes(); + max_observed.fetch_max(observed, Ordering::SeqCst); + drop(permit); + })); + } + tokio::time::sleep(Duration::from_millis(100)).await; + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, 0, + "no sub-cap reservation is granted against a backlog the atomic page \ + cannot join without crossing the watermark" + ); + } + assert_eq!( + spill_count(), + 0, + "no atomic page was admitted while the backlog stood (the socket is blocked)" + ); + assert_eq!( + max_observed.load(Ordering::SeqCst), + 0, + "no drain admitted anything before the queue drained" + ); + + // REAL consumption releases the admissions: the pump drains the + // backlog, the reservations grant (serialized by their atomic-page + // size), and every drain completes its page. + unblock(&capture); + for drain in drains { + tokio::time::timeout(Duration::from_secs(5), drain) + .await + .expect("each reserved sub-cap drain completes") + .unwrap(); + } + + // THE BOUND: zero drain-induced queue_overflow evictions, and the + // admitted aggregate never exceeded the watermark + one atomic page. + let observed = max_observed.load(Ordering::SeqCst); + assert_eq!( + spill_count(), + 0, + "THE BOUND: concurrent sub-cap pane drains must never evict the connection's \ + own pages by under-reserving their atomic pages (observed aggregate \ + {observed}B vs watermark {watermark}B + one atomic page {page_bytes}B)" + ); + assert!( + observed <= watermark + page_bytes, + "THE BOUND: the admitted aggregate ({observed}) must never exceed the \ + watermark ({watermark}) + one atomic page ({page_bytes})" + ); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.drain_reserved, 0, + "every reservation was released after its page became real backlog" + ); + } + + sender.stop_without_close(); + let _ = join(task).await; +} + #[tokio::test] async fn blocked_flush_does_not_block_producers_and_is_still_accounted() { let (mut sender, pump) = WriterSender::new(4096, 4096, Duration::from_secs(10)); @@ -467,6 +944,211 @@ async fn overflow_stops_a_pending_flush_without_waiting_for_send_timeout() { assert_eq!(text_frames(&capture).len(), 1); } +/// An oversized indivisible OUTPUT frame — larger than the ENTIRE queue cap — +/// spills immediately (output frames are themselves evictable, so the cap +/// evicts the oversize frame the moment it is admitted) instead of +/// accumulating unbounded bytes or closing the connection: the push +/// succeeds, the loss is the honest queue-overflow gap covering exactly that +/// frame, and pending bytes stay bounded at zero. (The CONTROL lane's +/// one-oversize-frame grace is separate — see +/// `overflow_stops_a_pending_flush_without_waiting_for_send_timeout`.) +#[tokio::test] +async fn oversized_indivisible_output_frame_spills_instead_of_accumulating() { + let (sender, pump) = overflow_writer(); + let probe = serde_json::to_string(&output(1)).unwrap().len(); + let mut huge = output(1); + if let ServerMessage::TerminalOutput(frame) = &mut huge { + // Serialized length comfortably exceeds the whole cap. + frame.data = "X".repeat(probe * 2); + } + assert!( + sender.push_server(huge), + "admission must survive an oversize frame (it spills, never wedges)" + ); + assert_eq!( + sender.pending_output_bytes(), + 0, + "the oversize frame must not accumulate: the queue stays bounded" + ); + let next = pump.take_next().unwrap().unwrap(); + let gap: serde_json::Value = serde_json::from_str(&leased_text(&next.frame)).unwrap(); + assert_eq!(gap["type"], "terminal.output.gap"); + assert_eq!(gap["reason"], "queue_overflow"); + assert_eq!( + gap["fromSeq"], 1, + "the gap covers exactly the oversize frame" + ); + assert_eq!(gap["toSeq"], 1); + pump.finish_frame(next.output_bytes, next.control_bytes); + assert!( + pump.take_next().unwrap().is_none(), + "nothing else was retained behind the oversize frame" + ); +} + +/// Task-007 review M2 (landed by task-010): the rate-limited +/// `ws.terminal_stream.queue_overflow_spill` event must fire at ADMISSION +/// time — the moment the eviction happens — not at gap-lease time, so a +/// connection that spills and then dies while backlogged (its coalesced +/// gap never surfaces from the stuck queue) still leaves spill evidence in +/// the live log. The pump is NEVER run until after the emit assertions: +/// nothing is leased, so a lease-time emit would produce no event at all. +#[tokio::test] +async fn spill_observability_fires_at_admission_time_even_if_the_gap_is_never_leased() { + let events = crate::invariants::capture::capture(); + // Cap sized to exactly ONE of THIS terminal's frames (the longer + // terminal id makes each serialized frame larger than `output(1)`'s + // probe, so `overflow_writer()` would self-evict frame 1). + let (sender, pump) = { + let probe = serde_json::to_string(&named_output("spill-admission", 1)) + .unwrap() + .len(); + WriterSender::new(probe, 4096, Duration::from_secs(10)) + }; + let spill_events = || { + events + .lock() + .unwrap() + .iter() + .filter(|e| { + e.message.contains("queue_overflow_spill") + && e.fields.get("terminal_id").map(String::as_str) == Some("spill-admission") + }) + .count() + }; + // Frame 1 alone fits the cap (probe bytes); frame 2's admission evicts + // frame 1; frame 3's admission evicts frame 2. + assert!(sender.push_server(named_output("spill-admission", 1))); + { + let queues = sender.shared.queues.lock().unwrap(); + assert!( + queues.spill_last_logged.is_none(), + "no spill bookkeeping before any eviction" + ); + } + assert_eq!(spill_events(), 0, "no eviction has happened yet"); + assert!(sender.push_server(named_output("spill-admission", 2))); + // The FIRST eviction emits immediately (rate limiter idle) with the + // evicted frame's own range — without any pump lease. + { + let queues = sender.shared.queues.lock().unwrap(); + assert!( + queues.spill_last_logged.is_some(), + "the eviction at admission time must record the spill" + ); + } + { + let captured: Vec<_> = events + .lock() + .unwrap() + .iter() + .filter(|e| { + e.message.contains("queue_overflow_spill") + && e.fields.get("terminal_id").map(String::as_str) == Some("spill-admission") + }) + .cloned() + .collect(); + assert_eq!( + captured.len(), + 1, + "admission-time eviction must emit exactly one spill event with no pump lease: {captured:?}" + ); + assert_eq!( + captured[0].fields.get("from_seq").map(String::as_str), + Some("1") + ); + assert_eq!( + captured[0].fields.get("to_seq").map(String::as_str), + Some("1") + ); + assert_eq!( + captured[0].fields.get("suppressed").map(String::as_str), + Some("0") + ); + } + // A second eviction inside the rate-limit window folds into the + // `suppressed` counter (one event per window). + assert!(sender.push_server(named_output("spill-admission", 3))); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.spill_suppressed, 1, + "the in-window eviction folds into suppressed" + ); + } + // NOW lease everything (including the coalesced [1..=2] gap): gap + // DELIVERY is not a spill occurrence — no new event, no extra fold. + let mut leased_gap = None; + while let Some(next) = pump.take_next().unwrap() { + let text = leased_text(&next.frame); + if text.contains("terminal.output.gap") { + leased_gap = Some(serde_json::from_str::(&text).unwrap()); + } + pump.finish_frame(next.output_bytes, next.control_bytes); + } + let gap = leased_gap.expect("the coalesced gap must lease"); + assert_eq!(gap["fromSeq"], 1); + assert_eq!(gap["toSeq"], 2); + { + let queues = sender.shared.queues.lock().unwrap(); + assert_eq!( + queues.spill_suppressed, 1, + "leasing the gap is not a spill occurrence" + ); + } + assert_eq!( + spill_events(), + 1, + "delivery of the gap must not emit a second spill event" + ); +} + +/// Drain-progress liveness (responsive-terminal-restore Workstream 3): the +/// completed-send counter moves ONLY on successful socket sends. Supersede +/// and eviction shrink queued bytes without touching it, and a leased frame +/// counts only once its flush finishes. +#[tokio::test] +async fn completed_sends_counts_only_successful_socket_sends() { + let (mut sender, pump) = WriterSender::new(1 << 20, 1 << 20, Duration::from_secs(10)); + assert_eq!(sender.completed_sends(), 0); + // Two queued output frames and one control. + assert!(sender.push_server(output(1))); + assert!(sender.push_server(output(2))); + sender.send(notice("control")).await.unwrap(); + // A superseding attach DISCARDS another terminal's queued output — a + // byte reduction that is not a send and must not count as progress. + assert!(sender.push_server(named_output("victim", 1))); + assert!(sender.push_server( + serde_json::from_value(serde_json::json!({ + "type":"terminal.attach.ready", "terminalId":"victim", "attachRequestId":"a2", + "streamId":"stream", "headSeq":1, "replayFromSeq":1, "replayToSeq":1 + })) + .unwrap() + )); + assert_eq!( + sender.completed_sends(), + 0, + "admission, eviction and supersede are not sends" + ); + // Drive the pump: every take_next + finish_frame pair is one successful + // send; the discarded victim frame never leases. + let mut sends = 0u64; + while let Some(next) = pump.take_next().unwrap() { + pump.finish_frame(next.output_bytes, next.control_bytes); + sends += 1; + assert_eq!( + sender.completed_sends(), + sends, + "exactly one count per completed send" + ); + } + assert_eq!( + sends, 4, + "control + superseding attach.ready + two output frames were sent" + ); + assert_eq!(sender.completed_sends(), 4); +} + fn named_output(terminal_id: &str, seq: i64) -> ServerMessage { let mut message = output(seq); if let ServerMessage::TerminalOutput(frame) = &mut message { @@ -483,6 +1165,7 @@ fn interest( revision, focused_terminal_id: focused.map(str::to_string), visible_terminal_ids: visible.iter().map(|s| s.to_string()).collect(), + claimed_terminal_ids: None, } } fn taken_terminal(pump: &WriterPump) -> String { @@ -562,3 +1245,194 @@ async fn attach_priority_works_for_clients_without_interest_capability() { sender.push_server(named_output("visible", 1)); assert_eq!(taken_terminal(&pump), "visible"); } + +/// Force exactly one queue-overflow eviction: the output limit admits the +/// first frame alone, so the second push evicts the first and materializes +/// its gap. The limit is derived from the probe frame's serialized size so +/// the eviction is deterministic without hardcoding byte counts. +fn overflow_writer() -> (WriterSender, WriterPump) { + let probe = serde_json::to_string(&output(1)).unwrap().len(); + WriterSender::new(probe, 4096, Duration::from_secs(10)) +} + +#[tokio::test] +async fn queue_overflow_gap_carries_restore_bounds_on_paced_connections() { + // Restore contract (responsive-terminal-restore): a connection that + // negotiated pacedTerminalReplayV1 (modeled here by an installed + // gap-bounds source) sees its queue-overflow gaps stamped with the + // terminal's CURRENT headSeq + oldestRetainedSeq, resolved at + // gap-emission (lease) time. + let (sender, pump) = overflow_writer(); + sender.set_paced_replay_gap_bounds(Arc::new(|_| { + Some(freshell_terminal::ReplayBounds { + head_seq: 421, + oldest_retained_seq: 7, + }) + })); + assert!(sender.push_server(output(1))); + assert!(sender.push_server(output(2))); // evicts output(1) -> queue_overflow gap + let next = pump.take_next().unwrap().unwrap(); + let gap: serde_json::Value = serde_json::from_str(&leased_text(&next.frame)).unwrap(); + assert_eq!(gap["type"], "terminal.output.gap"); + assert_eq!(gap["reason"], "queue_overflow"); + assert_eq!(gap["fromSeq"], 1); + assert_eq!(gap["toSeq"], 1); + assert_eq!(gap["headSeq"], 421, "negotiated gap carries headSeq: {gap}"); + assert_eq!( + gap["oldestRetainedSeq"], 7, + "negotiated gap carries oldestRetainedSeq: {gap}" + ); + pump.finish_frame(next.output_bytes, next.control_bytes); + // The evicted frame's successor still delivers. + let next = pump.take_next().unwrap().unwrap(); + assert!(leased_text(&next.frame).contains("data-2")); + pump.finish_frame(next.output_bytes, next.control_bytes); +} + +#[tokio::test] +async fn queue_overflow_gap_without_paced_negotiation_keeps_the_frozen_shape() { + // Compatibility invariant (load-bearing): a connection that did NOT + // negotiate sees gap frames byte-identical to the pre-contract wire — + // no headSeq, no oldestRetainedSeq, and exactly the frozen key set. + let (sender, pump) = overflow_writer(); + assert!(sender.push_server(output(1))); + assert!(sender.push_server(output(2))); // evicts output(1) -> queue_overflow gap + let next = pump.take_next().unwrap().unwrap(); + let gap: serde_json::Value = serde_json::from_str(&leased_text(&next.frame)).unwrap(); + assert_eq!(gap["type"], "terminal.output.gap"); + let mut keys: Vec<&str> = gap + .as_object() + .unwrap() + .keys() + .map(String::as_str) + .collect(); + keys.sort_unstable(); + assert_eq!( + keys, + vec![ + "attachRequestId", + "fromSeq", + "reason", + "streamId", + "terminalId", + "toSeq", + "type", + ], + "the non-negotiated gap frame keeps the pre-contract key set: {gap}" + ); + pump.finish_frame(next.output_bytes, next.control_bytes); +} + +/// A directly pushed `terminal.output.gap` (the paced path emits retention +/// gaps through `push_server`, unlike the queue's own lease-time gap +/// materialization): the gap is sequenced WITH the terminal's output — it +/// must lease strictly AFTER output admitted before it, never jumping ahead +/// on the preemptive control lane, and it must never be evictable or +/// byte-charged. +#[tokio::test] +async fn server_pushed_restore_gap_stays_ordered_with_the_terminals_output() { + let (sender, pump) = WriterSender::new(100_000, 4096, Duration::from_secs(10)); + assert!(sender.push_server(output(1))); + let gap = ServerMessage::TerminalOutputGap(freshell_protocol::TerminalOutputGap { + terminal_id: "term".into(), + stream_id: "stream".into(), + attach_request_id: Some("attach".into()), + from_seq: 1, + to_seq: 9, + reason: freshell_protocol::TerminalOutputGapReason::ReplayWindowExceeded, + head_seq: Some(12), + oldest_retained_seq: Some(10), + }); + assert!(sender.push_server(gap)); + + let first = pump.take_next().unwrap().unwrap(); + assert!( + leased_text(&first.frame).contains("data-1"), + "output admitted BEFORE the gap leases first" + ); + pump.finish_frame(first.output_bytes, first.control_bytes); + + let second = pump.take_next().unwrap().unwrap(); + let gap_json: serde_json::Value = serde_json::from_str(&leased_text(&second.frame)).unwrap(); + assert_eq!(gap_json["type"], "terminal.output.gap"); + assert_eq!( + second.output_bytes, 0, + "the gap leases as a zero-weight sequenced control (never byte-charged)" + ); + pump.finish_frame(second.output_bytes, second.control_bytes); +} + +/// The server-pushed restore gap must survive queue overflow eviction: like +/// `terminal.exit`, it is a non-evictable sequenced control — the byte cap +/// evicts payload frames, never the gap. +#[tokio::test] +async fn server_pushed_restore_gap_is_not_evictable_under_overflow() { + let (sender, pump) = overflow_writer(); + assert!(sender.push_server(output(1))); + let gap = ServerMessage::TerminalOutputGap(freshell_protocol::TerminalOutputGap { + terminal_id: "term".into(), + stream_id: "stream".into(), + attach_request_id: Some("attach".into()), + from_seq: 2, + to_seq: 2, + reason: freshell_protocol::TerminalOutputGapReason::ReplayWindowExceeded, + head_seq: None, + oldest_retained_seq: None, + }); + assert!(sender.push_server(gap)); + // Overflow: output(2) does not fit and evicts the OLDEST EVICTABLE entry — + // output(1) — never the non-evictable gap. + assert!(sender.push_server(output(2))); + + let first = pump.take_next().unwrap().unwrap(); + let first_json: serde_json::Value = serde_json::from_str(&leased_text(&first.frame)).unwrap(); + assert_eq!( + first_json["type"], "terminal.output.gap", + "the queue-overflow gap head leases first (the evicted output(1))" + ); + pump.finish_frame(first.output_bytes, first.control_bytes); + + let second = pump.take_next().unwrap().unwrap(); + let second_json: serde_json::Value = serde_json::from_str(&leased_text(&second.frame)).unwrap(); + assert_eq!( + second_json["type"], "terminal.output.gap", + "the server-pushed restore gap survives the overflow eviction" + ); + assert_eq!(second_json["fromSeq"], 2); + pump.finish_frame(second.output_bytes, second.control_bytes); + + let third = pump.take_next().unwrap().unwrap(); + assert!(leased_text(&third.frame).contains("data-2")); + pump.finish_frame(third.output_bytes, third.control_bytes); +} + +/// Task-2 review follow-up (Minor 2): a writer stop that lands while the Gap +/// arm has RELEASED the admission lock to resolve negotiated bounds (after +/// the pop, before the materialize) must yield NO flushed frame — the +/// re-acquire re-check of `closed` aborts the lease and the pump exits via +/// the stop watch, leaving the popped gap unflushed. +#[tokio::test] +async fn writer_stop_landing_during_gap_bounds_resolution_flushes_no_frame() { + let (sender, pump) = overflow_writer(); + let stopping = sender.clone(); + sender.set_paced_replay_gap_bounds(Arc::new(move |_| { + // The stop lands mid-resolution: the gap was already popped and the + // admission lock released. + stopping.stop_without_close(); + Some(freshell_terminal::ReplayBounds { + head_seq: 9, + oldest_retained_seq: 2, + }) + })); + assert!(sender.push_server(output(1))); + assert!(sender.push_server(output(2))); // evicts output(1) -> queue gap + + let capture = Arc::new(Capture::default()); + let task = tokio::spawn(pump.run(TestSink(Arc::clone(&capture)))); + assert_eq!(join(task).await, WriterExit::Stopped); + assert!( + text_frames(&capture).is_empty(), + "a stop before the lease must leave nothing flushed" + ); + assert_eq!(sender.pending_output_bytes(), 0); +} diff --git a/crates/freshell-ws/src/lib.rs b/crates/freshell-ws/src/lib.rs index d8bb107a0..2b6dcc728 100644 --- a/crates/freshell-ws/src/lib.rs +++ b/crates/freshell-ws/src/lib.rs @@ -56,6 +56,7 @@ pub mod opencode_association; pub mod opencode_lane; pub mod opencode_signal; pub mod origin; +pub(crate) mod paced_replay; pub mod pane_identity_binder; pub mod pane_ledger; pub mod reconcile; @@ -606,7 +607,7 @@ pub fn stuck_window_ms_from_env() -> i64 { /// would lose scrollback). On a truly fresh boot the registry is empty, so this stays /// byte-identical to the clean-boot handshake the oracle's T0/determinism tiers pin. pub async fn build_handshake(state: &WsState) -> Vec { - build_handshake_with_capabilities(state, false, false, false).await + build_handshake_with_capabilities(state, false, false, false, false, false).await } /// [`build_handshake`], parameterized on the connection's negotiated @@ -629,6 +630,8 @@ pub async fn build_handshake_with_capabilities( pane_reconcile_v1: bool, pane_reconcile_fresh_agent_v1: bool, terminal_interest_v1: bool, + paced_terminal_replay_v1: bool, + terminal_lifetime_claim_v1: bool, ) -> Vec { let boot_id = state.boot_id.as_ref().clone(); // kata b8ke Task 4 (reconnect-owner discovery, T1 rec A3): replay current @@ -682,11 +685,15 @@ pub async fn build_handshake_with_capabilities( build_id: ready_build_id(), capabilities: (pane_reconcile_v1 || pane_reconcile_fresh_agent_v1 - || terminal_interest_v1) + || terminal_interest_v1 + || paced_terminal_replay_v1 + || terminal_lifetime_claim_v1) .then_some(freshell_protocol::ReadyCapabilities { pane_reconcile_v1: pane_reconcile_v1.then_some(true), pane_reconcile_fresh_agent_v1: pane_reconcile_fresh_agent_v1.then_some(true), terminal_interest_v1: terminal_interest_v1.then_some(true), + paced_terminal_replay_v1: paced_terminal_replay_v1.then_some(true), + terminal_lifetime_claim_v1: terminal_lifetime_claim_v1.then_some(true), }), }), ServerMessage::SettingsUpdated(SettingsUpdated { @@ -906,6 +913,26 @@ async fn handle_socket( .and_then(serde_json::Value::as_bool) .unwrap_or(false); + // Responsive-terminal-restore Workstream 1 (paced replay negotiation): + // same opt-in gate as `paneReconcileV1` — the `ready` echo appears only + // when the client's `hello` opted in, so a frozen client's handshake + // stays byte-for-byte unchanged. + let paced_terminal_replay_v1 = value + .get("capabilities") + .and_then(|c| c.get("pacedTerminalReplayV1")) + .and_then(|v| v.as_bool()) + .unwrap_or(false); + + // Responsive-terminal-restore Workstream 1 (hidden-pane lifetime claims): + // same opt-in gate — the echo appears only when the client's `hello` + // opted in, and only then may its `terminal.interest` snapshots carry + // `claimedTerminalIds`. + let terminal_lifetime_claim_v1 = value + .get("capabilities") + .and_then(|c| c.get("terminalLifetimeClaimV1")) + .and_then(|v| v.as_bool()) + .unwrap_or(false); + // Authenticated: emit the ordered handshake. CFG-12: the builder is // async + per-connection so its `settings.updated` frame resolves the // LIVE settings tree (see `build_handshake_with_capabilities`). @@ -914,6 +941,8 @@ async fn handle_socket( pane_reconcile_v1, pane_reconcile_fresh_agent_v1, terminal_interest_v1, + paced_terminal_replay_v1, + terminal_lifetime_claim_v1, ) .await { @@ -969,12 +998,14 @@ async fn handle_socket( &state, bcast_rx, terminal_output_batch_v1, + paced_terminal_replay_v1, ui_screenshot_v1, pane_reconcile_v1, pane_reconcile_fresh_agent_v1, origin_kind, conn_identity, terminal_interest_v1, + terminal_lifetime_claim_v1, ) .await; } @@ -1158,7 +1189,8 @@ mod tests { #[tokio::test] async fn handshake_advertises_pane_reconcile_only_when_negotiated() { let s = state(); - let negotiated = build_handshake_with_capabilities(&s, true, false, false).await; + let negotiated = + build_handshake_with_capabilities(&s, true, false, false, false, false).await; let ready = serde_json::to_value(&negotiated[0]).unwrap(); assert_eq!( ready["capabilities"], @@ -1172,11 +1204,81 @@ mod tests { "non-negotiating hello must not change ready's shape: {ready}" ); // Same shape as an explicit `false` negotiation. - let unnegotiated = build_handshake_with_capabilities(&s, false, false, false).await; + let unnegotiated = + build_handshake_with_capabilities(&s, false, false, false, false, false).await; let ready2 = serde_json::to_value(&unnegotiated[0]).unwrap(); assert!(ready2.get("capabilities").is_none()); } + /// Responsive-terminal-restore Workstream 1 (paced replay negotiation): + /// `ready.capabilities.pacedTerminalReplayV1` is advertised ONLY for a + /// hello that opted in — a non-negotiating client's handshake stays + /// byte-identical to the pre-capability shape (frozen-client inertness). + #[tokio::test] + async fn handshake_advertises_paced_terminal_replay_only_when_negotiated() { + let s = state(); + let negotiated = + build_handshake_with_capabilities(&s, false, false, false, true, false).await; + let ready = serde_json::to_value(&negotiated[0]).unwrap(); + assert_eq!( + ready["capabilities"], + serde_json::json!({ "pacedTerminalReplayV1": true }) + ); + + // Non-paced negotiations keep the capabilities object byte-identical + // to today's output — no paced key is invented. + let pane_only = + build_handshake_with_capabilities(&s, true, false, false, false, false).await; + let ready = serde_json::to_value(&pane_only[0]).unwrap(); + assert_eq!( + ready["capabilities"], + serde_json::json!({ "paneReconcileV1": true }), + "a non-paced negotiation must not invent pacedTerminalReplayV1: {ready}" + ); + + // No negotiation at all: no capabilities object on the wire. + let default = build_handshake(&s).await; + let ready = serde_json::to_value(&default[0]).unwrap(); + assert!( + ready.get("capabilities").is_none(), + "non-negotiating hello must not change ready's shape: {ready}" + ); + } + + /// Responsive-terminal-restore Workstream 1 (hidden-pane lifetime claims): + /// `ready.capabilities.terminalLifetimeClaimV1` is advertised ONLY for a + /// hello that opted in — non-claiming negotiations and default hellos + /// stay byte-identical to today's shapes. + #[tokio::test] + async fn handshake_advertises_terminal_lifetime_claim_only_when_negotiated() { + let s = state(); + let negotiated = + build_handshake_with_capabilities(&s, false, false, true, false, true).await; + let ready = serde_json::to_value(&negotiated[0]).unwrap(); + assert_eq!( + ready["capabilities"], + serde_json::json!({ "terminalInterestV1": true, "terminalLifetimeClaimV1": true }) + ); + + // A claim-less negotiation must not invent the key. + let interest_only = + build_handshake_with_capabilities(&s, false, false, true, false, false).await; + let ready = serde_json::to_value(&interest_only[0]).unwrap(); + assert_eq!( + ready["capabilities"], + serde_json::json!({ "terminalInterestV1": true }), + "a non-claim negotiation must not invent terminalLifetimeClaimV1: {ready}" + ); + + // No negotiation at all: no capabilities object on the wire. + let default = build_handshake(&s).await; + let ready = serde_json::to_value(&default[0]).unwrap(); + assert!( + ready.get("capabilities").is_none(), + "non-negotiating hello must not change ready's shape: {ready}" + ); + } + #[tokio::test] async fn handshake_is_ordered_with_shared_bootid() { let msgs = build_handshake(&state()).await; diff --git a/crates/freshell-ws/src/paced_replay.rs b/crates/freshell-ws/src/paced_replay.rs new file mode 100644 index 000000000..5f8a09d12 --- /dev/null +++ b/crates/freshell-ws/src/paced_replay.rs @@ -0,0 +1,1214 @@ +//! Responsive-terminal-restore Workstream 1 — the connection-side pacing +//! coordinator for negotiated (`pacedTerminalReplayV1`) terminal replay. +//! +//! The registry owns the page reads ([`freshell_terminal::TerminalRegistry`] +//! `attach`'s paced start, `next_replay_page`, `complete_paced_tail`) and +//! the per-subscriber deferral; THIS module owns the session state the wire +//! protocol needs: the fixed catch-up target, the production cursor, the +//! credited-consumed window that gates continuation credits, and the two +//! execution sites that turn a valid [`TerminalReplayCredit`] into the +//! next page — the CREDITED replay phase (inline, one page per credit, +//! dispatcher-owned) and the UN-CREDITED drain (a spawned task that pages +//! the accumulated live range toward a FIXED target captured once at +//! drain start, so the connection dispatcher stays free to service input +//! and other panes for the drain's whole duration). +//! +//! Credit rules (all under the connection's negotiated capability; ignored +//! otherwise): a credit whose `attachRequestId` does not match the ACTIVE +//! session is a stale generation (superseded, completed, or DRAINING — +//! ignored); a `consumedSeq` outside `(credited, page_end]` is out of +//! window (ignored — no double-grant, no phantom grant past the last-sent +//! page). The grant is STRICTLY page-end: an in-window value below the +//! page end is a partial consumption report (observed as +//! `partial_consumption`, grants nothing — the prior batch is not yet +//! fully consumed); only `consumedSeq == page_end` produces exactly ONE +//! further replay page, so at most ONE unacknowledged page per +//! (connection, terminal) exists at any time and per-pane +//! unacknowledged replay stays bounded by one page budget. Gap rounds +//! continue within the same credit until a frame-carrying page is +//! produced — an accepted credit always makes byte progress, never a +//! zero-progress demand for more credit. +//! +//! E2R3 (the third sharpening of the natural-exit transition): the +//! session's PHASE-TRANSITION decision — "extend the credited phase +//! with a staged exit" vs "transfer to the uncredited tail drain" vs +//! "arm at session start" — is ATOMIC with respect to exit staging. +//! [`TerminalRegistry::paced_exit_transition`] reads the staged-exit +//! state and the frozen head under ONE registry lock hold and returns +//! the decision; every drive site applies it idempotently. The +//! load-bearing site is [`settle_exit_transition`], the POST-DRIVE +//! disposition that commits extend-vs-transfer: an exit staged during +//! the drive — after the pre-arm's read, inside the pre-fix +//! check-then-act window — is absorbed by the disposition's own atomic +//! read, the credited phase extends, and terminal.exit rides the +//! credit acknowledging the page that reaches the frozen head. An exit +//! staged strictly after the disposition's transfer confirmation is +//! the drain's documented uncredited tail content: the atomicity +//! boundary is exactly the transfer decision. +//! +//! When the replay cursor reaches the fixed attach-time target, the +//! accumulated live range drains as ordinary delivery (pages, not +//! credit-gated) in the SPAWNED drain task toward a FIXED target — the +//! head captured ONCE when the drain starts, NEVER re-captured. When +//! that target is covered, the COMPLETION BOUNDARY B is captured ONCE +//! under the same hold (the registry records it on the subscriber — +//! `TargetCovered`), and the staged post-target remainder pages toward +//! B and ONLY B through the bounded post-target handoff — the boundary +//! NEVER moves again, so a producer appending at-or-above drain speed +//! cannot postpone completion (the moving-head chase is structurally +//! gone: no completion-path code reads the terminal's current head +//! after completion start). The completing hold clears the deferral +//! atomically with its delivery, and the frames the producer staged +//! past B flow through the normal live fan-out path in that same hold. +//! Retention overrunning the handoff cursor mid-handoff is the plan:146 +//! bounded-baseline exit: the registry sinks the exact bounds-carrying +//! gap (ordered ahead of everything), sweeps the retained window through +//! the normal live path, clears the deferral, and the session COMPLETES +//! AT THE RING FRONT — the paged handoff never resumes toward the +//! unreachable boundary. Between pages the drain task holds a +//! RESERVATION on the connection queue's admission capacity +//! (reserve-then-admit, round-5): the reservation accounts for the page +//! it is about to admit and for every other concurrent pane drain's +//! in-flight reservation, so the un-credited drain is bounded by the +//! connection queue's REAL consumption and can never self-spill its own +//! unconsumed pages. + +use std::collections::HashMap; +use std::sync::Arc; + +use freshell_protocol::{ + ServerMessage, TerminalOutputGap, TerminalOutputGapReason, TerminalReplayCredit, +}; +use freshell_terminal::{ + FrameSink, PacedPage, PacedSessionDesc, PacedTailCompletion, TerminalRegistry, +}; + +/// One active paced replay session for a (connection, terminal). Owned by +/// the connection's dispatch loop; a re-attach replaces it, a detach or +/// socket drop discards it. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) struct PacedSession { + pub terminal_id: String, + pub stream_id: String, + /// The attach generation this session serves — the stale-credit key. + pub attach_request_id: String, + /// FIXED catch-up target: the head at attach time. Ongoing output never + /// extends it (frames past the target are tail delivery). + pub target: i64, + /// The retention-adjusted baseline the session started from. + pub effective_since: i64, + /// Highest credited consumed seq (starts at the baseline; the + /// double-grant guard). + pub credited: i64, + /// Production cursor: the last seq sent in a page. + pub page_end: i64, + /// The session's page budget (round-2 finding F3): the clamped + /// effective bound (`min(requested replayPageBytes, registry cap)`) + /// recorded at attach — credits, the tail drain, and the exit drain + /// all page at this SAME bound; nothing re-reads the registry cap. + pub page_budget: i64, + /// E2R1 finding 1: when the terminal EXITS NATURALLY while this + /// session is still in its credited phase, the registry STAGES the + /// exit on the subscriber and the credited phase extends ONCE + /// through the terminal's final head — captured at the staging (the + /// head is frozen by the exit; there is no moving-head chase). The + /// deferred final output then pages ONLY on the client's + /// continuation credits, and the session's completing verdict (the + /// drain's CaughtUp hold) delivers the staged exit after the last + /// page. `None` while the terminal lives: the ordinary fixed-target + /// semantics (frames past the target are tail delivery). + pub exit_head: Option, + /// Pages produced so far (observability). + pub pages: u64, +} + +impl PacedSession { + /// The credited phase's paging target: the fixed attach-time head, + /// extended once through the terminal's final head when a staged + /// exit armed the extension. + pub(crate) fn phase_target(&self) -> i64 { + match self.exit_head { + Some(exit_head) => self.target.max(exit_head), + None => self.target, + } + } + + /// Arm the staged-exit extension (idempotent, monotone): the + /// terminal's final head becomes part of the credited phase's + /// paging target, so the deferred final output pages only on + /// credits. + pub(crate) fn arm_staged_exit(&mut self, exit_head: i64) { + self.exit_head = Some(match self.exit_head { + Some(armed) => armed.max(exit_head), + None => exit_head, + }); + } + + /// Adopt the registry's attach-time session description (the first page + /// was produced under the attach lock and is sunk by the caller). + pub(crate) fn from_desc(desc: PacedSessionDesc) -> Self { + let started = desc.page_end > desc.effective_since; + Self { + terminal_id: desc.terminal_id, + stream_id: desc.stream_id, + attach_request_id: desc.attach_request_id, + target: desc.target, + effective_since: desc.effective_since, + credited: desc.effective_since, + page_end: desc.page_end, + page_budget: desc.page_budget, + exit_head: None, + pages: u64::from(started), + } + } +} + +/// E2R2 finding (the exit-arming race), invariant 1 — the PRE-ARM: +/// arming always precedes driving, at every drive site (the credit +/// path, the attach start, and the notify arm), so the drive pages +/// toward the extended phase target immediately. The registry is the +/// authority for a staged natural exit: the staging happens under the +/// terminal lock BEFORE the connection's notify hook fires, so this +/// query observes it regardless of the notify's dispatch order. +/// +/// E2R3: the arm reads the transition decision under ONE registry +/// lock hold ([`TerminalRegistry::paced_exit_transition`] — the staged +/// exit and the frozen head together); the pre-fix shape read them in +/// two separate holds. The arm is a HINT, never a transition +/// commitment: it only ever EXTENDS the session (monotone, idempotent), +/// and the disposition ([`settle_exit_transition`]) re-decides +/// atomically after the drive — so an exit this read missed because it +/// staged inside the credit's window is absorbed by the disposition's +/// own read. Because the arm precedes the drive, an arm that extends +/// the phase target beyond the credited cursor is always followed by a +/// drive that produces the next page: no state can exist where the +/// target exceeds the credited cursor and no page was just emitted +/// (the pre-fix drive-then-arm ordering kept such sessions with +/// nothing outstanding, permanently wedging the final output and +/// terminal.exit — nothing outstanding means no credit can ever +/// come). +/// +/// Returns `true` when THIS call armed (observability). +pub(crate) fn arm_staged_exit_from_registry( + registry: &TerminalRegistry, + conn_id: u64, + session: &mut PacedSession, +) -> bool { + if session.exit_head.is_some() { + // Already armed — the notify, the start, or an earlier arm won + // the race; the frozen head never moves and a second arm is + // inert. + return false; + } + let Some(exit_head) = registry + .paced_exit_transition(&session.terminal_id, conn_id) + .exit_head + else { + return false; + }; + session.arm_staged_exit(exit_head); + tracing::info!( + terminal_id = %session.terminal_id, + attach_request_id = %session.attach_request_id, + exit_head, + "ws.restore.paced_exit_armed" + ); + true +} + +/// The per-connection session table (at most one session per terminal). +#[derive(Default)] +pub(crate) struct PacedSessions { + inner: HashMap, +} + +impl PacedSessions { + /// Install (or replace — the re-attach supersede) a session. + pub(crate) fn insert(&mut self, session: PacedSession) { + self.inner.insert(session.terminal_id.clone(), session); + } + + pub(crate) fn get_mut(&mut self, terminal_id: &str) -> Option<&mut PacedSession> { + self.inner.get_mut(terminal_id) + } + + /// Cancel one terminal's session (`terminal.detach`). + pub(crate) fn cancel(&mut self, terminal_id: &str) { + self.inner.remove(terminal_id); + } + + pub(crate) fn remove(&mut self, terminal_id: &str) -> Option { + self.inner.remove(terminal_id) + } +} + +/// The verdict for one inbound credit (observability's status field). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum CreditVerdict { + Accepted, + /// In-window but below the last-sent page's end: honest consumption + /// progress the server records as an observation, but the prior batch + /// is not fully consumed, so it grants nothing. + PartialConsumption, + StaleGeneration, + /// The credit carries the current attach generation but the WRONG + /// stream id (round-2 finding F4): stream identity is part of the + /// continuation contract, so this is not the session's credit. A + /// distinct verdict from the stale generation — the generation is + /// current; the STREAM is foreign. + StreamMismatch, + BeyondWindow, + NonNegotiated, +} + +impl CreditVerdict { + pub(crate) fn as_str(self) -> &'static str { + match self { + Self::Accepted => "accepted", + Self::PartialConsumption => "partial_consumption", + Self::StaleGeneration => "stale_generation", + Self::StreamMismatch => "stream_mismatch", + Self::BeyondWindow => "beyond_window", + Self::NonNegotiated => "non_negotiated", + } + } +} + +/// Validate one credit against the session and, when valid, advance the +/// credited cursor (the grant). Pure — the drive loop does the producing. +/// +/// The grant is STRICTLY page-end: only a `consumed_seq` equal to the +/// last-sent page's end produces the next page (the plan's credit rule — +/// "Credit is granted only after the prior batch is consumed in order"). +/// An in-window value below the page end is a partial-consumption report: +/// observed (`partial_consumption`), inert — at most ONE unacknowledged +/// page per (connection, terminal) exists at any time, so per-pane +/// unacknowledged replay stays bounded by one page budget. +pub(crate) fn validate_credit( + session: &mut PacedSession, + credit: &TerminalReplayCredit, +) -> CreditVerdict { + if session.attach_request_id != credit.attach_request_id { + return CreditVerdict::StaleGeneration; + } + // Round-2 finding F4: stream identity is a required part of the + // continuation contract — a credit whose stream id does not match the + // session's stream belongs to a different stream's story, not this + // session, and must never page for it (no grant, wire shape unchanged). + if session.stream_id != credit.stream_id { + return CreditVerdict::StreamMismatch; + } + if credit.consumed_seq <= session.credited || credit.consumed_seq > session.page_end { + return CreditVerdict::BeyondWindow; + } + if credit.consumed_seq < session.page_end { + return CreditVerdict::PartialConsumption; + } + session.credited = credit.consumed_seq; + CreditVerdict::Accepted +} + +/// The negotiated retention gap for an `Expired` read: the exact lost +/// interval plus the task-2 bounds fields, stamped with the session's +/// generation. +fn retention_gap( + terminal_id: &str, + stream_id: &str, + attach_request_id: &str, + lost_from: i64, + lost_to: i64, + head_seq: i64, + oldest_retained_seq: i64, +) -> ServerMessage { + ServerMessage::TerminalOutputGap(TerminalOutputGap { + terminal_id: terminal_id.to_string(), + stream_id: stream_id.to_string(), + attach_request_id: Some(attach_request_id.to_string()), + from_seq: lost_from, + to_seq: lost_to, + reason: TerminalOutputGapReason::ReplayWindowExceeded, + head_seq: Some(head_seq), + oldest_retained_seq: Some(oldest_retained_seq), + }) +} + +/// The outcome of driving a session one round (attach start or one valid +/// credit). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum DriveOutcome { + /// A replay page is outstanding (uncredited); the session waits for the + /// next credit. + Active, + /// The credited replay phase is DONE — the fixed attach-time target is + /// covered. The caller hands the session to the spawned drain task + /// ([`spawn_paced_drain`]); the un-credited drain never runs on the + /// connection dispatcher. + DrainReady, + /// The terminal (or the connection's subscriber) disappeared — the + /// session is cancelled; the registry has no deferral left to clear. + Gone, +} + +/// Produce the next page for a session after a valid credit (or drain the +/// tail when the cursor reached the target). Exactly one replay page per +/// credit; `Expired` rounds emit the negotiated retention gap and continue +/// within the same credit until a frame-carrying page exists. The tail +/// drains without credit toward the FIXED head captured at tail-start +/// (never the moving current head — a producing terminal cannot postpone +/// completion indefinitely, and the drain's per-connection work stays +/// bounded), then completes through budget-bounded pages: the staged +/// remainder pages at the same page budget (never a full-suffix batch), +/// the deferral clears only at the completing verdict, and a retention +/// advance past the drain cursor reports the exact bounds-carrying gap +/// and resumes from the ring front so live frames can neither jump the +/// pages nor be lost. +/// +/// E2R1 finding 1: the replay phase pages toward the session's PHASE +/// TARGET — the fixed attach-time head, extended once through the +/// terminal's final head when a natural exit staged behind the +/// still-armed session (see [`PacedSession::exit_head`]). The deferred +/// final output therefore pages ONLY on the client's continuation +/// credits, and the exit sequences behind the session's credited +/// completion. +pub(crate) fn drive_session( + registry: &TerminalRegistry, + conn_id: u64, + sink: &FrameSink, + session: &mut PacedSession, + budget: i64, +) -> DriveOutcome { + // Replay phase: page (page_end, ...] toward the phase target. + let phase_target = session.phase_target(); + if session.page_end < phase_target { + loop { + match registry.next_replay_page( + &session.terminal_id, + conn_id, + session.page_end, + phase_target, + budget, + ) { + PacedPage::Frames { + messages, end_seq, .. + } => { + for message in messages { + sink(message); + } + session.page_end = end_seq; + session.pages += 1; + if session.page_end < phase_target { + return DriveOutcome::Active; + } + break; // cursor reached the phase target -> tail + } + PacedPage::Done => break, + PacedPage::Expired { + lost_from, + lost_to, + resume_from, + head_seq, + oldest_retained_seq, + } => { + sink(retention_gap( + &session.terminal_id, + &session.stream_id, + &session.attach_request_id, + lost_from, + lost_to, + head_seq, + oldest_retained_seq, + )); + tracing::info!( + terminal_id = %session.terminal_id, + lost_from, + lost_to, + resume_from, + "ws.restore.paced_expired" + ); + session.page_end = resume_from; + continue; + } + PacedPage::Gone => { + tracing::warn!( + terminal_id = %session.terminal_id, + attach_request_id = %session.attach_request_id, + "ws.restore.paced_gone" + ); + return DriveOutcome::Gone; + } + } + } + } + DriveOutcome::DrainReady +} + +/// E2R3 — the disposition a drive site commits from the ATOMIC +/// transition decision (see [`settle_exit_transition`]). +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) enum TransitionDisposition { + /// Keep the session in the credited phase: a page is outstanding, or + /// the extended phase still has window left to page on the client's + /// continuation credits. + Stay, + /// Hand the session to the uncredited tail drain + /// ([`spawn_paced_drain`]) — the atomic decision confirmed no + /// staged exit, or the armed exit head is fully acknowledged (the + /// drain's completing verdict delivers the staged exit after + /// everything the client already acknowledged). + Transfer, + /// The terminal (or the connection's subscriber) disappeared — + /// cancel the session; the registry has no deferral left to clear. + Gone, +} + +/// E2R3 — THE ATOMIC TRANSITION DISPOSITION. The phase-transition +/// decision — "extend the credited phase with a staged exit" vs +/// "transfer to the uncredited tail drain" — is committed HERE, from +/// ONE registry lock hold +/// ([`TerminalRegistry::paced_exit_transition`]: the staged-exit state +/// and the frozen head read together) AFTER the drive produced its +/// pages. +/// +/// THE LOCK-SCOPE MAP (decided under the lock vs applied after): +/// - UNDER the decision's hold: the staged-exit state and the frozen +/// head are read, and the decision value — extend, or the +/// atomically-confirmed absence of a staged exit — leaves the lock +/// WITH the read. No window exists in which a concurrently staged +/// exit can change which transition is correct. +/// - AFTER the hold (idempotent application, safe against concurrent +/// staging: an armed head is FROZEN by the exit, and the unarmed +/// transfer's boundary IS the decision): the session arms, the +/// wedge-guard may emit the extended phase's next page, the session +/// stays in the table, or it moves to the spawned drain. The +/// terminal lock is never held across the application — pages are +/// built under the drive's own per-page holds and sunk outside them, +/// exactly as before. +/// +/// WHY THE DECISION SITS AFTER THE DRIVE: the disposition must be +/// settled from state that accounts for the drive's own pages, and +/// the staging that matters is whatever happened-before THIS +/// decision's hold — including an exit staged DURING the drive, after +/// the pre-arm's read (the pre-fix check-then-act window: the arm read +/// "no exit", the drive ran, the exit staged, and the disposition then +/// decided from the stale read and transferred the session to the +/// drain, which dumped the remaining suffix and delivered +/// terminal.exit without the acknowledging credit). An exit staged +/// before the hold is absorbed here: the phase extends and the exit +/// rides the credit acknowledging the page reaching the frozen head. +/// An exit staged strictly after the hold is post-decision — at the +/// transfer site it is the drain's documented uncredited tail content. +/// The atomicity boundary is exactly this decision. +/// +/// The armed dispositions, in drive-site order: +/// - `Gone` from the drive: the registry owns the teardown — cancel. +/// - an uncredited page outstanding (`credited < page_end`): STAY — +/// including the page that just reached the armed exit head; the +/// removal rides the credit that acknowledges it (E2R2 invariant +/// 2), and the next credit's drive pages toward the extended target. +/// - all-acknowledged cursor with the armed head AHEAD +/// (`credited == page_end < exit_head`): THE WEDGE GUARD — the +/// extension outgrew everything outstanding, so the drive MUST +/// produce the extended phase's next page NOW (no state may exist +/// where the target exceeds the credited cursor and no page is +/// outstanding — nothing outstanding means no credit can ever come; +/// the guard emits at most the ONE page every drive site emits per +/// credit). +/// - armed, all acknowledged, head covered +/// (`credited == page_end >= exit_head`): TRANSFER — the spawned +/// drain's CaughtUp hold delivers the staged exit after everything +/// the client already acknowledged; the acknowledging credit is the +/// one being processed, so the exit rides it. +/// +/// The unarmed disposition is the drive's own outcome — the ordinary +/// semantics, byte-identical to the pre-E2R3 dispatch (Active → stay, +/// DrainReady → transfer, Gone → cancel) — and the atomically +/// confirmed no-staged-exit is what makes the transfer safe: no +/// happened-before staging was missed. +pub(crate) fn settle_exit_transition( + registry: &TerminalRegistry, + conn_id: u64, + sink: &FrameSink, + session: &mut PacedSession, + drive_outcome: DriveOutcome, +) -> TransitionDisposition { + let transition = registry.paced_exit_transition(&session.terminal_id, conn_id); + let Some(exit_head) = transition.exit_head else { + // (b) No exit staged — atomically confirmed under the + // decision's lock hold. Any staging after this hold is + // post-transfer tail content (the drain's documented + // uncredited semantics). + return match drive_outcome { + DriveOutcome::Active => TransitionDisposition::Stay, + DriveOutcome::DrainReady => TransitionDisposition::Transfer, + DriveOutcome::Gone => TransitionDisposition::Gone, + }; + }; + // (a) A staged exit is visible UNDER THIS HOLD: the credited phase + // extends through the frozen head — idempotent, monotone. + session.arm_staged_exit(exit_head); + if drive_outcome == DriveOutcome::Gone { + return TransitionDisposition::Gone; + } + if session.credited < session.page_end { + // An uncredited page is outstanding — including the page that + // just reached the armed exit head: the exit rides the credit + // that acknowledges it, never the drive that emitted it. + return TransitionDisposition::Stay; + } + if session.page_end < exit_head { + // THE WEDGE GUARD: the extension outgrew an all-acknowledged + // cursor — the drive must produce the extended phase's next + // page now. + return match drive_session(registry, conn_id, sink, session, session.page_budget) { + DriveOutcome::Active => TransitionDisposition::Stay, + DriveOutcome::DrainReady if session.credited < session.page_end => { + TransitionDisposition::Stay + } + DriveOutcome::DrainReady => TransitionDisposition::Transfer, + DriveOutcome::Gone => TransitionDisposition::Gone, + }; + } + // Armed, everything emitted is acknowledged, and the frozen head + // is covered: the spawned drain's CaughtUp hold delivers the + // staged exit now, after everything the client already + // acknowledged — the acknowledging credit is the one being + // processed, so the exit rides it. + TransitionDisposition::Transfer +} + +/// The drain's per-iteration admission reservation (E2R1 finding 2b): +/// the reservation must account what the drain can actually admit. A +/// page packs within the session's budget, but the page builder always +/// includes the window's FIRST frame even when it alone exceeds the +/// budget — the ATOMIC single-frame page — so a sub-cap budget (a +/// client's small `replayPageBytes` request) can still admit a page up +/// to the fragment-cap ceiling. Reserving only the requested budget let +/// concurrent pane drains each be granted against the same +/// just-under-watermark backlog and each admit a full atomic page on +/// top of it, overbooking the admission gate and self-spilling the +/// connection's own queued pages. The reservation is therefore +/// max(session budget, the atomic page ceiling); a budget at-or-above +/// the ceiling (every supported default) reserves exactly the budget, +/// byte-identical to the pre-fix behavior. +pub(crate) fn drain_admission_bytes(page_budget: i64) -> usize { + let atomic_ceiling = freshell_terminal::paced_atomic_page_serialized_ceiling() as i64; + page_budget.max(atomic_ceiling).max(0) as usize +} + +/// Spawn the session's UN-CREDITED drain (the accumulated live range after +/// the credited replay covered its fixed target). The drain runs OFF the +/// connection dispatcher — a sustained producer can hold it open through +/// real backpressure for as long as the client takes to consume the +/// backlog, and the dispatcher must keep servicing input, other panes, +/// and control traffic meanwhile (the inline-drain structure this +/// replaces monopolized the dispatcher for the drain's whole duration, +/// starving same-connection input). +/// +/// The drain target is the head captured ONCE at drain start — NEVER +/// re-captured. When the target is covered, the registry's +/// `TargetCovered` verdict captures the COMPLETION BOUNDARY B ONCE under +/// that hold (recorded on the subscriber) and the staged post-target +/// remainder pages toward B and ONLY B through the bounded post-target +/// handoff (`handoff_paced_tail`) — page-budget-sized per lock hold, the +/// lock released between chunks, the same admission-reserved loop. The +/// completing hold clears the deferral atomically with its delivery and +/// the frames staged past B flow through the normal live fan-out path in +/// that hold; a retention overrun past the handoff cursor mid-handoff is +/// the plan:146 bounded-baseline exit (`GapCompleted`): the exact +/// bounds-carrying gap is sunk (ordered ahead of everything), the +/// retained window sweeps through the normal live path, and the session +/// COMPLETES AT THE RING FRONT — never a resumption toward the +/// unreachable boundary. No completion or handoff path ever bulk-clones +/// a retained suffix or admits pages outside the page budget; a +/// sustained producer can never turn the drain into a moving-head +/// chase. Between pages the task holds a RESERVATION on the connection +/// queue's admission capacity (round-5 finding 1, reserve-then-admit): +/// the grant accounts for the page about to be admitted and for every +/// other concurrent pane drain's in-flight reservation — the sink +/// itself admits without yielding and evicts past the byte limit, so +/// the reservation is what bounds the drain and keeps it from +/// self-spilling its own unconsumed pages. `cancel` fires on the +/// connection's teardown (any exit reason), bounding the task's +/// lifetime with the connection's own. +/// +/// Page order, the deferral contract, and lock discipline are preserved: +/// the session leaves `PacedSessions` when it enters the drain (credits +/// during the drain are stale generations — observed, inert), only this +/// task produces the drain's pages (single producer, ascending seq), the +/// registry's per-terminal lock is never held across pages, and the +/// generation guard inside `TerminalRegistry::complete_paced_tail` / +/// `TerminalRegistry::handoff_paced_tail` refuses the drain the moment a +/// re-attach supersedes its attach generation (the off-dispatch drain can +/// race the dispatcher's re-attach handling; the guard makes that race +/// inert). +pub(crate) fn spawn_paced_drain( + registry: TerminalRegistry, + conn_id: u64, + writer: crate::terminal::WsSink, + sink: FrameSink, + mut session: PacedSession, + budget: i64, + mut cancel: tokio::sync::watch::Receiver, +) { + tokio::spawn(async move { + let terminal_id = session.terminal_id.clone(); + let stream_id = session.stream_id.clone(); + let attach_request_id = session.attach_request_id.clone(); + // The FIXED drain target: captured ONCE, never reassigned. + let drain_target = match registry.replay_bounds(&terminal_id) { + Some(bounds) => bounds.head_seq, + None => { + tracing::warn!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + "ws.restore.paced_gone" + ); + return; + } + }; + // The drain's two phases share ONE gated loop: paging toward the + // FIXED target, then — once it is covered — the bounded post-target + // handoff toward the terminal's current head. The phase never + // re-captures the drain target. + // + // RESERVE-THEN-ADMIT (round-5 finding 1): every page and every + // handoff chunk reserves its admission bytes under the connection + // queue's own lock BEFORE the registry builds and sinks it, so the + // reservation accounts for the page about to be admitted AND for + // every other concurrent pane drain's in-flight reservation — the + // concurrent wakes can no longer each admit a full page on top of + // the same pre-admission backlog. The permit is held across the + // whole iteration (the page's bytes become real backlog before + // the release; an iteration that sinks a retention gap instead + // releases the reservation unused). The server boot clamps the + // page budget to the connection queue's admission ceiling, so the + // grant is always reachable — and the gate itself can never + // deadlock whatever the budget (a page larger than the watermark + // admits into a fully drained queue). E2R1 finding 2b: the + // reserved bytes are [`drain_admission_bytes`] — max(budget, the + // atomic page ceiling) — because a sub-cap budget can still admit + // the builder's atomic single-frame page, and a reservation of + // only the requested bytes under-books exactly that page. + let admission_bytes = drain_admission_bytes(budget); + let mut handing_off = false; + loop { + // The RAII permit is held to the END of the loop body (dropped + // at each `return` inside the match and after the match for + // continuing arms): its only role is the reservation's + // lifetime, hence the underscore binding. + let _permit = tokio::select! { + permit = writer.reserve_drain_admission(admission_bytes) => { + match permit { + Some(permit) => permit, + None => { + // The writer pump is gone; the cancel path owns + // the exit. + return; + } + } + } + _ = cancel.changed() => { + tracing::debug!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + "ws.restore.paced_drain_cancelled" + ); + return; + } + }; + let verdict = if handing_off { + registry.handoff_paced_tail( + &terminal_id, + conn_id, + &attach_request_id, + session.page_end, + budget, + ) + } else { + registry.complete_paced_tail( + &terminal_id, + conn_id, + &attach_request_id, + session.page_end, + drain_target, + budget, + ) + }; + match verdict { + PacedTailCompletion::Handoff { + end_seq, + serialized_bytes, + } => { + tracing::debug!( + terminal_id = %terminal_id, + end_seq, + serialized_bytes, + "ws.restore.paced_tail_page" + ); + session.page_end = end_seq; + session.pages += 1; + } + PacedTailCompletion::TargetCovered { + end_seq, + serialized_bytes, + } => { + // THE FIXED TARGET IS COVERED — COMPLETION START (never + // re-captured): the registry recorded the completion + // boundary B (the head under this hold) on the + // subscriber; the staged post-target remainder now pages + // toward B and ONLY B through the bounded handoff + // (page-budget-sized chunks, gated, lock released + // between them) — never a bulk re-fan, never a moving + // current-head target. + tracing::debug!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + end_seq, + serialized_bytes, + "ws.restore.paced_target_covered" + ); + session.page_end = end_seq; + session.pages += 1; + handing_off = true; + } + PacedTailCompletion::Expired { + lost_from, + lost_to, + resume_from, + head_seq, + oldest_retained_seq, + } => { + // DRAIN-PHASE retention overrun (before the target was + // covered): the exact bounds-carrying gap, then continue + // toward the still-FIXED target from the resumed front. + // (Mid-handoff overruns take the plan:146 exit below — + // the paged handoff never resumes.) + sink(retention_gap( + &terminal_id, + &stream_id, + &attach_request_id, + lost_from, + lost_to, + head_seq, + oldest_retained_seq, + )); + tracing::info!( + terminal_id = %terminal_id, + lost_from, + lost_to, + resume_from, + "ws.restore.paced_expired" + ); + session.page_end = resume_from; + } + PacedTailCompletion::GapCompleted { + lost_from, + lost_to, + end_seq, + serialized_bytes: _, + reason, + } => { + // THE plan:146 BOUNDED-BASELINE EXIT (mid-handoff + // retention overrun): the registry already sank the + // exact bounds-carrying gap (ordered ahead of the + // swept retained window) and CLEARED the deferral in + // the same hold — the session COMPLETES AT THE RING + // FRONT with the gap recorded; the paged handoff never + // resumes toward the unreachable boundary, and the + // client's bounded baseline recovery owns the post-gap + // state. Round-5 finding 3: the event carries the + // MANDATORY gap-exit reason so production diagnostics + // distinguish a genuine unfetchable retention overrun + // (`retention_overrun`) from the ordinary fetchable + // fixed-boundary residual exit + // (`handoff_boundary_residual`). + tracing::info!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + reason = reason.as_str(), + lost_from, + lost_to, + last_seq = end_seq, + pages = session.pages + 1, + "ws.restore.paced_expired" + ); + session.page_end = session.page_end.max(end_seq); + session.pages += 1; + tracing::info!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + last_seq = session.page_end, + pages = session.pages, + "ws.restore.paced_complete" + ); + return; + } + PacedTailCompletion::CaughtUp => { + // The ring drained at or below the cursor (a quiet or + // slower terminal): the registry's atomic clear already + // fired inside the page read. + tracing::info!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + last_seq = session.page_end, + pages = session.pages, + "ws.restore.paced_complete" + ); + return; + } + PacedTailCompletion::Completed { end_seq, .. } => { + // The completing hold: the chunk covered the FIXED + // boundary B, everything staged past it flowed through + // the normal live fan-out path in the same hold, and + // the deferral cleared atomically with that delivery. + session.page_end = session.page_end.max(end_seq); + session.pages += 1; + tracing::info!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + last_seq = session.page_end, + pages = session.pages, + "ws.restore.paced_complete" + ); + return; + } + PacedTailCompletion::Gone => { + tracing::warn!( + terminal_id = %terminal_id, + attach_request_id = %attach_request_id, + "ws.restore.paced_gone" + ); + return; + } + } + } + }); +} + +/// Begin one negotiated session after a paced attach: sink the first page +/// (produced under the attach lock; sunk now that it is released), emit +/// `ws.restore.paced_start`, and settle the start's phase transition — +/// ATOMICALLY with respect to exit staging (E2R3). The start's arm +/// reads the transition decision under ONE registry lock hold, so an +/// exit staged during the attach is seen in the SAME decision; when the +/// arm extends an all-acknowledged cursor (the empty-first-page state), +/// the start's drive settles the disposition through +/// [`settle_exit_transition`]: the extended phase's first page is +/// produced NOW (the wedge guard), and the session stays credited-phase +/// active. When the first page already covered the target and nothing +/// is staged, the session hands straight to the spawned drain (a short +/// or empty replay drains to completion without ever needing a credit, +/// still OFF the dispatcher). A first page that is a bounded prefix +/// leaves the session ACTIVE with exactly ONE outstanding page: the +/// next page is produced on the first credit, never before. +#[allow(clippy::too_many_arguments)] +pub(crate) fn start_session( + registry: &TerminalRegistry, + conn_id: u64, + sink: &FrameSink, + sessions: &mut PacedSessions, + start: freshell_terminal::PacedAttachStart, + requested_since_seq: i64, + max_replay_bytes: Option, + writer: crate::terminal::WsSink, + cancel: tokio::sync::watch::Receiver, +) { + let page_bytes = start.session.page_bytes; + let mut session = PacedSession::from_desc(start.session); + for message in start.first_page { + sink(message); + } + tracing::info!( + terminal_id = %session.terminal_id, + attach_request_id = %session.attach_request_id, + requested_since = requested_since_seq, + effective_since = session.effective_since, + target = session.target, + page_budget = session.page_budget, + page_bytes, + max_replay_bytes = ?max_replay_bytes, + "ws.restore.paced_start" + ); + // E2R3: the start's ATOMIC arm/decide — ONE registry lock hold + // reads the staged-exit state; an exit staged during the attach is + // seen HERE, in the same decision (the notify arm and the first + // credit's disposition are later, also-atomic re-decisions). + // Arming precedes every branch below, so no branch can hand an + // armed session to the uncredited drain by accident of ordering. + arm_staged_exit_from_registry(registry, conn_id, &mut session); + if session.page_end < session.phase_target() && session.credited < session.page_end { + // A bounded first page is outstanding: the credited phase + // continues on the client's credits — no transition is + // committed at start (the next drive site's disposition is + // atomic). + sessions.insert(session); + return; + } + // The target is covered (with any extension), or the arm extended + // an all-acknowledged cursor with the frozen head ahead (the + // wedge-guard state: the drive MUST produce the extended phase's + // next page now). Either way the disposition is settled from the + // ATOMIC decision after the drive. + let budget = session.page_budget; + let outcome = drive_session(registry, conn_id, sink, &mut session, budget); + match settle_exit_transition(registry, conn_id, sink, &mut session, outcome) { + TransitionDisposition::Stay => { + sessions.insert(session); + } + TransitionDisposition::Transfer => { + spawn_paced_drain( + registry.clone(), + conn_id, + writer, + Arc::clone(sink), + session, + budget, + cancel, + ); + } + TransitionDisposition::Gone => {} + } +} + +#[cfg(test)] +mod tests { + use super::*; + + fn session_fixture() -> PacedSession { + PacedSession { + terminal_id: "T".into(), + stream_id: "S".into(), + attach_request_id: "arid-1".into(), + target: 100, + effective_since: 10, + credited: 10, + page_end: 50, + page_budget: 4096, + exit_head: None, + pages: 1, + } + } + + #[test] + fn staged_exit_extends_the_credited_phase_once_and_monotonically() { + // E2R1 finding 1: a natural exit behind a still-credited session + // extends the credited phase's paging target ONCE through the + // terminal's final head — the deferred final output pages only + // on credits. The extension is idempotent and monotone (a credit + // that raced ahead of the notify must not shrink or duplicate + // it), and while no exit is staged the phase target stays the + // ordinary fixed attach-time head. + let mut session = session_fixture(); + assert_eq!( + session.phase_target(), + 100, + "no staged exit: the phase target is the fixed attach-time head" + ); + session.arm_staged_exit(140); + assert_eq!( + session.phase_target(), + 140, + "the staged exit extends the phase target to the terminal's final head" + ); + session.arm_staged_exit(120); + assert_eq!( + session.phase_target(), + 140, + "a stale re-arm never shrinks the extension (monotone)" + ); + session.arm_staged_exit(160); + assert_eq!( + session.phase_target(), + 160, + "only the terminal's frozen head moves it — and it is frozen by death" + ); + } + + fn credit(arid: &str, consumed_seq: i64) -> TerminalReplayCredit { + credit_with_stream("S", arid, consumed_seq) + } + + fn credit_with_stream(stream_id: &str, arid: &str, consumed_seq: i64) -> TerminalReplayCredit { + TerminalReplayCredit { + terminal_id: "T".into(), + stream_id: stream_id.into(), + attach_request_id: arid.into(), + consumed_seq, + } + } + + #[test] + fn drain_admission_never_under_reserves_for_an_atomic_page() { + // E2R1 finding 2(b): a sub-cap page budget can still admit the + // page builder's ATOMIC single-frame result (a frame larger than + // the request forms its own page, bounded by the fragment-cap + // ceiling), so the reservation must cover that ceiling — + // reserving only the request under-books the admission gate for + // exactly the pages that exceed it, and concurrent drains then + // overbook the gate (the ws sub-cap canary's queue_overflow RED). + let ceiling = freshell_terminal::paced_atomic_page_serialized_ceiling() as i64; + assert!( + drain_admission_bytes(2048) >= ceiling as usize, + "a sub-cap budget reserves at least the atomic page ceiling \ + (got {})", + drain_admission_bytes(2048) + ); + assert_eq!( + drain_admission_bytes(128 * 1024), + 128 * 1024, + "a budget at-or-above the ceiling reserves the budget (the supported default)" + ); + assert_eq!( + drain_admission_bytes(ceiling), + ceiling as usize, + "at the ceiling the reservation is the ceiling" + ); + assert_eq!( + drain_admission_bytes(0), + ceiling as usize, + "a degenerate zero budget still reserves the atomic ceiling \ + (its pages are single frames, each bounded by the ceiling)" + ); + } + + #[test] + fn valid_credit_advances_the_credited_cursor() { + let mut session = session_fixture(); + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 50)), + CreditVerdict::Accepted + ); + assert_eq!(session.credited, 50); + // The production position stays at the last-sent page end. + assert_eq!(session.page_end, 50); + } + + #[test] + fn stale_generation_credit_is_ignored() { + let mut session = session_fixture(); + assert_eq!( + validate_credit(&mut session, &credit("arid-old", 50)), + CreditVerdict::StaleGeneration + ); + assert_eq!(session.credited, 10, "a stale credit grants nothing"); + } + + #[test] + fn wrong_stream_credit_with_the_current_generation_grants_nothing() { + // Round-2 finding F4: the continuation contract keys on stream + // identity as well as the attach generation — a credit carrying the + // CURRENT attachRequestId but the WRONG stream id is not this + // session's credit and must grant nothing (the stale-generation + // guard is otherwise incomplete: the arid matches, so without this + // check the credit would page for a session it does not belong to). + let mut session = session_fixture(); + assert_eq!( + validate_credit( + &mut session, + &credit_with_stream("OTHER-STREAM", "arid-1", 50) + ), + CreditVerdict::StreamMismatch + ); + assert_eq!(session.credited, 10, "a wrong-stream credit grants nothing"); + // The matching-stream credit still grants after the ignored one. + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 50)), + CreditVerdict::Accepted + ); + assert_eq!(session.credited, 50); + } + + #[test] + fn duplicate_and_beyond_window_credits_are_ignored() { + let mut session = session_fixture(); + // Beyond the last-sent page's end. + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 51)), + CreditVerdict::BeyondWindow + ); + // At-or-below the already-credited cursor (double grant). + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 10)), + CreditVerdict::BeyondWindow + ); + assert_eq!(session.credited, 10); + // A VALID credit still works after the ignored ones. + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 50)), + CreditVerdict::Accepted + ); + } + + #[test] + fn partial_page_credit_grants_nothing_until_the_page_end() { + let mut session = session_fixture(); + // A mid-page consumption report is honest progress, but the prior + // batch is not fully consumed: the plan's credit rule ("Credit is + // granted only after the prior batch is consumed in order") means + // it must NOT produce the next page — at most ONE unacknowledged + // page may exist per (connection, terminal). + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 30)), + CreditVerdict::PartialConsumption + ); + assert_eq!(session.credited, 10, "a partial credit advances nothing"); + assert_eq!(session.page_end, 50, "a partial credit grants no page"); + // Repeated partial reports (coalesced client ticks) stay inert. + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 40)), + CreditVerdict::PartialConsumption + ); + assert_eq!(session.credited, 10); + // The page-end value is the grant. + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 50)), + CreditVerdict::Accepted + ); + assert_eq!(session.credited, 50); + // The same value cannot grant twice (the double-grant guard). + assert_eq!( + validate_credit(&mut session, &credit("arid-1", 50)), + CreditVerdict::BeyondWindow + ); + assert_eq!(session.credited, 50); + } + + #[test] + fn session_adoption_counts_the_first_page() { + let desc = PacedSessionDesc { + terminal_id: "T".into(), + stream_id: "S".into(), + attach_request_id: "a".into(), + target: 9, + effective_since: 3, + page_end: 7, + page_budget: 4096, + page_bytes: 123, + }; + let session = PacedSession::from_desc(desc); + assert_eq!(session.credited, 3, "crediting starts at the baseline"); + assert_eq!(session.page_end, 7); + assert_eq!( + session.page_budget, 4096, + "the session carries the attach's effective page budget" + ); + assert_eq!(session.pages, 1); + + // An attach with nothing to replay starts with no pages. + let empty = PacedSessionDesc { + page_end: 3, + effective_since: 3, + ..PacedSessionDesc { + terminal_id: "T".into(), + stream_id: "S".into(), + attach_request_id: "a".into(), + target: 3, + effective_since: 3, + page_end: 3, + page_budget: 4096, + page_bytes: 0, + } + }; + assert_eq!(PacedSession::from_desc(empty).pages, 0); + } +} diff --git a/crates/freshell-ws/src/terminal.rs b/crates/freshell-ws/src/terminal.rs index f0db90dab..c23bde233 100644 --- a/crates/freshell-ws/src/terminal.rs +++ b/crates/freshell-ws/src/terminal.rs @@ -279,12 +279,14 @@ pub async fn run( state: &WsState, bcast_rx: tokio::sync::broadcast::Receiver, terminal_output_batch_v1: bool, + paced_terminal_replay_v1: bool, ui_screenshot_v1: bool, pane_reconcile_v1: bool, pane_reconcile_fresh_agent_v1: bool, origin_kind: &'static str, conn_identity: ConnectionIdentity, terminal_interest_v1: bool, + terminal_lifetime_claim_v1: bool, ) { let (ws_tx, ws_rx) = socket.split(); @@ -319,6 +321,7 @@ pub async fn run( state, bcast_rx, terminal_output_batch_v1, + paced_terminal_replay_v1, ui_screenshot_v1, pane_reconcile_v1, pane_reconcile_fresh_agent_v1, @@ -326,6 +329,7 @@ pub async fn run( origin_kind, conn_identity, terminal_interest_v1, + terminal_lifetime_claim_v1, ) .instrument(span) .await; @@ -341,6 +345,7 @@ async fn run_loop( state: &WsState, mut bcast_rx: tokio::sync::broadcast::Receiver, terminal_output_batch_v1: bool, + paced_terminal_replay_v1: bool, ui_screenshot_v1: bool, pane_reconcile_v1: bool, pane_reconcile_fresh_agent_v1: bool, @@ -348,6 +353,7 @@ async fn run_loop( origin_kind: &'static str, mut conn_identity: ConnectionIdentity, terminal_interest_v1: bool, + terminal_lifetime_claim_v1: bool, ) { // One independently supervised socket writer. The read/dispatch path // never awaits socket capacity; output is reconsidered one frame at a time. @@ -367,9 +373,59 @@ async fn run_loop( if terminal_interest_v1 { ws_tx.enable_terminal_interest(); } + if terminal_lifetime_claim_v1 { + // Hidden-pane lifetime claims (responsive-terminal-restore WS1): this + // connection's terminal.interest snapshots may carry + // claimedTerminalIds. A connection that never negotiated keeps its + // claim fields ignored server-side. + ws_tx.enable_terminal_lifetime_claims(); + } + if paced_terminal_replay_v1 { + // Restore contract (responsive-terminal-restore): a negotiated + // connection's queue-overflow gaps carry the terminal's CURRENT + // replay-retention bounds, resolved from the registry at + // gap-emission time. Installed BEFORE the pump is spawned (the same + // pre-spawn setup rule as `enable_terminal_interest`), so a gap can + // never be leased before the source exists. Non-negotiated + // connections leave the source unset and their gaps stay + // byte-identical to the pre-capability wire shape. + let registry = state.registry.clone(); + ws_tx.set_paced_replay_gap_bounds(Arc::new(move |terminal_id: &str| { + registry.replay_bounds(terminal_id) + })); + } let mut writer_task = tokio::spawn(writer.run(socket_tx).instrument(tracing::Span::current())); let _writer_lifetime = connection_writer::AbortWriterOnDrop(writer_task.abort_handle()); let mut writer_finished = false; + // Responsive-terminal-restore W1: this connection's paced replay + // sessions. Owned by the loop — a socket drop discards them (the + // registry-side deferrals die with the subscribers remove_connection + // sweeps), a detach cancels one, a re-attach replaces it. + let mut paced_sessions = crate::paced_replay::PacedSessions::default(); + // Round-2 finding F1 + E2R1 finding 1 + E2R2 finding: the + // connection's staged-exit routing. A natural exit while a paced + // session's deferral is armed STAGES the exit registry-side (final + // output first) and fires the per-subscriber notify hook; this + // channel carries the event back to THIS loop, whose select arm + // ARMS the still-CREDITED session's phase extension through the + // terminal's frozen final head (the dispatch-side mirror of the + // credit path's arm-first query — the registry is the authority + // either way, and the arm is monotone + idempotent). The deferred + // final output pages ONLY on continuation credits, and terminal.exit + // rides the credit that acknowledges the page reaching the armed + // exit head (no uncredited dump, no wedge). Unbounded mpsc: the + // sender lives on registry subscribers and fires at most once per + // subscriber; the credited session bounds the staged window itself. + let (paced_exit_tx, mut paced_exit_rx) = mpsc::unbounded_channel::<(String, i64)>(); + let paced_exit_notify: Option = paced_terminal_replay_v1 + .then(|| { + let tx = paced_exit_tx.clone(); + let notify: freshell_terminal::PacedExitNotify = + Arc::new(move |terminal_id: &str, exit_code: i64| { + let _ = tx.send((terminal_id.to_string(), exit_code)); + }); + notify + }); let conn_sink: FrameSink = { let sender = ws_tx.clone(); Arc::new(move |msg| { @@ -409,6 +465,11 @@ async fn run_loop( let mut catastrophic_ticker = tokio::time::interval(std::time::Duration::from_millis( (state.term09.catastrophic_stall_ms / 4).max(10), )); + // Drain-progress liveness (responsive-terminal-restore W3): the last + // snapshot of the writer's completed-send counter, so each monitor tick + // feeds the DELTA — successful sends since the previous tick — into the + // window decision. Slow consumption alone is not a dead socket. + let mut last_completed_sends: u64 = 0; // Task 9 (host-pressure pane): THIS connection's last `hoststats.refresh` // stamp — the per-connection 1s floor (legacy parity: @@ -506,12 +567,15 @@ async fn run_loop( conn_id, &conn_sink, terminal_output_batch_v1, + paced_terminal_replay_v1, pane_reconcile_v1, pane_reconcile_fresh_agent_v1, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut paced_sessions, + &paced_exit_notify, ) .await { @@ -542,16 +606,65 @@ async fn run_loop( _ => {} } } - // TERM-09 catastrophic backpressure: this connection's queued - // output has stayed above the threshold continuously for the - // full stall duration -- close now (mirrors `broker.ts`'s - // `catastrophicBlocked` closing with 4008 "Catastrophic backpressure"). + // E2R3 (the atomic exit-transition decision): the notify arm + // is the DISPATCH-side PRE-ARM — an extension hint, never a + // transition commitment. A session still in its credited + // phase — which always has exactly ONE uncredited page + // outstanding — gets its phase extended through the + // terminal's frozen final head HERE when the notify wins the + // dispatch race, so the next credit's drive pages toward it + // immediately; the arm reads the transition decision under + // ONE registry lock hold (monotone, idempotent — whichever + // arm lands first wins and the others are inert). NOTHING + // is ever delivered here and the session never leaves the + // table here: the disposition that commits extend-vs-transfer + // is the drive sites' `settle_exit_transition`, whose own + // atomic read absorbs any staging this arm missed — so a + // lost, delayed, or out-raced notify is a paging hint + // difference only, never a correctness difference. + Some((terminal_id, exit_code)) = paced_exit_rx.recv() => { + let _ = exit_code; // the armed log names the frozen head; the drain logs the code + if let Some(session) = paced_sessions.get_mut(&terminal_id) { + crate::paced_replay::arm_staged_exit_from_registry( + &state.registry, + conn_id, + session, + ); + } + // A session NOT in the credited table is owned by its + // drain task (its completing verdict delivers the staged + // exit) or already completed (the exit was delivered at + // staging, the non-paced shape) — nothing to do. + } _ = catastrophic_ticker.tick() => { - if catastrophic.tick(ws_tx.pending_output_bytes()) { + // TERM-09 catastrophic backpressure: this connection's queued + // output has stayed above the threshold continuously for the + // full stall duration WITH ZERO successful socket sends — close + // now (mirrors `broker.ts`'s `catastrophicBlocked` closing with + // 4008 "Catastrophic backpressure"). A slow-but-progressing + // client resets the window every tick it completes a send. + let completed_sends = ws_tx.completed_sends(); + let sends_since_last_tick = completed_sends.saturating_sub(last_completed_sends); + last_completed_sends = completed_sends; + if let Some(fire) = + catastrophic.tick(ws_tx.pending_output_bytes(), sends_since_last_tick) + { + // Task-007 review M3 (landed by task-010): the event + // must be diagnosable from the log line alone. + // `sends_in_window` is the per-occurrence evidence — + // completed sends DURING the deciding window for THIS + // close, structurally zero (any send resets the window) + // — while `total_sends` carries the connection's + // lifetime history; the pair distinguishes a + // wedge-after-progress episode (large total, silent + // window) from a never-sent socket (both zero). tracing::warn!( connection_id = conn_id, pending_bytes = ws_tx.pending_output_bytes(), threshold = state.term09.catastrophic_buffered_bytes, + total_sends = completed_sends, + sends_in_window = fire.sends_in_window, + window_ms = state.term09.catastrophic_stall_ms, "ws.terminal_stream.catastrophic_close" ); use axum::extract::ws::CloseFrame; @@ -750,6 +863,7 @@ async fn handle_client_text( conn_id: u64, conn_sink: &FrameSink, terminal_output_batch_v1: bool, + paced_terminal_replay_v1: bool, pane_reconcile_v1: bool, pane_reconcile_fresh_agent_v1: bool, interactive_create_tx: &mpsc::Sender, @@ -759,6 +873,14 @@ async fn handle_client_text( // D8: the connection's hello-stamped client identity (refreshed by // `tabs.sync.push` below) — the provenance source for ledger stamps. conn_identity: &mut ConnectionIdentity, + // Responsive-terminal-restore W1: this connection's paced replay + // sessions (one per attached terminal; the pacing coordinator's state). + paced_sessions: &mut crate::paced_replay::PacedSessions, + // Round-2 finding F1: the connection's staged-exit notification hook. + // A natural exit while a paced session's deferral is armed STAGES the + // exit registry-side and fires this hook; the run_loop's + // paced_exit_rx arm routes it into the exit-drain. + paced_exit_notify: &Option, ) -> bool { // Accept-and-strip: unknown/unparseable frames are ignored (matches the // runtime's tolerance; the handshake already gated auth). @@ -905,7 +1027,25 @@ async fn handle_client_text( } match message { ClientMessage::TerminalInterest(interest) => match ws_tx.set_terminal_interest(&interest) { - Ok(()) => true, + // Hidden-pane lifetime claims (responsive-terminal-restore WS1): + // the accepted snapshot's claim diff is applied to the registry — + // claims never attach, never grant replay, and never change + // delivery priority (the priority recompute happened inside + // set_terminal_interest). A connection that did not negotiate + // `terminalLifetimeClaimV1` produces an empty diff here, so its + // claimedTerminalIds (if any) are ignored server-side. + Ok(Some(claims)) => { + if !claims.is_empty() { + for terminal_id in &claims.added { + state.registry.claim_terminal(terminal_id, conn_id); + } + for terminal_id in &claims.removed { + state.registry.withdraw_claim(terminal_id, conn_id); + } + } + true + } + Ok(None) => true, Err(message) => { send( ws_tx, @@ -1458,15 +1598,61 @@ async fn handle_client_text( // session lost before its first snapshot. let asserted_at = now_ms(); maybe_restamp_on_attach(&attach, state, conn_identity, asserted_at).await; + // Observability inputs captured before the attach moves. + let requested_since_seq = attach.since_seq.unwrap_or(0); + let attach_max_replay_bytes = attach.max_replay_bytes; + // Responsive-terminal-restore binding point 6 (supersede): + // EVERY successful re-attach for the same (connection, + // terminal) cancels the connection's previous paced session + // for that terminal — a cancelled session's late credits are + // ignored as a stale generation. This must cover the LEGACY + // reply too (an arid-less re-attach from a negotiated + // connection, or an attach to an exited terminal): the + // registry already replaced the subscriber, so a surviving + // ws-layer session would be fed by a NEW subscriber's ring + // reads and could still produce phantom pages for a + // stale-generation credit. A FAILED attach (Error) cancels + // nothing — the previous session still matches its live + // subscriber and deferral. Each arm cancels BEFORE the paced + // insert below, so the new session is never the one removed. + let attach_terminal_id = attach.terminal_id.clone(); let attached = match handle_attach( attach, state, conn_id, conn_sink, terminal_output_batch_v1, + paced_terminal_replay_v1, + paced_exit_notify.clone(), ) { - Some(err) => send(ws_tx, &err).await, - None => true, + AttachReply::Error(err) => send(ws_tx, &err).await, + AttachReply::Legacy => { + paced_sessions.cancel(&attach_terminal_id); + true + } + AttachReply::Paced(start) => { + paced_sessions.cancel(&attach_terminal_id); + // The paced replay core (responsive-terminal-restore + // W1): sink the first page (the registry produced it + // under the attach lock), emit the session start, and + // — when the first page already covers the target — + // hand the session to the off-dispatch drain task (a + // short or empty replay drains to completion without + // ever needing a credit, still never on the + // dispatcher). + crate::paced_replay::start_session( + &state.registry, + conn_id, + conn_sink, + paced_sessions, + *start, + requested_since_seq, + attach_max_replay_bytes, + ws_tx.clone(), + create_cancel_rx.clone(), + ); + true + } }; // The window closes with the guard (exactly-once). drop(attach_guard); @@ -1542,8 +1728,37 @@ async fn handle_client_text( } } ClientMessage::TerminalDetach(detach) => { + // Responsive-terminal-restore: an explicit detach cancels the + // connection's paced session for this terminal (the registry-side + // deferral dies with the subscriber the detach removes). + paced_sessions.cancel(&detach.terminal_id); handle_detach(&detach, ws_tx, state, conn_id).await } + // Responsive-terminal-restore Workstream 1: the paced replay + // continuation credit. All rules sit under the connection's + // negotiated capability — a non-negotiated connection's credits are + // inert (logged, then ignored). + ClientMessage::TerminalReplayCredit(replay_credit) => { + if !paced_terminal_replay_v1 { + tracing::info!( + terminal_id = %replay_credit.terminal_id, + consumed_seq = replay_credit.consumed_seq, + status = crate::paced_replay::CreditVerdict::NonNegotiated.as_str(), + "ws.restore.credit" + ); + true + } else { + handle_replay_credit( + &replay_credit, + &state.registry, + conn_id, + conn_sink, + paced_sessions, + ws_tx, + create_cancel_rx, + ) + } + } ClientMessage::TerminalKill(kill) => { // b8ke ext r24 F2: the kill's coordinator transitions record // the connection's real device/client identity as the @@ -7088,23 +7303,44 @@ async fn maybe_restamp_on_attach( } /// `terminal.attach` — resolve the terminal in the shared registry and attach THIS -/// connection to it: the registry enqueues `terminal.attach.ready`, replays the -/// scrollback (seq-ordered, stamped with this attach's id + `source:'replay'`), and -/// registers the connection so live output fans out — all onto `conn_sink`, which -/// the select loop drains to the socket. Attaching to an unknown terminal returns -/// the reference's `error{INVALID_TERMINAL_ID, "Terminal not running"}` frame for -/// the caller to send (`ws-handler.ts:2730-2735`; restored by kata dtfn — the SPA's -/// recovery ladder recreates the pane). `None` = attached. +/// connection to it: the registry enqueues `terminal.attach.ready` and replays the +/// scrollback (seq-ordered, stamped with this attach's id + `source:'replay'`) onto +/// `conn_sink`, which the select loop drains to the socket. Attaching to an +/// unknown terminal returns the reference's `error{INVALID_TERMINAL_ID, +/// "Terminal not running"}` frame for the caller to send +/// (`ws-handler.ts:2730-2735`; restored by kata dtfn — the SPA's recovery ladder +/// recreates the pane). `Ok(None)` = attached with no reply. +/// +/// Responsive-terminal-restore Workstream 1: a NEGOTIATED +/// (`pacedTerminalReplayV1`) attach to a Running terminal with an +/// attachRequestId returns the paced session start instead — the registry +/// armed the subscriber's deferral and produced the FIRST page under the +/// attach lock; the caller sinks that page after the lock is released and +/// owns the pacing session (credits, tail drain, completion). +/// +/// TERM-07: the attach's `maxReplayBytes` threads through BOTH +/// geometry-authorized and geometry-skipped paths into the registry call, +/// where it is recorded on the subscriber with no delivery-behavior change +/// (the paced-start observability event reports it; the increment-3 +/// snapshot work consumes it). +enum AttachReply { + Legacy, + Error(Box), + Paced(Box), +} + fn handle_attach( attach: TerminalAttach, state: &WsState, conn_id: u64, conn_sink: &FrameSink, terminal_output_batch_v1: bool, -) -> Option { + paced_terminal_replay_v1: bool, + paced_exit_notify: Option, +) -> AttachReply { // STATE-SYNC FIX 1 increment 2a: stamp the canonical identity onto // `attach.ready` from the shared identity registry (create-time - // resume ids AND locator-associated ids both live there); the + // resume ids AND locator-associated ids both live here); the // registry crate is identity-agnostic, so it's resolved here. let canonical_session_ref = state.identity.session_ref_for(&attach.terminal_id); @@ -7118,9 +7354,28 @@ fn handle_attach( attach.expected_session_ref.as_ref(), canonical_session_ref.as_ref(), ); + // TERM-07 seam: the client's replay-budget request rides both paths. + let max_replay_bytes = attach.max_replay_bytes; + // Round-2 finding F3: the negotiated forward-page upper bound rides + // the same paths as a PacedAttachOptions input — the registry clamps + // it to its own cap and records the effective budget on the session + // so every later page honors the same bound. Round-2 finding F1: the + // connection's staged-exit notification hook installs with the paced + // subscriber ATOMICALLY with the attach (a post-attach registration + // would race the very natural exit it exists to sequence), so a + // natural exit while the deferral is armed can move this connection's + // session into its exit-drain. + let paced_options = freshell_terminal::PacedAttachOptions { + replay_page_bytes: attach.replay_page_bytes, + paced_exit_notify, + }; let outcome = if geometry_identity_ok { let cols = attach.cols.clamp(0, u16::MAX as i64) as u16; let rows = attach.rows.clamp(0, u16::MAX as i64) as u16; + // `paced_terminal_replay_v1` gates the paced replay core + // (responsive-terminal-restore): a negotiated attach with an + // attachRequestId to a Running terminal returns the paced session + // start; every other shape keeps the legacy inline replay. state.registry.attach_with_geometry( &attach.terminal_id, conn_id, @@ -7128,14 +7383,17 @@ fn handle_attach( attach.attach_request_id.clone(), attach.since_seq.unwrap_or(0), terminal_output_batch_v1, + paced_terminal_replay_v1, canonical_session_ref, // Mode replay-sync: the client's positive surface-fresh marker // (xterm recreation / user reset). Forwards the wire field 1:1; the // registry owns the emit-vs-skip gating. attach.surface_reset, + max_replay_bytes, attach.intent, cols, rows, + paced_options, ) } else { state.registry.attach( @@ -7145,12 +7403,18 @@ fn handle_attach( attach.attach_request_id.clone(), attach.since_seq.unwrap_or(0), terminal_output_batch_v1, + paced_terminal_replay_v1, canonical_session_ref, attach.surface_reset, + max_replay_bytes, + paced_options, ) }; if outcome.found { - return None; + return match outcome.paced { + Some(start) => AttachReply::Paced(Box::new(start)), + None => AttachReply::Legacy, + }; } // Kata dtfn: `AttachOutcome{found:false}` was silently discarded here, // wedging any attach against an unknown id (stale pre-restart id, typo'd @@ -7159,7 +7423,7 @@ fn handle_attach( // gate accepts it (attachRequestIds live in the `pane:N:nanoid` namespace, // never colliding with createRequestIds — see ws-client's // clearTrackedCreate-on-error behavior). - Some(ServerMessage::Error(ErrorMsg { + AttachReply::Error(Box::new(ServerMessage::Error(ErrorMsg { owner_kind: None, owner_generation: None, owner_epoch: None, @@ -7173,7 +7437,99 @@ fn handle_attach( terminal_id: Some(attach.terminal_id), terminal_exit_code: None, live_terminal_id: None, - })) + }))) +} + +/// One `terminal.replay.credit` on a negotiated connection +/// (responsive-terminal-restore Workstream 1): validate against the active +/// session's generation + outstanding-page window, and on acceptance +/// produce the next page (or the negotiated retention gap + continuation, +/// or — once the cursor reaches the target — the tail drain that completes +/// the session). Every non-accepted verdict is inert (observed via +/// `ws.restore.credit`, never a client-visible error). +fn handle_replay_credit( + replay_credit: &freshell_protocol::TerminalReplayCredit, + registry: &freshell_terminal::TerminalRegistry, + conn_id: u64, + conn_sink: &FrameSink, + paced_sessions: &mut crate::paced_replay::PacedSessions, + writer: &connection_writer::WriterSender, + cancel: &tokio::sync::watch::Receiver, +) -> bool { + use crate::paced_replay::TransitionDisposition; + // Identifiers/measurements only, per the restore observability contract. + let observe = |verdict: crate::paced_replay::CreditVerdict| { + tracing::info!( + terminal_id = %replay_credit.terminal_id, + consumed_seq = replay_credit.consumed_seq, + status = verdict.as_str(), + "ws.restore.credit" + ); + }; + let Some(session) = paced_sessions.get_mut(&replay_credit.terminal_id) else { + // No active session accepts this credit (completed, detached, + // draining, or a terminal never paced): a stale generation. + observe(crate::paced_replay::CreditVerdict::StaleGeneration); + return true; + }; + let verdict = crate::paced_replay::validate_credit(session, replay_credit); + observe(verdict); + if verdict != crate::paced_replay::CreditVerdict::Accepted { + return true; + } + // Round-2 finding F3: the SESSION's effective page budget (the + // attach's `replayPageBytes` request clamped to the registry cap, + // recorded on the session at attach) sizes the credited pages — the + // whole session honors the requested bound, not just the first page. + let budget = session.page_budget; + // E2R2 finding (the exit-arming race) — invariant 1, THE PRE-ARM: + // arming precedes the drive so the drive pages toward the extended + // phase target immediately. The arm is monotone/idempotent and + // NEVER a transition commitment: the disposition below re-decides + // atomically after the drive, so an exit staged inside this + // credit's window — after this arm's read, during the drive — is + // absorbed by the disposition's own single-hold read (E2R3: the + // pre-fix check-then-act window is closed structurally, not + // re-ordered). + crate::paced_replay::arm_staged_exit_from_registry(registry, conn_id, session); + let outcome = crate::paced_replay::drive_session(registry, conn_id, conn_sink, session, budget); + // E2R3 — THE ATOMIC TRANSITION DISPOSITION: extend-vs-transfer is + // committed from ONE registry lock hold AFTER the drive. An exit + // staged before the disposition's hold extends the credited phase + // (the exit rides the acknowledging credit of the page reaching the + // frozen head — E2R2 invariant 2 lives in the disposition now); the + // atomically-confirmed absence of a staged exit is what makes the + // transfer below safe; an exit staged strictly after the hold is + // the drain's documented uncredited tail content (the atomicity + // boundary is exactly the transfer decision). + match crate::paced_replay::settle_exit_transition( + registry, conn_id, conn_sink, session, outcome, + ) { + TransitionDisposition::Stay => {} + // The credited phase covered its target and everything it must + // acknowledge is acknowledged (or the armed exit head is fully + // covered): the session moves WHOLE into the spawned drain task + // — the connection dispatcher stays free (input, other panes, + // controls) while the un-credited drain pages, and credits that + // arrive during the drain are inert stale generations. + TransitionDisposition::Transfer => { + if let Some(session) = paced_sessions.remove(&replay_credit.terminal_id) { + crate::paced_replay::spawn_paced_drain( + registry.clone(), + conn_id, + writer.clone(), + Arc::clone(conn_sink), + session, + budget, + cancel.clone(), + ); + } + } + TransitionDisposition::Gone => { + paced_sessions.remove(&replay_credit.terminal_id); + } + } + true } /// Node's `resizeIfSessionMatches` identity guard @@ -10540,12 +10896,15 @@ mod pane_reconcile_gate_tests { 1, &conn_sink, false, + false, false, // pane_reconcile_v1: NOT negotiated on this connection false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!( @@ -10566,10 +10925,13 @@ mod pane_reconcile_gate_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(pong_ok); @@ -10610,10 +10972,13 @@ mod pane_reconcile_gate_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok); @@ -10634,10 +10999,13 @@ mod pane_reconcile_gate_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok, "attempt {attempt}: a full queue must be answered"); @@ -10975,10 +11343,13 @@ mod host_stats_dispatch_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok); @@ -11006,10 +11377,13 @@ mod host_stats_dispatch_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok); @@ -11027,10 +11401,13 @@ mod host_stats_dispatch_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(pong_ok); @@ -11069,10 +11446,13 @@ mod host_stats_dispatch_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok); @@ -11096,10 +11476,13 @@ mod host_stats_dispatch_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok); @@ -11141,10 +11524,13 @@ mod host_stats_dispatch_tests { false, false, false, + false, &interactive_create_tx, &create_cancel_rx, &mut host_stats_last_refresh_at, &mut conn_identity, + &mut Default::default(), + &None, ) .await; assert!(ok); @@ -11158,3 +11544,1015 @@ mod host_stats_dispatch_tests { assert!(host_stats_last_refresh_at.is_none()); } } + +/// E2R2 finding (the credited natural-exit exit-arming race): focused, +/// DETERMINISTIC exercises of the transition's two race windows. The +/// production entry points run against a REAL in-process PTY registry — +/// the staging is real (`finish_pty_exit` from the PTY reader thread), +/// the pages are real, the spawned drain is a real tokio task — but the +/// DISPATCHER is the test itself: each credit is a direct +/// `handle_replay_credit` call, so the "notify queued but not yet +/// dispatched" window (the exact race state the integration socket cannot +/// order deterministically) is constructed by simply not dispatching +/// anything else. Credit timing is controlled explicitly — the +/// parser-consumption boundary is the thing under test. +/// +/// THE INVARIANT under test (E2R2, stated per the finding): +/// 1. Arming always precedes driving: an arm that extends `phase_target` +/// beyond `credited` MUST be followed by a drive that produces the +/// next page — no state may exist where the target exceeds the +/// credited cursor and no page was just emitted (the WEDGE). +/// 2. `terminal.exit` rides the CREDIT verdict that acknowledges +/// consumption of the page reaching `exit_head` — never the drive +/// that emits it. +/// 3. Monotone arming: a restaging never moves `exit_head` backward and +/// never double-delivers. +/// 4. Retention loss mid-wait reports the exact bounds-carrying gap; +/// the exit still rides the acknowledging credit. +/// +/// THE INVARIANT under test (E2R3, the third sharpening — the phase +/// transition is ATOMIC with respect to exit staging): +/// 5. The session's phase-transition decision — extend the credited +/// phase with a staged exit vs transfer to the uncredited tail +/// drain vs arm at session start — reads the staged-exit state and +/// commits the disposition under ONE registry lock hold. No +/// check-then-act window may remain in which a concurrently staged +/// exit can change which transition was correct: an exit staged +/// before the decision's hold is absorbed by the SAME handling (the +/// phase extends; the exit rides the acknowledging credit of the +/// page reaching the frozen head), and an exit staged strictly after +/// the atomic transfer decision is the drain's documented +/// uncredited tail content. +/// +/// The E2R3 races are modeled deterministically (never sleep-based): +/// the registry's ONE-SHOT staging hook fires the natural-exit staging +/// INSIDE a staged-exit read's own lock hold, immediately after that +/// read — the PTY reader's concurrent staging at a precise point of +/// the decision path. The quiet script never exits for these tests: a +/// real exit would stage at its own uncontrolled moment. +#[cfg(test)] +mod paced_exit_race_tests { + use super::*; + use std::sync::Mutex; + + /// Everything the session sinks (pages, gaps, exits), in order. + type Collector = Arc>>; + + fn collector_sink(collector: &Collector) -> FrameSink { + let collector = Arc::clone(collector); + Arc::new(move |message| { + collector.lock().expect("collector lock").push(message); + }) + } + + fn outputs(collector: &Collector) -> Vec { + collector + .lock() + .expect("collector lock") + .iter() + .filter_map(|m| match m { + ServerMessage::TerminalOutput(frame) => Some(frame.data.clone()), + _ => None, + }) + .collect() + } + + fn exit_count(collector: &Collector) -> usize { + collector + .lock() + .expect("collector lock") + .iter() + .filter(|m| matches!(m, ServerMessage::TerminalExit(_))) + .count() + } + + fn last_is_exit(collector: &Collector) -> bool { + collector + .lock() + .expect("collector lock") + .last() + .is_some_and(|m| matches!(m, ServerMessage::TerminalExit(_))) + } + + /// Poll `probe` until it returns `Some` or the deadline passes — + /// the deterministic observation points (output landed in the ring, + /// the exit staged, the drain delivered). + async fn wait_for(mut probe: impl FnMut() -> Option, what: &str) -> T { + let deadline = std::time::Instant::now() + std::time::Duration::from_secs(10); + loop { + if let Some(value) = probe() { + return value; + } + assert!( + std::time::Instant::now() < deadline, + "timed out waiting for {what}" + ); + tokio::time::sleep(std::time::Duration::from_millis(10)).await; + } + } + + /// The deterministic terminal under test: a shell script that produces + /// output only in response to input lines, then a final marker, then + /// exits. Nothing is emitted before the first input, so the ring's + /// head is deterministically 0 until the test writes. Each pad step + /// prints a UNIQUE marker (`PAD-STEP-`): `step`'s marker wait is a + /// substring match over the whole retained ring, so a repeated + /// marker would let step i+1 return on step i's output before its + /// own ingestion — the attach would then race the remaining pads' + /// asynchronous ingestion (a real flake: the first page could cover + /// the whole raced window and fail the bounded-prefix assert). + fn race_script(pads: usize, final_marker: &str) -> String { + let mut script = String::new(); + for pad in 0..pads { + script.push_str(&format!("read x; printf 'PAD-STEP-{pad}\\n'; ")); + } + script.push_str(&format!("read x; printf '{}\\n'; exit\n", final_marker)); + script + } + + /// The QUIET twin (E2R3): identical output behavior, but the script + /// NEVER exits and echo is OFF — the concurrent-staging + /// transition-race tests stage the natural exit themselves via the + /// registry's deterministic in-lock hook, at a precise point + /// inside the decision path, and they COUNT FRAMES (budget 0: one + /// frame per page, so the credit whose drive reaches the attach + /// target is nameable in advance). A real script exit would stage + /// at its own uncontrolled moment, and a live PTY's input echo + /// coalesces with the step's output nondeterministically (sometimes + /// one frame, sometimes two) — `stty -echo` plus the ECHO-OFF + /// banner (the harness's [`RaceHarness::wait_ready`] gate) makes + /// every post-banner step produce EXACTLY its printf frame. Pad + /// markers are unique per step for the same reason as the exiting + /// script's (see [`race_script`]). + fn quiet_race_script(pads: usize, final_marker: &str) -> String { + let mut script = String::from("stty -echo; printf 'ECHO-OFF\\n'; "); + for pad in 0..pads { + script.push_str(&format!("read x; printf 'PAD-STEP-{pad}\\n'; ")); + } + script.push_str(&format!("read x; printf '{}\\n'; ", final_marker)); + script.push_str("while :; do read x; done"); + script + } + + struct RaceHarness { + registry: freshell_terminal::TerminalRegistry, + terminal_id: String, + collector: Collector, + sink: FrameSink, + writer: connection_writer::WriterSender, + /// The cancel channel's sender stays alive with the harness (the + /// production run_loop holds it for the connection's lifetime) — + /// a closed channel wakes the drain's cancel arm immediately. + _cancel_tx: tokio::sync::watch::Sender, + /// The writer pump stays alive with the harness (in production it + /// owns the socket until the connection ends): its Drop closes the + /// writer queue, and a closed queue makes every drain-admission + /// reservation return None (the drain would exit silently). It is + /// never polled here — the pages and the exit sink through the + /// registry subscriber, not the writer. + _writer_pump: connection_writer::WriterPump, + cancel_rx: tokio::sync::watch::Receiver, + sessions: crate::paced_replay::PacedSessions, + conn_id: u64, + arid: String, + } + + impl RaceHarness { + /// Spawn the script PTY, negotiate nothing (the registry attach is + /// the production paced path), and wire the session table, the + /// collector sink, and a real (pump-less) writer. The writer's + /// admission gate grants a reservation whenever the queue is empty, + /// so the spawned drain completes its CaughtUp hold in-process. + fn new(name: &str, pads: usize, final_marker: &str) -> Self { + // Small pages: every drive emits at most a couple of frames, + // so the credit-by-credit walk crosses the target boundary in + // observable steps. + Self::new_with_script(name, race_script(pads, final_marker), 240) + } + + /// The QUIET twin (E2R3): the script never exits and the page + /// budget is 0 — ONE frame per page, the registry's own + /// deterministic-cursor idiom — so the concurrent-staging tests + /// can name the EXACT credit whose drive reaches the attach + /// target (the racing credit) without probing page boundaries. + fn new_quiet(name: &str, pads: usize, final_marker: &str) -> Self { + Self::new_with_script(name, quiet_race_script(pads, final_marker), 0) + } + + fn new_with_script(name: &str, script: String, page_budget: i64) -> Self { + let registry = freshell_terminal::TerminalRegistry::new(); + registry.set_paced_page_max_bytes(page_budget); + let terminal_id = format!("T-{name}"); + let exit_registry = registry.clone(); + let exit_terminal_id = terminal_id.clone(); + let on_exit: freshell_terminal::pty::ExitHook = Box::new(move |exit_code: i64| { + exit_registry.finish_pty_exit(&exit_terminal_id, exit_code); + }); + let spec = freshell_platform::SpawnSpec { + program: "/bin/sh".into(), + args: vec!["-c".into(), script], + env_overrides: BTreeMap::new(), + cwd: None, + cols: 120, + rows: 30, + }; + registry + .create( + &spec, + &BTreeMap::new(), + terminal_id.clone(), + "S".into(), + "shell", + None, + None, + None, + Some(on_exit), + ) + .expect("spawn race-script PTY"); + let collector: Collector = Arc::new(Mutex::new(Vec::new())); + let sink = collector_sink(&collector); + let (writer, writer_pump) = connection_writer::WriterSender::new( + 16 * 1024 * 1024, + 1024 * 1024, + std::time::Duration::from_secs(5), + ); + let (cancel_tx, cancel_rx) = tokio::sync::watch::channel(false); + Self { + registry, + terminal_id, + collector, + sink, + writer, + _cancel_tx: cancel_tx, + _writer_pump: writer_pump, + cancel_rx, + sessions: crate::paced_replay::PacedSessions::default(), + conn_id: 1, + arid: format!("arid-{name}"), + } + } + + fn attach_paced(&self, since_seq: i64) -> freshell_terminal::PacedAttachStart { + let outcome = self.registry.attach( + &self.terminal_id, + self.conn_id, + Arc::clone(&self.sink), + Some(self.arid.clone()), + since_seq, + false, + true, + None, + None, + None, + freshell_terminal::PacedAttachOptions::default(), + ); + outcome.paced.expect("the paced attach path") + } + + /// Wait for the quiet script's ECHO-OFF banner: the `stty -echo` + /// before it has taken effect by then, so every later input line + /// produces EXACTLY its printf frame — no echo frames, no + /// echo/output coalescing. The banner itself is the ring's + /// deterministic frame 1, which the frame-counting tests fold + /// into their seeded windows. + async fn wait_ready(&self) { + let registry = self.registry.clone(); + let terminal_id = self.terminal_id.clone(); + wait_for( + move || { + registry + .directory() + .iter() + .find(|entry| entry.terminal_id == terminal_id) + .map(|entry| entry.snapshot.clone()) + .filter(|snapshot| snapshot.contains("ECHO-OFF")) + }, + "the quiet script's ECHO-OFF banner", + ) + .await; + } + + fn start_session(&mut self, start: freshell_terminal::PacedAttachStart, since: i64) { + crate::paced_replay::start_session( + &self.registry, + self.conn_id, + &self.sink, + &mut self.sessions, + start, + since, + None, + self.writer.clone(), + self.cancel_rx.clone(), + ); + } + + fn session(&mut self) -> crate::paced_replay::PacedSession { + self.try_session() + .expect("the session is in the credited table") + } + + fn try_session(&mut self) -> Option { + self.sessions + .get_mut(&self.terminal_id) + .map(|session| session.clone()) + } + + fn credit(&mut self, consumed_seq: i64) { + let credit = freshell_protocol::TerminalReplayCredit { + terminal_id: self.terminal_id.clone(), + stream_id: "S".into(), + attach_request_id: self.arid.clone(), + consumed_seq, + }; + handle_replay_credit( + &credit, + &self.registry, + self.conn_id, + &self.sink, + &mut self.sessions, + &self.writer, + &self.cancel_rx, + ); + } + + /// Write one input line and wait for the marker it produces to be + /// observable in the ring (the deterministic per-step barrier). + async fn step(&self, line: &str, marker: &str) { + let registry = self.registry.clone(); + let terminal_id = self.terminal_id.clone(); + let wanted = marker.to_string(); + let outcome = self + .registry + .input(&self.terminal_id, format!("{line}\n").as_bytes()); + assert!(outcome.found, "the input write reaches the live PTY"); + wait_for( + move || { + registry + .directory() + .iter() + .find(|entry| entry.terminal_id == terminal_id) + .map(|entry| entry.snapshot.clone()) + .filter(|snapshot| snapshot.contains(&wanted)) + }, + &format!("the ring to hold {marker}"), + ) + .await; + } + + async fn wait_staged(&self) -> i64 { + let registry = self.registry.clone(); + let terminal_id = self.terminal_id.clone(); + let conn_id = self.conn_id; + wait_for( + move || registry.staged_paced_exit(&terminal_id, conn_id), + "the natural exit to stage behind the armed deferral", + ) + .await + } + + fn head(&self) -> i64 { + self.registry + .replay_bounds(&self.terminal_id) + .expect("replay bounds") + .head_seq + } + + /// The ring head once in-flight PTY chunks have landed. A small + /// write can split across PTY read chunks — a marker's trailing + /// bytes may land as a separate ring frame microseconds after the + /// marker text first appears (observed on 2-core CI runners, + /// never on the 96-core dev box) — so frame counts are only + /// stable once the ring is quiet. Progress-based: the quiet + /// clock resets on every observed head advance, and the fixed + /// bound trips only on a dead PTY, never a slow one. + async fn settled_head(&self) -> i64 { + let deadline = tokio::time::Instant::now() + std::time::Duration::from_secs(30); + let mut last = self.head(); + let mut quiet_since = tokio::time::Instant::now(); + loop { + tokio::time::sleep(std::time::Duration::from_millis(5)).await; + let now = self.head(); + if now != last { + last = now; + quiet_since = tokio::time::Instant::now(); + } else if quiet_since.elapsed() >= std::time::Duration::from_millis(25) { + return now; + } + assert!( + tokio::time::Instant::now() < deadline, + "the ring head never settled — a dead PTY (the quiet window \ + resets on every advance, so this trips only on a dead one)" + ); + } + } + + /// Assert NO terminal.exit is delivered while an exit page sits + /// uncredited — the invariant-2 hold, over a deterministic window. + async fn assert_exit_held(&self) { + let before = exit_count(&self.collector); + let hold = std::time::Instant::now() + std::time::Duration::from_millis(1_500); + while std::time::Instant::now() < hold { + tokio::time::sleep(std::time::Duration::from_millis(50)).await; + let now = exit_count(&self.collector); + assert_eq!( + now, before, + "terminal.exit must NOT deliver while the page reaching the \ + armed exit head is uncredited — it rides the credit that \ + acknowledges that page (invariant 2)" + ); + } + } + + async fn assert_exit_arrives_and_is_last(&self, expected_code: i64) { + let collector = Arc::clone(&self.collector); + wait_for( + move || (exit_count(&collector) > 0).then_some(()), + "terminal.exit to arrive on the acknowledging credit", + ) + .await; + assert_eq!( + exit_count(&self.collector), + 1, + "exactly one terminal.exit may ever arrive (invariant 3)" + ); + assert!( + last_is_exit(&self.collector), + "terminal.exit must be the client's last frame" + ); + let code = self + .collector + .lock() + .expect("collector lock") + .iter() + .find_map(|m| match m { + ServerMessage::TerminalExit(exit) => Some(exit.exit_code), + _ => None, + }) + .expect("the exit frame"); + assert_eq!(code, expected_code); + } + } + + /// E2R3, (a) THE CONCURRENT-STAGING RACE at the credit path: the + /// ordering the finding pins at the credit handler, in the one state + /// where it strands the stream: the client's first page was EMPTY (its + /// cursor already sat at the attach head: `credited == page_end == + /// target`), and the exit staged between the attach and + /// `start_session` with NEW output past that target. The arm extends + /// the phase target beyond the credited cursor, so the drive MUST + /// produce the next page — a keep-with-no-page wedges the session + /// forever (nothing outstanding, so no credit can ever come). + #[tokio::test] + async fn empty_first_page_exit_race_extends_the_drive_not_a_wedge() { + let mut harness = RaceHarness::new("wedge", 0, "FINAL-WEDGE"); + // The client attaches fully caught up: since == head == 0 (the + // script emits nothing before its first input), so the first page + // is empty and the session starts with nothing outstanding. + let head_at_attach = harness.head(); + assert_eq!(head_at_attach, 0, "the script is quiet until driven"); + let start = harness.attach_paced(head_at_attach); + assert_eq!( + start.session.page_end, start.session.effective_since, + "the empty first page leaves nothing outstanding" + ); + // The exit stages with new output while the notify is still + // queued (the test IS the undispatched dispatcher window). + harness.step("go", "FINAL-WEDGE").await; + let exit_code = harness.wait_staged().await; + let exit_head = harness.head(); + assert!( + exit_head > start.session.target, + "the exit added output past the original target" + ); + harness.start_session(start, head_at_attach); + // INVARIANT 1: the arm extended the target beyond the credited + // cursor, so the drive MUST have produced the next page — the + // pre-fix keep-with-no-page wedged here with NOTHING outstanding + // (no page, no exit, ever). The session owes exactly one + // uncredited page and its first frames are already sunk. + let session = harness.session(); + assert!( + session.credited < session.page_end, + "the drive produced the extended phase's next page \ + (no wedge state: credited {}, page_end {})", + session.credited, + session.page_end, + ); + assert!( + !outputs(&harness.collector).is_empty(), + "the extended phase's first page was sunk to the client" + ); + // INVARIANT 2: the exit waits for the credit that acknowledges + // the page reaching the armed exit head. Credit the pages one by + // one until that page is outstanding, hold it uncredited, then + // acknowledge it — the exit rides THAT credit. + let mut guards = 0; + loop { + let session = harness.session(); + assert!( + session.page_end <= exit_head, + "pages never overshoot the frozen head" + ); + if session.page_end == exit_head { + break; + } + harness.assert_exit_held().await; + let consumed = session.page_end; + harness.credit(consumed); + guards += 1; + assert!(guards < 10_000, "the credit walk must converge"); + } + harness.assert_exit_held().await; + harness.credit(exit_head); + harness.assert_exit_arrives_and_is_last(exit_code).await; + assert!( + outputs(&harness.collector) + .iter() + .any(|data| data.contains("FINAL-WEDGE")), + "the deferred final output was delivered before the exit" + ); + assert!( + harness.sessions.get_mut(&harness.terminal_id).is_none(), + "the session left the credited table on the acknowledging credit" + ); + assert_eq!( + harness + .registry + .staged_paced_exit(&harness.terminal_id, harness.conn_id), + None, + "the subscriber retired with the delivered exit" + ); + } + + /// (b) THE PREMATURE EXIT, through the credit path: the exit stages + /// mid-restore (the staging is in the registry before any credit + /// runs — the notify-undispatched window), the credits drive the + /// session through the original target and on to the exit page, and + /// the page that reaches the armed exit head is read but NOT + /// credited. No exit may deliver on the drive that emitted it — the + /// removal rides the credit that acknowledges that page. + #[tokio::test] + async fn exit_page_read_uncredited_holds_the_exit_for_its_credit() { + let mut harness = RaceHarness::new("preempt", 8, "FINAL-PREEMPT"); + // Seed the pre-exit window: every pad step lands in the ring + // deterministically before the paced attach (each step waits for + // its OWN unique marker, so the attach cannot race a pad's + // asynchronous ingestion). + for step in 0..8 { + harness.step("pad", &format!("PAD-STEP-{step}")).await; + } + let head_at_attach = harness.head(); + assert!(head_at_attach > 0, "the pre-exit window is non-empty"); + let start = harness.attach_paced(0); + assert!( + start.session.page_end < start.session.target, + "the first page is a bounded prefix (mid-restore)" + ); + harness.start_session(start, 0); + // The exit stages with new output past the attach target, before + // any credit runs (the undispatched-notify race window). + harness.step("go", "FINAL-PREEMPT").await; + let exit_code = harness.wait_staged().await; + let exit_head = harness.head(); + assert!( + exit_head > head_at_attach, + "the exit added output past the original target" + ); + // Credit the pages one by one — the drive must keep producing + // (invariant 1) — until the page reaching the FROZEN exit head is + // outstanding. THE PARSER-CONSUMPTION BOUNDARY: that page is + // read (sunk) but NOT credited. + let mut guards = 0; + let converged = loop { + let Some(session) = harness.try_session() else { + // The session left the credited table mid-walk — the + // pre-fix premature removal (the armed session was + // removed on the drive that EMITTED the exit page, while + // that page was still uncredited). + break false; + }; + assert!( + session.page_end <= exit_head, + "pages never overshoot the frozen head" + ); + if session.page_end == exit_head { + break true; + } + let consumed = session.page_end; + harness.credit(consumed); + guards += 1; + assert!(guards < 10_000, "the credit walk must converge"); + }; + assert!( + outputs(&harness.collector) + .iter() + .any(|data| data.contains("FINAL-PREEMPT")), + "the exit page carrying the final marker was read" + ); + if !converged { + // THE PREMATURE-EXIT RACE (RED pre-fix): the removal already + // delivered (or is about to deliver) terminal.exit through the + // drain's CaughtUp hold — BEFORE any credit could acknowledge + // the page reaching the armed exit head. + let collector = Arc::clone(&harness.collector); + wait_for( + move || (exit_count(&collector) > 0).then_some(()), + "the premature exit (the violation under test)", + ) + .await; + panic!( + "invariant 2 violated: the session was removed from the \ + credited table on the drive that EMITTED the page reaching \ + the armed exit head, and terminal.exit delivered while that \ + page was uncredited — the exit must ride the credit that \ + acknowledges it" + ); + } + // THE HOLD (invariant 2): the page reaching exit_head is + // uncredited — the pre-fix removal delivered terminal.exit here. + harness.assert_exit_held().await; + // The acknowledging credit: the exit arrives and is the last frame. + harness.credit(exit_head); + harness.assert_exit_arrives_and_is_last(exit_code).await; + assert!( + harness.sessions.get_mut(&harness.terminal_id).is_none(), + "the session left the credited table on the acknowledging credit" + ); + // (c) no double-delivery: a spurious duplicate credit after the + // exit is inert, and the terminal cannot stage a second exit. + let before = exit_count(&harness.collector); + harness.credit(exit_head); + assert_eq!( + exit_count(&harness.collector), + before, + "a post-exit credit grants nothing (no second exit)" + ); + assert!( + !harness.registry.finish_pty_exit(&harness.terminal_id, 99), + "a second exit never restages (monotone, once-only)" + ); + } + + /// E2R3, (a) THE CONCURRENT-STAGING RACE at the credit path: the + /// natural exit stages INSIDE the credit handling's window — the + /// registry's one-shot hook fires the staging inside the arm read's + /// own lock scope, immediately after the read observes the + /// pre-staging state (the deterministic model of the PTY reader + /// staging concurrently with the decision's use of its read; never + /// a sleep-based race). The racing credit is the one whose drive + /// reaches the attach target — with budget 0 (one frame per page) + /// the walk is countable, so the racing credit is named, not probed. + /// Absolute frame counts are deliberately NOT pinned: a small write + /// can split across PTY read chunks on slow runners (a marker's + /// trailing bytes landing as a separate ring frame microseconds + /// later), so the walk pins head PROGRESSION via settled reads. + /// + /// PRE-FIX (RED): the arm's read went stale (check-then-act) — the + /// handler sees `exit_head == None`, `uncredited_exit_page()` is + /// false, and the session transfers to the uncredited drain; the + /// drain dumps the remaining suffix and delivers terminal.exit + /// WITHOUT the credit that would have acknowledged the page + /// reaching the frozen head. + /// + /// POST-FIX (GREEN): the phase-transition decision is ATOMIC — the + /// disposition re-reads the staged-exit state under ONE registry + /// lock hold after the drive, so the concurrently staged exit is + /// absorbed by the SAME handling: the credited phase extends + /// through the frozen head, the final output pages only on + /// continuation credits, and terminal.exit rides the credit that + /// acknowledges the page reaching the armed exit head. + #[tokio::test] + async fn credit_path_exit_staged_inside_the_window_extends_the_credited_phase() { + let mut harness = RaceHarness::new_quiet("creditrace", 1, "FINAL-CREDRACE"); + harness.wait_ready().await; + let banner_head = harness.settled_head().await; + assert!( + banner_head >= 1, + "the ECHO-OFF banner is in the ring before any step" + ); + // Seed the replay window: one pad step lands the pad marker (the + // harness script keeps echo off). Transport may split a step's + // bytes across ring frames on slow runners, so the walk pins head + // PROGRESSION with settled reads, not absolute frame counts; the + // first page (one frame per page) leaves the rest for the racing + // credit's drive. + harness.step("pad", "PAD-STEP-0").await; + let target = harness.settled_head().await; + assert!( + target > banner_head, + "the pad step advanced the ring past the banner (echo off)" + ); + let start = harness.attach_paced(0); + assert_eq!( + start.session.page_end, 1, + "budget 0: the first page is exactly frame 1" + ); + harness.start_session(start, 0); + // The "final output": frames past the attach target — the exit + // will freeze THIS (settled) head. + harness.step("go", "FINAL-CREDRACE").await; + let exit_head = harness.settled_head().await; + assert!( + exit_head > target, + "the final-marker step advanced the head past the attach target" + ); + let exit_code = 0; + // THE RACING CREDIT: acknowledges frame 1 — its drive produces + // the page reaching the attach target, and the exit stages + // INSIDE this credit's arm→decision window (the hook fires + // inside the arm read's lock scope, right after the read). + harness + .registry + .set_paced_exit_stage_hook_for_tests(harness.conn_id, exit_code); + harness.credit(1); + let Some(session) = harness.try_session() else { + // THE CONCURRENT-STAGING VIOLATION (RED pre-fix): the credit + // whose drive reached the target transferred the session to + // the uncredited drain even though the natural exit staged + // INSIDE the credit's window — the drain then dumped the + // remaining suffix and delivered terminal.exit without the + // credit acknowledging the page reaching the frozen head. + let collector = Arc::clone(&harness.collector); + wait_for( + move || (exit_count(&collector) > 0).then_some(()), + "the premature exit (the violation under test)", + ) + .await; + panic!( + "the phase transition was decided from a stale staged-exit \ + read: the session transferred to the uncredited drain while \ + the natural exit staged concurrently with the credit's arm \ + — the transition decision must be atomic with respect to \ + exit staging" + ); + }; + // GREEN: the atomic disposition absorbed the concurrently staged + // exit — the credited phase EXTENDED through the frozen head. + assert_eq!( + session.exit_head, + Some(exit_head), + "the disposition armed the staged exit's frozen head" + ); + assert!( + session.credited < session.page_end, + "the racing credit's page (reaching the original target) is the \ + one outstanding uncredited page" + ); + // Pages flow ONLY on continuation credits; the exit rides the + // credit acknowledging the page reaching the armed exit head. + let mut guards = 0; + loop { + let session = harness.session(); + assert!( + session.page_end <= exit_head, + "pages never overshoot the frozen head" + ); + if session.page_end == exit_head { + break; + } + harness.assert_exit_held().await; + let consumed = session.page_end; + harness.credit(consumed); + guards += 1; + assert!(guards < 10_000, "the credit walk must converge"); + } + harness.assert_exit_held().await; + harness.credit(exit_head); + harness.assert_exit_arrives_and_is_last(exit_code).await; + assert!( + outputs(&harness.collector) + .iter() + .any(|data| data.contains("FINAL-CREDRACE")), + "the deferred final output was delivered before the exit" + ); + assert!( + harness.sessions.get_mut(&harness.terminal_id).is_none(), + "the session left the credited table on the acknowledging credit" + ); + assert_eq!( + harness + .registry + .staged_paced_exit(&harness.terminal_id, harness.conn_id), + None, + "the subscriber retired with the delivered exit" + ); + } + + /// E2R3, (b) THE SAME INTERLEAVE at start_session: the exit stages + /// during the attach/start window — the hook fires inside the + /// start's arm read's lock scope, immediately after it. The EMPTY + /// first page (the client's cursor already sat at the attach head) + /// is the start state whose disposition the window poisons. + /// + /// PRE-FIX (RED): the start's arm read goes stale — start_session + /// transfers the session to the uncredited drain with everything + /// acknowledged, and the drain dumps the final suffix + + /// terminal.exit with no credit at all. + /// + /// POST-FIX (GREEN): the session STARTS with the exit armed, pages + /// flow on continuation credits, and the exit rides the final + /// acknowledging credit. + #[tokio::test] + async fn start_session_exit_staged_during_attach_arms_before_any_transfer() { + let mut harness = RaceHarness::new_quiet("startrace", 0, "FINAL-STARTRACE"); + harness.wait_ready().await; + // The client attaches fully caught up: since == head == 1 (the + // ECHO-OFF banner frame) — the first page is EMPTY (nothing + // outstanding at start). + let head_at_attach = harness.head(); + assert_eq!( + head_at_attach, 1, + "the ECHO-OFF banner is the ring's frame 1" + ); + let start = harness.attach_paced(head_at_attach); + assert_eq!( + start.session.page_end, start.session.effective_since, + "the empty first page leaves nothing outstanding" + ); + // The exit stages with final output past the attach head while + // the start path is the in-flight decision (the hook fires + // inside the start arm read's lock scope, right after it). + harness.step("go", "FINAL-STARTRACE").await; + let exit_head = harness.head(); + assert_eq!( + exit_head, 2, + "the final-marker step adds exactly one frame past the head" + ); + let exit_code = 0; + harness + .registry + .set_paced_exit_stage_hook_for_tests(harness.conn_id, exit_code); + harness.start_session(start, head_at_attach); + let Some(session) = harness.try_session() else { + // THE CONCURRENT-STAGING VIOLATION (RED pre-fix): the start + // transferred the session to the uncredited drain even though + // the natural exit staged during the attach/start window. + let collector = Arc::clone(&harness.collector); + wait_for( + move || (exit_count(&collector) > 0).then_some(()), + "the premature exit (the violation under test)", + ) + .await; + panic!( + "start_session decided the phase transition from a stale \ + staged-exit read: the session transferred to the uncredited \ + drain while the natural exit staged during the attach \ + window — the transition decision must be atomic with \ + respect to exit staging" + ); + }; + // GREEN: the session STARTS with the exit armed and the + // extended phase's first page outstanding (no wedge state). + assert_eq!( + session.exit_head, + Some(exit_head), + "the start armed the staged exit's frozen head" + ); + assert!( + session.credited < session.page_end, + "the start's wedge-guard drive produced the extended phase's \ + first page" + ); + assert!( + !outputs(&harness.collector).is_empty(), + "the extended phase's first page was sunk to the client" + ); + // Pages flow on credits; the exit rides the final acknowledging + // credit. + let mut guards = 0; + loop { + let session = harness.session(); + assert!( + session.page_end <= exit_head, + "pages never overshoot the frozen head" + ); + if session.page_end == exit_head { + break; + } + harness.assert_exit_held().await; + let consumed = session.page_end; + harness.credit(consumed); + guards += 1; + assert!(guards < 10_000, "the credit walk must converge"); + } + harness.assert_exit_held().await; + harness.credit(exit_head); + harness.assert_exit_arrives_and_is_last(exit_code).await; + assert!( + outputs(&harness.collector) + .iter() + .any(|data| data.contains("FINAL-STARTRACE")), + "the deferred final output was delivered before the exit" + ); + assert!( + harness.sessions.get_mut(&harness.terminal_id).is_none(), + "the session left the credited table on the acknowledging credit" + ); + } + + /// E2R3, (c) THE ATOMICITY BOUNDARY — the legitimately + /// post-transfer exit: staged STRICTLY AFTER the atomic transfer + /// decision (the disposition that confirmed, under its one lock + /// hold, that no exit was staged), it is the uncredited drain's + /// documented tail content. The drain delivers the remaining tail + /// pages and then the staged exit at its completing verdict — + /// WITHOUT any credit after the transfer. That is exactly the + /// documented tail semantics, not a loss and not a re-entry into + /// the credited phase: the atomicity scope ends at the transfer + /// decision's lock hold. + /// + /// The staging lands in the synchronous window after the + /// transferring credit and before the spawned drain task's first + /// poll (the current-thread runtime has not yielded), so the + /// "strictly after the atomic transfer" ordering is deterministic. + #[tokio::test] + async fn post_transfer_staged_exit_is_the_drains_documented_tail_content() { + let mut harness = RaceHarness::new_quiet("boundary", 1, "FINAL-BOUNDARY"); + harness.wait_ready().await; + let banner_head = harness.settled_head().await; + assert!( + banner_head >= 1, + "the ECHO-OFF banner is in the ring before any step" + ); + harness.step("pad", "PAD-STEP-0").await; + let target = harness.settled_head().await; + assert!( + target > banner_head, + "the pad step advanced the ring past the banner (echo off)" + ); + let start = harness.attach_paced(0); + assert_eq!( + start.session.page_end, 1, + "budget 0: the first page is exactly frame 1" + ); + harness.start_session(start, 0); + // The tail the drain will page: frames past the attach + // target, ingested BEFORE the transferring credit (the drain's + // fixed target captures it at its start). + harness.step("go", "FINAL-BOUNDARY").await; + let head_at_transfer = harness.settled_head().await; + assert!( + head_at_transfer > target, + "the final-marker step advanced the head past the attach target" + ); + // The transferring credits: their drives walk the pages up to + // the attach target with NO exit staged — the disposition + // atomically confirms that under its lock hold and transfers. + // No hook is armed: nothing stages concurrently here. (On a + // fast box the first credit is the transferring one; transport + // frame-splitting on slow runners may add intermediate pages, + // so credit until the drive reaches the target. Every call is + // synchronous — the current-thread runtime cannot yield between + // the transferring credit and the strictly-post-transfer + // staging below.) + let mut transfer_steps = 0; + loop { + if harness.try_session().is_none() { + break; + } + let consumed = harness.session().page_end; + harness.credit(consumed); + transfer_steps += 1; + assert!(transfer_steps < 10_000, "the transfer walk must converge"); + } + assert!( + harness.try_session().is_none(), + "the credited phase completed and the session moved to the drain" + ); + // STRICTLY POST-TRANSFER staging: from outside any decision, + // after the atomic transfer and before the drain's first poll. + assert!( + harness + .registry + .stage_natural_exit_for_test(&harness.terminal_id, harness.conn_id, 0), + "the post-transfer staging lands on the live deferred subscriber" + ); + // The drain delivers the tail pages and THEN the staged exit at + // its completing verdict — no credit is ever sent after the + // transfer (the documented uncredited tail semantics). + harness.assert_exit_arrives_and_is_last(0).await; + assert!( + outputs(&harness.collector) + .iter() + .any(|data| data.contains("FINAL-BOUNDARY")), + "the tail frames were delivered before the exit" + ); + // A post-exit credit is inert: the session left the credited + // table at the transfer and the subscriber retired with the + // exit. + let before = exit_count(&harness.collector); + harness.credit(head_at_transfer); + assert_eq!( + exit_count(&harness.collector), + before, + "a post-exit credit grants nothing (stale generation)" + ); + assert_eq!( + harness + .registry + .staged_paced_exit(&harness.terminal_id, harness.conn_id), + None, + "the subscriber retired with the delivered exit" + ); + } +} diff --git a/crates/freshell-ws/src/terminal_delivery_queue.rs b/crates/freshell-ws/src/terminal_delivery_queue.rs index e9f84c636..8eda8f9d3 100644 --- a/crates/freshell-ws/src/terminal_delivery_queue.rs +++ b/crates/freshell-ws/src/terminal_delivery_queue.rs @@ -40,6 +40,21 @@ pub enum Delivery { Gap { terminal_id: String, range: Range }, } +/// One admission-time eviction record (responsive-terminal-restore W3 +/// observability, task-007 review M2): the evicted frame's own terminal and +/// range, in global-oldest eviction order. The connection writer drains +/// these after `push` (under the same admission lock) and emits its +/// rate-limited `ws.terminal_stream.queue_overflow_spill` event at the +/// moment the eviction happens — a connection that dies while backlogged +/// (its coalesced gap never leased) still leaves spill evidence in the log. +/// Supersede discards (`discard_terminal`) are not evictions and produce no +/// records. +#[derive(Clone, Debug)] +pub struct EvictedOutput { + pub terminal_id: String, + pub range: Range, +} + #[derive(Clone, Copy, Debug, PartialEq, Eq)] pub enum CapacityError { MetadataLimit, @@ -85,6 +100,12 @@ pub struct DeliveryQueue { byte_limit: usize, metadata_limit: usize, gap_count: usize, + /// Admission-time eviction records (task-007 review M2): drained by the + /// caller via [`Self::take_evictions`] right after `push`. `Vec::new()` + /// allocates nothing, so admissions that evict nothing pay no cost, and + /// `std::mem::take` returns the buffer without retaining a flood-peak + /// capacity on the queue. + evictions: Vec, class_served: [u128; 3], class_clock: u128, lane_clock: [u128; 3], @@ -102,6 +123,7 @@ impl DeliveryQueue { byte_limit: byte_limit.max(1), metadata_limit: metadata_limit.max(1), gap_count: 0, + evictions: Vec::new(), class_served: [0; 3], class_clock: 0, lane_clock: [0; 3], @@ -193,6 +215,10 @@ impl DeliveryQueue { .expect("entry has lane"); lane.ids.remove(&id); let range = entry.range.expect("only output is evictable"); + self.evictions.push(EvictedOutput { + terminal_id: entry.terminal_id, + range: range.clone(), + }); if let Some(last) = lane.gaps.back_mut() { if last.stream_id == range.stream_id && last.attach_request_id == range.attach_request_id @@ -212,6 +238,15 @@ impl DeliveryQueue { Ok(()) } + /// Drain the eviction records accumulated by [`Self::push`] (task-007 + /// review M2): one record per evicted entry, in eviction order, taken + /// atomically so the next admission starts from a clean slate. Call + /// under the caller's admission lock immediately after `push`; an empty + /// return means no eviction happened. + pub fn take_evictions(&mut self) -> Vec { + std::mem::take(&mut self.evictions) + } + /// Atomic caller snapshot: update scheduling only, never attachment, /// generation, geometry, sequence state, or the PTY process. pub fn update_priorities(&mut self, mut priority: impl FnMut(&str) -> Priority) { diff --git a/crates/freshell-ws/src/terminal_delivery_queue_tests.rs b/crates/freshell-ws/src/terminal_delivery_queue_tests.rs index 65e951952..f2b474e81 100644 --- a/crates/freshell-ws/src/terminal_delivery_queue_tests.rs +++ b/crates/freshell-ws/src/terminal_delivery_queue_tests.rs @@ -192,6 +192,40 @@ fn late_arriving_focused_lane_has_no_historical_background_debt() { put(&mut q, "f", Priority::Focused, 0, 100); assert_eq!(next(&mut q).0, "f"); } +/// Task-007 review M2 (landed by task-010): every eviction surfaces at +/// ADMISSION time as one record per evicted entry (the evicted frame's own +/// terminal + range, in global-oldest eviction order), so the connection +/// writer can emit the rate-limited spill event when the eviction HAPPENS — +/// not when (or if) the coalesced gap is later leased to the socket. Records +/// drain exactly once; supersede discards are NOT spills and surface nothing. +#[test] +fn eviction_records_surface_at_admission_time_per_evicted_entry() { + let mut q = DeliveryQueue::new(200, 1000); + put(&mut q, "t", Priority::Focused, 1, 120); + put(&mut q, "t", Priority::Focused, 2, 120); + put(&mut q, "t", Priority::Focused, 3, 120); + let evicted = q.take_evictions(); + assert_eq!( + evicted.len(), + 2, + "pushes 2 and 3 each evicted the global-oldest entry: {evicted:?}" + ); + assert_eq!(evicted[0].terminal_id, "t"); + assert_eq!((evicted[0].range.from_seq, evicted[0].range.to_seq), (1, 1)); + assert_eq!((evicted[1].range.from_seq, evicted[1].range.to_seq), (2, 2)); + assert!( + q.take_evictions().is_empty(), + "eviction records drain in one take" + ); + // Supersede (discard_terminal) removes the RETAINED frame's queued + // bytes WITHOUT being an eviction episode — it must not fabricate spill + // records. + q.discard_terminal("t"); + assert!( + q.take_evictions().is_empty(), + "supersede discards are not spills" + ); +} #[test] fn global_oldest_eviction_remains_explicit_and_generation_scoped() { let mut q = DeliveryQueue::new(250, 100); diff --git a/crates/freshell-ws/src/terminal_interest.rs b/crates/freshell-ws/src/terminal_interest.rs index dd1fe1e44..157a88b4d 100644 --- a/crates/freshell-ws/src/terminal_interest.rs +++ b/crates/freshell-ws/src/terminal_interest.rs @@ -1,6 +1,12 @@ //! Connection-local presentation interest. It changes delivery order only. //! Omitted terminals in an authoritative snapshot are background terminals; //! before the first snapshot, the existing attach.priority is the fallback. +//! +//! Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1) +//! ride the same snapshots on the negotiated `claimedTerminalIds` field: the +//! connection-local diff (`InterestClaimChange`) is handed back to the +//! dispatcher, which applies it to the terminal registry. The claim never +//! changes delivery priority — only `focused`/`visible` do. use super::delivery::Priority; use freshell_protocol::client_messages::TerminalInterest; @@ -9,6 +15,21 @@ use std::collections::{BTreeMap, BTreeSet}; pub(super) const MAX_INTEREST_TERMINALS: usize = 1024; const MAX_SAFE_REVISION: u64 = 9_007_199_254_740_991; +/// One applied snapshot's claim-set diff (negotiated connections only): the +/// ids this connection newly claims and the ids its previous snapshot +/// claimed that this one no longer does (the explicit withdrawal). +#[derive(Debug, Clone, PartialEq, Eq, Default)] +pub(crate) struct InterestClaimChange { + pub added: Vec, + pub removed: Vec, +} + +impl InterestClaimChange { + pub(crate) fn is_empty(&self) -> bool { + self.added.is_empty() && self.removed.is_empty() + } +} + #[derive(Default)] pub(super) struct InterestState { enabled: bool, @@ -16,12 +37,21 @@ pub(super) struct InterestState { focused: Option, visible: BTreeSet, attachments: BTreeMap, + /// The connection negotiated `terminalLifetimeClaimV1`: its snapshots may + /// carry `claimedTerminalIds` (hidden-pane lifetime claims). Without it + /// the field is ignored server-side (validated nowhere, applied nowhere). + claims_enabled: bool, + /// The last ACCEPTED claim set (superseded snapshot-by-snapshot). + claimed: BTreeSet, } impl InterestState { pub(super) fn enable(&mut self) { self.enabled = true; } + pub(super) fn enable_claims(&mut self) { + self.claims_enabled = true; + } pub(super) fn priority(&self, terminal_id: &str) -> Priority { if self.revision.is_some() { if self.focused.as_deref() == Some(terminal_id) { @@ -67,7 +97,14 @@ impl InterestState { pub(super) fn detach(&mut self, terminal_id: &str) { self.attachments.remove(terminal_id); } - pub(super) fn apply(&mut self, snapshot: &TerminalInterest) -> Result { + /// Apply one snapshot. `Ok(Some(change))` = accepted (the caller applies + /// `change` to the terminal registry and recomputes delivery priorities); + /// `Ok(None)` = stale revision, nothing changed; `Err` = rejected without + /// replacing the last accepted state. + pub(super) fn apply( + &mut self, + snapshot: &TerminalInterest, + ) -> Result, &'static str> { if !self.enabled { return Err("terminalInterestV1 was not negotiated"); } @@ -94,16 +131,46 @@ impl InterestState { { return Err("Focused terminal must be visible"); } + // Claim-set validation runs only for a connection that negotiated the + // capability: a non-negotiated sender's field is ignored entirely + // (robustness — the client gates it send-side, the server must not + // depend on that). + let claimed_input = match (&self.claims_enabled, &snapshot.claimed_terminal_ids) { + (true, Some(claimed)) => { + if claimed.len() > MAX_INTEREST_TERMINALS { + return Err("Too many claimed terminal identifiers"); + } + if !claimed.iter().all(|id| valid_id(id)) { + return Err("Invalid terminal interest identifier"); + } + Some(claimed) + } + _ => None, + }; if self .revision .is_some_and(|revision| snapshot.revision <= revision) { - return Ok(false); + return Ok(None); } self.revision = Some(snapshot.revision); self.focused.clone_from(&snapshot.focused_terminal_id); self.visible = visible; - Ok(true) + // Latest snapshot's claim set wins per connection: `Some(set)` + // supersedes; `None` carries no claim information and leaves the last + // accepted set standing. + let mut change = InterestClaimChange::default(); + if let Some(claimed) = claimed_input { + let next: BTreeSet = claimed.iter().cloned().collect(); + for id in next.difference(&self.claimed) { + change.added.push(id.clone()); + } + for id in self.claimed.difference(&next) { + change.removed.push(id.clone()); + } + self.claimed = next; + } + Ok(Some(change)) } } @@ -115,8 +182,12 @@ mod tests { revision, focused_terminal_id: focused.map(str::to_string), visible_terminal_ids: visible.iter().map(|s| s.to_string()).collect(), + claimed_terminal_ids: None, } } + fn applied() -> InterestClaimChange { + InterestClaimChange::default() + } #[test] fn requires_negotiation() { assert!(InterestState::default() @@ -135,13 +206,16 @@ mod tests { fn full_snapshot_is_authoritative_and_newer_revision_wins() { let mut state = InterestState::default(); state.enable(); - assert_eq!(state.apply(&snapshot(3, Some("a"), &["a", "b"])), Ok(true)); + assert_eq!( + state.apply(&snapshot(3, Some("a"), &["a", "b"])), + Ok(Some(applied())) + ); assert_eq!(state.priority("a"), Priority::Focused); assert_eq!(state.priority("b"), Priority::Visible); assert_eq!(state.priority("c"), Priority::Background); - assert_eq!(state.apply(&snapshot(2, Some("b"), &["b"])), Ok(false)); + assert_eq!(state.apply(&snapshot(2, Some("b"), &["b"])), Ok(None)); assert_eq!(state.priority("a"), Priority::Focused); - assert_eq!(state.apply(&snapshot(4, None, &[])), Ok(true)); + assert_eq!(state.apply(&snapshot(4, None, &[])), Ok(Some(applied()))); assert_eq!(state.priority("a"), Priority::Background); } #[test] @@ -180,4 +254,134 @@ mod tests { state.apply(&snapshot(1, Some("old"), &["old"])).unwrap(); assert_eq!(state.priority("replacement"), Priority::Background); } + + // ── Hidden-pane lifetime claims (responsive-terminal-restore WS1) ── + + fn claimed_snapshot( + revision: u64, + focused: Option<&str>, + visible: &[&str], + claimed: Option<&[&str]>, + ) -> TerminalInterest { + TerminalInterest { + revision, + focused_terminal_id: focused.map(str::to_string), + visible_terminal_ids: visible.iter().map(|s| s.to_string()).collect(), + claimed_terminal_ids: claimed.map(|ids| ids.iter().map(|s| s.to_string()).collect()), + } + } + + #[test] + fn claims_require_the_connection_negotiation() { + // A connection that did NOT negotiate terminalLifetimeClaimV1 has its + // claimedTerminalIds ignored server-side: the snapshot still applies + // for delivery priority, but the claim diff stays empty (no claim is + // recorded, so nothing can ever be withdrawn either). + let mut state = InterestState::default(); + state.enable(); + assert_eq!( + state.apply(&claimed_snapshot(1, None, &[], Some(&["T"]))), + Ok(Some(InterestClaimChange { + added: vec![], + removed: vec![] + })) + ); + assert!(state.claimed.is_empty()); + } + + #[test] + fn claim_snapshot_supersedes_previous_claim_set_per_connection() { + let mut state = InterestState::default(); + state.enable(); + state.enable_claims(); + // First snapshot claims T and U. + assert_eq!( + state.apply(&claimed_snapshot(1, None, &[], Some(&["T", "U"]))), + Ok(Some(InterestClaimChange { + added: vec!["T".to_string(), "U".to_string()], + removed: vec![] + })) + ); + // A later snapshot claiming only U withdraws T (latest wins). + assert_eq!( + state.apply(&claimed_snapshot(2, None, &[], Some(&["U"]))), + Ok(Some(InterestClaimChange { + added: vec![], + removed: vec!["T".to_string()] + })) + ); + // An empty claim set withdraws everything. + assert_eq!( + state.apply(&claimed_snapshot(3, None, &[], Some(&[]))), + Ok(Some(InterestClaimChange { + added: vec![], + removed: vec!["U".to_string()] + })) + ); + // A snapshot with NO claim field carries no claim information: the + // previous (now-empty) set stands and the diff is empty. + assert_eq!( + state.apply(&claimed_snapshot(4, None, &[], None)), + Ok(Some(InterestClaimChange { + added: vec![], + removed: vec![] + })) + ); + } + + #[test] + fn stale_snapshots_do_not_apply_claims() { + let mut state = InterestState::default(); + state.enable(); + state.enable_claims(); + assert_eq!( + state.apply(&claimed_snapshot(5, None, &[], Some(&["T"]))), + Ok(Some(InterestClaimChange { + added: vec!["T".to_string()], + removed: vec![] + })) + ); + // A STALE revision (5 <= 5) claiming U must be ignored entirely. + assert_eq!( + state.apply(&claimed_snapshot(5, None, &[], Some(&["U"]))), + Ok(None) + ); + // The accepted claim set still says T only. + assert_eq!( + state.apply(&claimed_snapshot(6, None, &[], Some(&["T"]))), + Ok(Some(InterestClaimChange { + added: vec![], + removed: vec![] + })) + ); + } + + #[test] + fn invalid_claim_snapshots_are_rejected_without_replacing_previous_state() { + let mut state = InterestState::default(); + state.enable(); + state.enable_claims(); + state + .apply(&claimed_snapshot(1, None, &[], Some(&["T"]))) + .unwrap(); + // Empty id and an oversized claim list are both invalid. + assert!(state + .apply(&claimed_snapshot(2, None, &[], Some(&[""]))) + .is_err()); + let too_many_valid_ids: Vec = (0..=MAX_INTEREST_TERMINALS) + .map(|i| format!("t-{i}")) + .collect(); + let too_many_refs: Vec<&str> = too_many_valid_ids.iter().map(String::as_str).collect(); + assert!(state + .apply(&claimed_snapshot(3, None, &[], Some(&too_many_refs))) + .is_err()); + // Rejected snapshots did not disturb the accepted claim set. + assert_eq!( + state.apply(&claimed_snapshot(4, None, &[], Some(&["T"]))), + Ok(Some(InterestClaimChange { + added: vec![], + removed: vec![] + })) + ); + } } diff --git a/crates/freshell-ws/tests/codex_fork_rebind.rs b/crates/freshell-ws/tests/codex_fork_rebind.rs index a2f740bb9..178e9380f 100644 --- a/crates/freshell-ws/tests/codex_fork_rebind.rs +++ b/crates/freshell-ws/tests/codex_fork_rebind.rs @@ -347,6 +347,7 @@ fn session_meta_line(thread_id: &str, cwd: &str, forked_from: Option<&str>) -> S } #[cfg(unix)] +#[ignore = "load-flaky: the 10 s wall-clock pre-fork baseline frame-wait starves when full gates run concurrently (2026-09-23 evidence; green focused and in all prior gates); user-directed bypass — kata n115 owns the progress-based rework, un-ignore when it lands"] #[tokio::test(flavor = "multi_thread")] async fn in_tui_fork_rebinds_the_pane_identity() { const OLD: &str = "019fa60f-aaaa-4bbb-8ccc-000000000001"; diff --git a/crates/freshell-ws/tests/hello_capabilities.rs b/crates/freshell-ws/tests/hello_capabilities.rs new file mode 100644 index 000000000..723537934 --- /dev/null +++ b/crates/freshell-ws/tests/hello_capabilities.rs @@ -0,0 +1,496 @@ +//! End-to-end capability-negotiation tests for the `/ws` hello→ready handshake +//! (responsive-terminal-restore Workstream 1: `pacedTerminalReplayV1`). +//! +//! These run a REAL axum server on an ephemeral loopback port (never a fixed/ +//! reserved one) and a REAL tokio-tungstenite WS client, so they exercise the +//! actual `handle_socket` path: the raw-JSON `hello` capability extraction and +//! the `ready` advertisement gate — the layers an in-file +//! `build_handshake_with_capabilities` unit test cannot reach (a typo'd wire +//! key in the extraction compiles fine and silently disables negotiation). +//! +//! Backwards-compat contract under test: a hello WITHOUT the capability must +//! produce a `ready` byte-identical to today's output (no `pacedTerminalReplayV1` +//! key, no new capabilities object), on both sides of the change. + +use std::sync::Arc; +use std::time::Duration; + +use futures_util::{SinkExt, StreamExt}; +use tokio::net::TcpListener; +use tokio_tungstenite::tungstenite::Message as WsMessage; + +use freshell_ws::WsState; + +const AUTH_TOKEN: &str = "s3cr3t-token-abcdef"; + +fn test_settings_value() -> serde_json::Value { + serde_json::json!({ + "ai": {}, + "codingCli": { "enabledProviders": [], "mcpServer": true, "providers": {} }, + "editor": { "externalEditor": "auto" }, + "extensions": { "disabled": [] }, + "freshAgent": { "defaultPlugins": [], "enabled": false, "providers": {} }, + "logging": { "debug": false }, + "network": { "configured": true, "host": "127.0.0.1" }, + "panes": { "defaultNewPane": "ask" }, + "safety": { "autoKillIdleMinutes": 15 }, + "sidebar": { + "autoGenerateTitles": true, + "excludeFirstChatMustStart": false, + "excludeFirstChatSubstrings": [] + }, + "terminal": { "scrollback": 10000 } + }) +} + +/// Build a `WsState`, spin up a real axum server on an ephemeral loopback port +/// (`127.0.0.1:0`, never a fixed/reserved port), and return its `ws://` URL. +async fn spawn_server() -> String { + let auth_token = Arc::new(AUTH_TOKEN.to_string()); + let broadcast_tx = Arc::new(tokio::sync::broadcast::channel::(16).0); + let settings = + Arc::new(serde_json::from_value(test_settings_value()).expect("valid settings fixture")); + + let state = WsState { + pane_ledger: std::sync::Arc::new(freshell_ws::pane_ledger::PaneLedger::disabled()), + layout: Default::default(), + identity: freshell_ws::identity::TerminalIdentityRegistry::new(), + terminal_meta: Default::default(), + auth_token: Arc::clone(&auth_token), + server_instance_id: Arc::new("srv-test".to_string()), + boot_id: Arc::new("boot-test".to_string()), + settings, + handshake_settings: Arc::new(tokio::sync::RwLock::new( + serde_json::from_value(test_settings_value()).expect("valid settings fixture"), + )), + broadcast_tx: Arc::clone(&broadcast_tx), + auto_resume_tx: tokio::sync::mpsc::unbounded_channel().0, + auto_resume_cancels: Default::default(), + fresh_codex: freshell_freshagent::FreshCodexState::new( + Arc::clone(&auth_token), + Arc::clone(&broadcast_tx), + serde_json::json!({ "freshAgent": { "enabled": false } }), + ), + fresh_claude: freshell_freshagent::FreshClaudeState::new(Arc::clone(&broadcast_tx)), + fresh_opencode: freshell_freshagent::FreshOpencodeState::new( + freshell_freshagent::FreshAgentState::new( + Arc::clone(&auth_token), + Arc::clone(&broadcast_tx), + ), + ), + registry: freshell_terminal::TerminalRegistry::new(), + tabs: freshell_ws::tabs::TabsRegistry::new(), + screenshots: freshell_ws::screenshot::ScreenshotBroker::new(Arc::clone(&broadcast_tx)), + subagent_interest: Default::default(), + host_stats: Default::default(), + terminals_revision: Arc::new(std::sync::atomic::AtomicI64::new(0)), + sessions_revision: Arc::new(std::sync::atomic::AtomicI64::new(0)), + cli_commands: Arc::new(Vec::new()), + shutdown: Arc::new(tokio::sync::Notify::new()), + ping_interval_ms: 30_000, + hello_timeout_ms: 5_000, + allowed_origins: Arc::new(freshell_ws::origin::default_allowed_origins()), + ws_max_payload_bytes: 16 * 1024 * 1024, + term09: freshell_ws::backpressure::Term09Config::default(), + create_protect: freshell_ws::create_limit::CreateProtectConfig::default(), + spawn_gate: std::sync::Arc::new(freshell_ws::spawn_gate::SpawnGate::new(4, 64)), + shutdown_started: std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)), + create_dedupe: std::sync::Arc::new(freshell_ws::create_dedupe::CreateDedupe::default()), + config_fallback: None, + opencode_locator: None, + codex_locator: None, + activity: None, + session_existence: std::sync::Arc::new(freshell_ws::existence::NoIndexProbe::default()), + reconcile_deferral_budget_ms: freshell_ws::reconcile::RECONCILE_DEFERRAL_BUDGET_MS_DEFAULT, + fresh_agent_respawn_counts: Default::default(), + ownership: None, + }; + + let router = freshell_ws::router(state); + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("bind ephemeral loopback port"); + let addr = listener.local_addr().expect("local addr"); + tokio::spawn(async move { + let _ = axum::serve(listener, router).await; + }); + + format!("ws://{addr}/ws", addr = addr) +} + +type WsClient = + tokio_tungstenite::WebSocketStream>; + +/// Send a `hello` and read the `ready` frame back as its RAW serialized text +/// (no JSON re-parse): the byte-identity pins below compare literal wire +/// bytes — serde `Value` equality would silently accept reordered keys, +/// reformatted numbers, or spacing changes that old clients may still pin. +async fn hello_ready_text(ws: &mut WsClient, capabilities: serde_json::Value) -> String { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "hello", + "token": AUTH_TOKEN, + "protocolVersion": freshell_protocol::WS_PROTOCOL_VERSION, + "capabilities": capabilities, + }) + .to_string(), + )) + .await + .expect("send hello"); + + let msg = tokio::time::timeout(Duration::from_secs(5), ws.next()) + .await + .expect("ready within timeout") + .expect("stream not ended") + .expect("no ws error"); + let WsMessage::Text(text) = msg else { + panic!("expected the ready text frame, got {msg:?}"); + }; + text +} + +/// Send a `hello` and read the `ready` frame back as JSON (the first message +/// of the connect handshake). +async fn hello_ready(ws: &mut WsClient, capabilities: serde_json::Value) -> serde_json::Value { + let text = hello_ready_text(ws, capabilities).await; + serde_json::from_str(&text).expect("ready is JSON") +} + +/// A hello WITH `pacedTerminalReplayV1: true` gets the echo advertised back in +/// `ready.capabilities` — the full negotiation rail (raw-JSON extraction → +/// handshake builder gate) at the real socket. +#[tokio::test] +async fn negotiated_hello_gets_paced_terminal_replay_echo_in_ready() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let ready = hello_ready( + &mut ws, + serde_json::json!({ "terminalOutputBatchV1": true, "pacedTerminalReplayV1": true }), + ) + .await; + assert_eq!( + ready["capabilities"], + serde_json::json!({ "pacedTerminalReplayV1": true }), + "a negotiated hello must get exactly the paced-replay echo: {ready}" + ); +} + +/// A hello that negotiates other capabilities but NOT the paced one gets a +/// `ready.capabilities` object byte-identical to today's output — the new key +/// never leaks to a non-opting client. +#[tokio::test] +async fn non_paced_negotiation_keeps_capabilities_byte_identical() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let ready = hello_ready( + &mut ws, + serde_json::json!({ "paneReconcileV1": true, "terminalInterestV1": true }), + ) + .await; + assert_eq!( + ready["capabilities"], + serde_json::json!({ "paneReconcileV1": true, "terminalInterestV1": true }), + "a non-paced negotiation must keep today's capabilities shape: {ready}" + ); +} + +/// The frozen client (no capabilities at all) still gets a `ready` with NO +/// capabilities object — byte-identical to the pre-capability handshake. +#[tokio::test] +async fn capability_free_hello_ready_has_no_capabilities_object() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let ready = hello_ready(&mut ws, serde_json::json!({})).await; + assert!( + ready.get("capabilities").is_none(), + "a capability-free hello must not change ready's shape: {ready}" + ); +} + +/// A hello WITH `terminalLifetimeClaimV1: true` gets the echo advertised back +/// in `ready.capabilities` (responsive-terminal-restore Workstream 1: the +/// hidden-pane lifetime-claim negotiation rail) — raw-JSON extraction → +/// handshake builder gate at the real socket. +#[tokio::test] +async fn negotiated_hello_gets_terminal_lifetime_claim_echo_in_ready() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let ready = hello_ready( + &mut ws, + serde_json::json!({ "terminalInterestV1": true, "terminalLifetimeClaimV1": true }), + ) + .await; + assert_eq!( + ready["capabilities"], + serde_json::json!({ "terminalInterestV1": true, "terminalLifetimeClaimV1": true }), + "a negotiated hello must get exactly the lifetime-claim echo: {ready}" + ); +} + +/// A hello that negotiates other capabilities but NOT the lifetime-claim one +/// gets a `ready.capabilities` object byte-identical to today's output — the +/// new key never leaks to a non-opting client. +#[tokio::test] +async fn non_claim_negotiation_keeps_capabilities_byte_identical() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let ready = hello_ready( + &mut ws, + serde_json::json!({ "paneReconcileV1": true, "terminalInterestV1": true, "pacedTerminalReplayV1": true }), + ) + .await; + assert_eq!( + ready["capabilities"], + serde_json::json!({ "paneReconcileV1": true, "terminalInterestV1": true, "pacedTerminalReplayV1": true }), + "a non-claim negotiation must keep today's capabilities shape: {ready}" + ); +} + +/// Read the next JSON text frame from the socket (bounded). +async fn next_json(ws: &mut WsClient) -> serde_json::Value { + let msg = tokio::time::timeout(Duration::from_secs(5), ws.next()) + .await + .expect("frame within timeout") + .expect("stream not ended") + .expect("no ws error"); + let WsMessage::Text(text) = msg else { + panic!("expected a text frame, got {msg:?}"); + }; + serde_json::from_str(&text).expect("frame is JSON") +} + +/// Send `terminal.create` (shell) and return the created terminalId. +async fn create_shell_terminal(ws: &mut WsClient, request_id: &str) -> String { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.create", + "requestId": request_id, + "mode": "shell", + "shell": "system", + }) + .to_string(), + )) + .await + .expect("send terminal.create"); + + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + let value = next_json(ws).await; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.created") + && value.get("requestId").and_then(|v| v.as_str()) == Some(request_id) + { + return value + .get("terminalId") + .and_then(|v| v.as_str()) + .expect("terminal.created carries terminalId") + .to_string(); + } + } + panic!("terminal.created never arrived"); +} + +/// Send `terminal.attach` and return the matching `terminal.attach.ready` +/// frame's JSON (skipping any frames in between). +async fn attach_and_read_ready( + ws: &mut WsClient, + terminal_id: &str, + attach_request_id: &str, +) -> serde_json::Value { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.attach", + "terminalId": terminal_id, + "intent": "viewport_hydrate", + "cols": 80, + "rows": 24, + "attachRequestId": attach_request_id, + }) + .to_string(), + )) + .await + .expect("send terminal.attach"); + + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + let value = next_json(ws).await; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.attach.ready") + && value.get("attachRequestId").and_then(|v| v.as_str()) == Some(attach_request_id) + { + return value; + } + } + panic!("terminal.attach.ready never arrived"); +} + +/// Negotiation gating e2e (restore contract): a `pacedTerminalReplayV1` +/// hello attaching over the REAL socket must see `oldestRetainedSeq` on +/// `terminal.attach.ready` — the full rail (raw hello JSON extraction → +/// `handle_attach` → registry population) that in-file unit tests cannot +/// reach. A fresh terminal reports 1 either way the ring sits (empty ring: +/// headSeq 0 + 1; retained first frame: seqStart 1). +#[tokio::test] +async fn negotiated_attach_ready_carries_oldest_retained_seq() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + let _ready = hello_ready( + &mut ws, + serde_json::json!({ "pacedTerminalReplayV1": true }), + ) + .await; + + let terminal_id = create_shell_terminal(&mut ws, "create-paced").await; + let ready = attach_and_read_ready(&mut ws, &terminal_id, "attach-paced").await; + assert_eq!( + ready["oldestRetainedSeq"], 1, + "a negotiated attach's ready frame must carry the retention bound: {ready}" + ); + assert!( + ready["headSeq"].as_i64().is_some_and(|h| h >= 0), + "headSeq remains the honest current head: {ready}" + ); +} + +/// The compatibility twin: a hello WITHOUT the capability must see a ready +/// frame with NO `oldestRetainedSeq` key at all — byte-identical to the +/// pre-contract shape on the real socket. +#[tokio::test] +async fn plain_attach_ready_omits_oldest_retained_seq() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + let _ready = hello_ready(&mut ws, serde_json::json!({})).await; + + let terminal_id = create_shell_terminal(&mut ws, "create-plain").await; + let ready = attach_and_read_ready(&mut ws, &terminal_id, "attach-plain").await; + assert!( + ready.get("oldestRetainedSeq").is_none(), + "a non-negotiated ready frame must not gain any new key: {ready}" + ); +} + +// ── Mixed-version compatibility matrix (responsive-terminal-restore, +// task-008) ───────────────────────────────────────────────────────────── + +/// Old client → new server, READY-FRAME BYTE IDENTITY (carried task-1 review +/// Nit 1): a capability-free hello must produce a `ready` whose RAW wire +/// text is byte-identical to the pre-branch shape — pinned as literal +/// serialized text, not serde `Value` equality (Value equality would silently +/// accept reordered keys, `1`-vs-`1.0`, or spacing changes). Only the +/// timestamp and buildId values are dynamic; every static byte around them, +/// the key order, and the absence of every newer key are pinned exactly. +#[tokio::test] +async fn capability_free_ready_frame_raw_text_is_byte_identical_to_the_pre_branch_shape() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let text = hello_ready_text(&mut ws, serde_json::json!({})).await; + + // The tag leads, then the fields in the frozen struct order: timestamp, + // bootId, serverInstanceId, buildId — and NOTHING else (capabilities and + // runtimeOwners are omitted for this fixture: no capability keys, no + // ownership coordinator injected). + assert!( + text.starts_with(r#"{"type":"ready","timestamp":""#), + "the ready frame's literal prefix must stay byte-identical: {text}" + ); + assert!( + text.contains(r#""bootId":"boot-test","serverInstanceId":"srv-test","buildId":""#), + "the static fields must serialize in the frozen order and shape: {text}" + ); + assert!( + text.ends_with(r#""}"#), + "buildId is the last field — no key may follow it: {text}" + ); + for forbidden in [ + "capabilities", + "pacedTerminalReplayV1", + "terminalLifetimeClaimV1", + "runtimeOwners", + "oldestRetainedSeq", + ] { + assert!( + !text.contains(forbidden), + "an old client's ready frame must not contain the new key {forbidden}: {text}" + ); + } +} + +/// Old client → new server, sibling-negotiation BYTE IDENTITY: a hello that +/// negotiates pre-branch capabilities but none of the new ones must get a +/// `ready.capabilities` object whose RAW serialized text is the literal +/// pre-branch bytes — exact key order, no spacing, no new keys interleaved +/// (the Value-equality twin above cannot catch a reorder or a reformat). +#[tokio::test] +async fn sibling_negotiation_ready_capabilities_raw_text_is_byte_identical() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let text = hello_ready_text( + &mut ws, + serde_json::json!({ "paneReconcileV1": true, "terminalInterestV1": true }), + ) + .await; + + assert!( + text.contains(r#""capabilities":{"paneReconcileV1":true,"terminalInterestV1":true}"#), + "the echoed capabilities object must be the literal pre-branch bytes \ + (key order, separators, no new keys): {text}" + ); + for forbidden in ["pacedTerminalReplayV1", "terminalLifetimeClaimV1"] { + assert!( + !text.contains(forbidden), + "a non-opting client's ready must not mention {forbidden}: {text}" + ); + } +} + +/// Combined negotiation (carried task-1 review Nit 2): a hello carrying +/// `pacedTerminalReplayV1` AND a sibling capability produces a ready that +/// carries BOTH keys — the paced echo never drops or swallows a sibling, and +/// the raw text pins the combined literal shape. +#[tokio::test] +async fn combined_negotiation_ready_carries_both_the_paced_and_sibling_keys() { + let url = spawn_server().await; + let (mut ws, _resp) = tokio_tungstenite::connect_async(&url) + .await + .expect("ws connect"); + + let text = hello_ready_text( + &mut ws, + serde_json::json!({ "pacedTerminalReplayV1": true, "paneReconcileV1": true }), + ) + .await; + let ready: serde_json::Value = serde_json::from_str(&text).expect("ready is JSON"); + + assert_eq!( + ready["capabilities"], + serde_json::json!({ "paneReconcileV1": true, "pacedTerminalReplayV1": true }), + "a combined negotiation must echo BOTH keys: {ready}" + ); + assert!( + text.contains(r#""capabilities":{"paneReconcileV1":true,"pacedTerminalReplayV1":true}"#), + "the combined echo must keep the frozen literal byte order: {text}" + ); +} diff --git a/crates/freshell-ws/tests/paced_replay.rs b/crates/freshell-ws/tests/paced_replay.rs new file mode 100644 index 000000000..e04aba653 --- /dev/null +++ b/crates/freshell-ws/tests/paced_replay.rs @@ -0,0 +1,4483 @@ +//! Responsive-terminal-restore Workstream 1 — the paced replay core, +//! end-to-end on REAL sockets: a REAL axum server (ephemeral loopback +//! port), a REAL PTY (`TerminalRegistry`), and REAL tokio-tungstenite WS +//! clients, one of which negotiates `pacedTerminalReplayV1` and drives +//! continuation credits over the raw socket. +//! +//! The harness follows `hello_capabilities.rs` (the `connect_async` +//! `WsClient` type — the two files' WS client types are deliberately NOT +//! shared). The page budget and the scrollback ring are per-server registry +//! knobs, so fixtures are deterministic without env races. +//! +//! Core contract under test: a negotiated attach gets `attach.ready` bounds +//! plus ONE bounded first page; further replay pages flow only on valid +//! `terminal.replay.credit` (stale generations, out-of-window values, and +//! credits from non-negotiated connections are ignored); the grant is +//! strictly page-end (a partial consumption report is observed, grants +//! nothing); live output produced during the replay is delivered strictly +//! AFTER the pages covering it, in seq order; the tail drain completes +//! against a continuously-producing terminal (fixed tail target + the +//! one-shot live handoff — never a chase of the moving head); retention +//! expiry mid-replay reports the exact lost interval with the task-2 +//! bounds fields and continues from the new baseline; a re-attach +//! supersedes the old session; a disconnect mid-replay leaves the +//! terminal running and re-attachable. + +use std::collections::BTreeMap; +use std::sync::Arc; +use std::time::Duration; + +use futures_util::{SinkExt, StreamExt}; +use tokio::net::TcpListener; +use tokio_tungstenite::tungstenite::Message as WsMessage; + +use freshell_ws::WsState; + +// ── capturing tracing layer (dev-only test facility). PROCESS-GLOBAL by +// deliberate choice (the diag01_lifecycle_events.rs `global_capture` / +// invariants.rs e08g pattern, extended with `record_u64` so the paced +// events' u64 fields — `page_bytes`, `pages` — are captured too): a +// thread-local `set_default` capture is UNSOUND for callsites shared with +// sibling tests running in parallel — tracing-core caches each callsite's +// Interest process-wide on first registration, and a subscriber-less +// sibling thread executing a shared emission site first (e.g. +// `credit_on_a_non_negotiated_connection_is_inert` firing the +// `non_negotiated` callsite) caches `Interest::never`, so a thread-local +// capture then never sees its OWN thread's emissions (kata 59nb). One +// global subscriber sees every thread's events; every read below MUST +// filter by the per-test-unique `terminal_id` because ALL tests in this +// binary share the vec. ───────────────────────────────────────────────── + +use std::sync::Mutex; +use tracing::field::{Field, Visit}; +use tracing::{Event, Subscriber}; +use tracing_subscriber::layer::{Context, SubscriberExt}; +use tracing_subscriber::Layer; + +#[derive(Debug, Clone, Default)] +struct CapturedEvent { + message: String, + fields: BTreeMap, +} + +#[derive(Default)] +struct FieldVisitor { + message: String, + fields: BTreeMap, +} + +impl Visit for FieldVisitor { + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + let rendered = format!("{value:?}"); + if field.name() == "message" { + self.message = rendered; + } else { + self.fields.insert(field.name().to_string(), rendered); + } + } + + fn record_str(&mut self, field: &Field, value: &str) { + if field.name() == "message" { + self.message = value.to_string(); + } else { + self.fields + .insert(field.name().to_string(), value.to_string()); + } + } + + fn record_i64(&mut self, field: &Field, value: i64) { + self.fields + .insert(field.name().to_string(), value.to_string()); + } + + fn record_u64(&mut self, field: &Field, value: u64) { + self.fields + .insert(field.name().to_string(), value.to_string()); + } +} + +struct CaptureLayer { + events: Arc>>, +} + +impl Layer for CaptureLayer { + fn on_event(&self, event: &Event<'_>, _ctx: Context<'_, S>) { + let mut visitor = FieldVisitor::default(); + event.record(&mut visitor); + self.events + .lock() + .expect("capture lock") + .push(CapturedEvent { + message: visitor.message, + fields: visitor.fields, + }); + } +} + +/// Process-global capture for this test binary (first caller installs; +/// `get_or_init` is the synchronization). This binary installs no other +/// global subscriber; `.expect()` turns any future second installer into an +/// immediate diagnosable panic instead of a silently-empty capture. +fn global_capture() -> Arc>> { + static EVENTS: std::sync::OnceLock>>> = std::sync::OnceLock::new(); + Arc::clone(EVENTS.get_or_init(|| { + let events = Arc::new(Mutex::new(Vec::new())); + let layer = CaptureLayer { + events: Arc::clone(&events), + }; + // Target/level-filter the process-global registry (task-010b + // hygiene): an unfiltered registry enables every callsite + // process-wide — a latent perf and determinism footgun. Unlike the + // term09 twin (a plain WARN level filter), this binary's + // assertions read INFO-level `ws.restore.*` events, emitted by + // exactly two modules: `freshell_ws::paced_replay` (paced_start / + // paced_complete / paced_expired; paced_gone is warn) and + // `freshell_ws::terminal` (the credit verdicts and + // restore-unavailable). The filter admits exactly those targets at + // INFO plus WARN-or-above everywhere else — nothing the tests read + // is disabled. + let subscriber = tracing_subscriber::registry().with(layer).with( + tracing_subscriber::filter::Targets::new() + .with_default(tracing_subscriber::filter::LevelFilter::WARN) + .with_target( + "freshell_ws::paced_replay", + tracing_subscriber::filter::LevelFilter::INFO, + ) + .with_target( + "freshell_ws::terminal", + tracing_subscriber::filter::LevelFilter::INFO, + ), + ); + tracing::subscriber::set_global_default(subscriber) + .expect("this test binary installs exactly one global subscriber"); + events + })) +} + +/// Poll the capture until an event for THIS test's terminal (the vec is +/// shared by every test in the binary — the unique `terminal_id` is the +/// per-test discriminator) with this message AND `fields[field] == value` +/// lands (or the 5s deadline passes). The `ws.restore.credit` events share +/// one message name, so the verdict `status` field is the selector. +async fn wait_for_restore_event( + events: &Arc>>, + terminal_id: &str, + message: &str, + field: &str, + value: &str, +) -> Option { + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + loop { + { + let captured = events.lock().unwrap(); + if let Some(found) = captured.iter().find(|e| { + e.message == message + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id) + && e.fields.get(field).map(String::as_str) == Some(value) + }) { + return Some(found.clone()); + } + } + if tokio::time::Instant::now() >= deadline { + return None; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } +} + +/// [`wait_for_restore_event`] without the extra field selector — for the +/// one-event-per-terminal messages (`ws.restore.paced_start`, +/// `ws.restore.paced_complete`). +async fn wait_for_restore_event_of_terminal( + events: &Arc>>, + terminal_id: &str, + message: &str, +) -> Option { + let deadline = tokio::time::Instant::now() + Duration::from_secs(5); + loop { + { + let captured = events.lock().unwrap(); + if let Some(found) = captured.iter().find(|e| { + e.message == message + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id) + }) { + return Some(found.clone()); + } + } + if tokio::time::Instant::now() >= deadline { + return None; + } + tokio::time::sleep(Duration::from_millis(20)).await; + } +} + +const AUTH_TOKEN: &str = "s3cr3t-token-abcdef"; +/// Deliberately small page budget: deterministic multi-page fixtures with +/// small floods (the production default is 128 KiB). +const PAGE_BUDGET: i64 = 4096; + +fn test_settings_value() -> serde_json::Value { + serde_json::json!({ + "ai": {}, + "codingCli": { "enabledProviders": [], "mcpServer": true, "providers": {} }, + "editor": { "externalEditor": "auto" }, + "extensions": { "disabled": [] }, + "freshAgent": { "defaultPlugins": [], "enabled": false, "providers": {} }, + "logging": { "debug": false }, + "network": { "configured": true, "host": "127.0.0.1" }, + "panes": { "defaultNewPane": "ask" }, + "safety": { "autoKillIdleMinutes": 15 }, + "sidebar": { + "autoGenerateTitles": true, + "excludeFirstChatMustStart": false, + "excludeFirstChatSubstrings": [] + }, + "terminal": { "scrollback": 10000 } + }) +} + +/// Spawn a real server with a paced-page budget of [`PAGE_BUDGET`] and a +/// scrollback ring of `ring_chars` UTF-16 units (per-server registry knobs — +/// no env races between parallel tests). +async fn spawn_server(ring_chars: i64) -> String { + spawn_server_with(ring_chars, PAGE_BUDGET, None).await +} + +/// [`spawn_server`] with per-test registry/backpressure overrides: the +/// paced page budget (default [`PAGE_BUDGET`]) and an optional +/// `queue_max_bytes` TERM-09 output-queue cap (None keeps +/// [`freshell_ws::backpressure::Term09Config::default`] — the production +/// 16 MiB lane; Some makes the drain's connection backpressure observable +/// at test scale). +async fn spawn_server_with( + ring_chars: i64, + page_budget: i64, + queue_max_bytes: Option, +) -> String { + let auth_token = Arc::new(AUTH_TOKEN.to_string()); + let broadcast_tx = Arc::new(tokio::sync::broadcast::channel::(16).0); + let settings = + Arc::new(serde_json::from_value(test_settings_value()).expect("valid settings fixture")); + + let registry = freshell_terminal::TerminalRegistry::new(); + // Round-5 finding 1 (degenerate settings): mirror the server boot's + // page-budget clamp — the boot applies the queue-derived admission + // ceiling right after TERM-09 resolution, so a paced page always fits + // the connection queue's drain-admission watermark. The harness + // applies the SAME relationship so every paced fixture runs the + // production shape (with default knobs this is a no-op). + let effective_page_budget = + page_budget.min(freshell_ws::backpressure::paced_page_budget_ceiling( + queue_max_bytes.unwrap_or_else(|| { + freshell_ws::backpressure::Term09Config::default().queue_max_bytes + }), + )); + registry.set_paced_page_max_bytes(effective_page_budget); + registry.set_scrollback_max_bytes(ring_chars); + + let state = WsState { + pane_ledger: std::sync::Arc::new(freshell_ws::pane_ledger::PaneLedger::disabled()), + layout: Default::default(), + identity: freshell_ws::identity::TerminalIdentityRegistry::new(), + terminal_meta: Default::default(), + auth_token: Arc::clone(&auth_token), + server_instance_id: Arc::new("srv-test".to_string()), + boot_id: Arc::new("boot-test".to_string()), + settings, + handshake_settings: Arc::new(tokio::sync::RwLock::new( + serde_json::from_value(test_settings_value()).expect("valid settings fixture"), + )), + broadcast_tx: Arc::clone(&broadcast_tx), + auto_resume_tx: tokio::sync::mpsc::unbounded_channel().0, + auto_resume_cancels: Default::default(), + fresh_codex: freshell_freshagent::FreshCodexState::new( + Arc::clone(&auth_token), + Arc::clone(&broadcast_tx), + serde_json::json!({ "freshAgent": { "enabled": false } }), + ), + fresh_claude: freshell_freshagent::FreshClaudeState::new(Arc::clone(&broadcast_tx)), + fresh_opencode: freshell_freshagent::FreshOpencodeState::new( + freshell_freshagent::FreshAgentState::new( + Arc::clone(&auth_token), + Arc::clone(&broadcast_tx), + ), + ), + registry, + tabs: freshell_ws::tabs::TabsRegistry::new(), + screenshots: freshell_ws::screenshot::ScreenshotBroker::new(Arc::clone(&broadcast_tx)), + subagent_interest: Default::default(), + host_stats: Default::default(), + terminals_revision: Arc::new(std::sync::atomic::AtomicI64::new(0)), + sessions_revision: Arc::new(std::sync::atomic::AtomicI64::new(0)), + cli_commands: Arc::new(Vec::new()), + shutdown: Arc::new(tokio::sync::Notify::new()), + ping_interval_ms: 30_000, + hello_timeout_ms: 5_000, + allowed_origins: Arc::new(freshell_ws::origin::default_allowed_origins()), + ws_max_payload_bytes: 64 * 1024 * 1024, + term09: freshell_ws::backpressure::Term09Config { + queue_max_bytes: queue_max_bytes.unwrap_or_else(|| { + freshell_ws::backpressure::Term09Config::default().queue_max_bytes + }), + ..freshell_ws::backpressure::Term09Config::default() + }, + create_protect: freshell_ws::create_limit::CreateProtectConfig::default(), + spawn_gate: std::sync::Arc::new(freshell_ws::spawn_gate::SpawnGate::new(4, 64)), + shutdown_started: std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)), + create_dedupe: std::sync::Arc::new(freshell_ws::create_dedupe::CreateDedupe::default()), + config_fallback: None, + opencode_locator: None, + codex_locator: None, + activity: None, + session_existence: std::sync::Arc::new(freshell_ws::existence::NoIndexProbe::default()), + reconcile_deferral_budget_ms: freshell_ws::reconcile::RECONCILE_DEFERRAL_BUDGET_MS_DEFAULT, + fresh_agent_respawn_counts: Default::default(), + ownership: None, + }; + + let router = freshell_ws::router(state); + let listener = TcpListener::bind("127.0.0.1:0") + .await + .expect("bind ephemeral loopback port"); + let addr = listener.local_addr().expect("local addr"); + tokio::spawn(async move { + let _ = axum::serve(listener, router).await; + }); + + format!("ws://{addr}/ws", addr = addr) +} + +type WsClient = + tokio_tungstenite::WebSocketStream>; + +async fn connect(url: &str) -> WsClient { + let (ws, _resp) = tokio_tungstenite::connect_async(url) + .await + .expect("ws connect"); + ws +} + +/// Complete the hello handshake, optionally negotiating the paced capability +/// (and echo-reading the 4 handshake frames). +async fn hello(ws: &mut WsClient, paced: bool) { + let capabilities = if paced { + serde_json::json!({ "pacedTerminalReplayV1": true }) + } else { + serde_json::json!({}) + }; + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "hello", + "token": AUTH_TOKEN, + "protocolVersion": freshell_protocol::WS_PROTOCOL_VERSION, + "capabilities": capabilities, + }) + .to_string(), + )) + .await + .expect("send hello"); + for _ in 0..4u8 { + let msg = tokio::time::timeout(Duration::from_secs(5), ws.next()) + .await + .expect("handshake frame within timeout") + .expect("stream not ended") + .expect("no ws error"); + assert!(matches!(msg, WsMessage::Text(_))); + } +} + +async fn next_json(ws: &mut WsClient) -> serde_json::Value { + let msg = tokio::time::timeout(Duration::from_secs(10), ws.next()) + .await + .expect("frame within timeout") + .expect("stream not ended") + .expect("no ws error"); + let WsMessage::Text(text) = msg else { + panic!("expected a text frame, got {msg:?}"); + }; + serde_json::from_str(&text).expect("frame is JSON") +} + +/// Read the next JSON frame, or `None` if no frame at all arrives within +/// `window`. A keepalive Ping/Pong landing INSIDE the window is NOT +/// silence — the burst-complete heuristic treats only text frames as +/// signal, so a server ping mid-read cannot truncate a page read (the +/// 30s ping interval makes this rare vs the ~1-2s bursts, but the harness +/// must not depend on that timing). +async fn next_json_or_timeout(ws: &mut WsClient, window: Duration) -> Option { + let deadline = tokio::time::Instant::now() + window; + loop { + let remaining = deadline.saturating_duration_since(tokio::time::Instant::now()); + if remaining.is_zero() { + return None; + } + match tokio::time::timeout(remaining, ws.next()).await { + Ok(Some(Ok(WsMessage::Text(text)))) => { + return Some(serde_json::from_str(&text).expect("frame is JSON")); + } + // Control frames never count as the burst's end: keep waiting + // for text until the window actually elapses. + Ok(Some(Ok(WsMessage::Ping(_) | WsMessage::Pong(_)))) => continue, + _ => return None, + } + } +} + +async fn create_shell_terminal(ws: &mut WsClient, request_id: &str) -> String { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.create", + "requestId": request_id, + "mode": "shell", + "shell": "system", + }) + .to_string(), + )) + .await + .expect("send terminal.create"); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + let value = next_json(ws).await; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.created") + && value.get("requestId").and_then(|v| v.as_str()) == Some(request_id) + { + return value + .get("terminalId") + .and_then(|v| v.as_str()) + .expect("terminal.created carries terminalId") + .to_string(); + } + } + panic!("terminal.created never arrived"); +} + +/// Send `terminal.attach` (viewport hydrate, sinceSeq 0 by default). +async fn attach(ws: &mut WsClient, terminal_id: &str, attach_request_id: &str) { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.attach", + "terminalId": terminal_id, + "intent": "viewport_hydrate", + "cols": 80, + "rows": 24, + "attachRequestId": attach_request_id, + "sinceSeq": 0, + }) + .to_string(), + )) + .await + .expect("send terminal.attach"); +} + +/// Send `terminal.attach` WITHOUT an `attachRequestId` — the uncorrelated +/// legacy shape: even a negotiated connection falls back to the inline +/// full-replay path (credits cannot be correlated without a generation key). +async fn attach_without_arid(ws: &mut WsClient, terminal_id: &str) { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.attach", + "terminalId": terminal_id, + "intent": "viewport_hydrate", + "cols": 80, + "rows": 24, + "sinceSeq": 0, + }) + .to_string(), + )) + .await + .expect("send arid-less terminal.attach"); +} + +/// Send one continuation credit. +async fn credit(ws: &mut WsClient, terminal_id: &str, arid: &str, consumed_seq: i64) { + credit_with_stream( + ws, + terminal_id, + &known_stream(terminal_id), + arid, + consumed_seq, + ) + .await; +} + +/// Send `terminal.replay.credit` with an EXPLICIT stream id — the honest +/// client shape (the stream comes from the terminal's `attach.ready`), +/// used by tests that credit without having observed a ready for the +/// terminal (round-2 finding F4 made stream identity part of the +/// continuation contract, so the old fixture value +/// "ignored-by-server" is now correctly rejected as a stream mismatch). +async fn credit_with_stream( + ws: &mut WsClient, + terminal_id: &str, + stream_id: &str, + arid: &str, + consumed_seq: i64, +) { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.replay.credit", + "terminalId": terminal_id, + "streamId": stream_id, + "attachRequestId": arid, + "consumedSeq": consumed_seq, + }) + .to_string(), + )) + .await + .expect("send terminal.replay.credit"); +} + +/// The stream ids observed in `terminal.attach.ready` frames, keyed by +/// terminal id (terminal ids are unique per test, so the map is race-free +/// across parallel tests in this binary). `credit()` sends the terminal's +/// REAL stream id from here — see [`credit_with_stream`]. Poison-tolerant: +/// the map is a test fixture, and one test's panic must not cascade +/// through every other test's lock acquisition. +fn note_ready_stream(ready: &serde_json::Value) { + let terminal_id = ready["terminalId"].as_str().expect("ready.terminalId"); + let stream_id = ready["streamId"].as_str().expect("ready.streamId"); + test_ready_streams() + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .insert(terminal_id.to_string(), stream_id.to_string()); +} + +fn known_stream(terminal_id: &str) -> String { + test_ready_streams() + .lock() + .unwrap_or_else(|poisoned| poisoned.into_inner()) + .get(terminal_id) + .cloned() + .unwrap_or_else(|| { + panic!( + "no attach.ready observed for {terminal_id}: read the ready via a helper \ + (paced_attach_first_page / attach_burst_collecting_gaps) or use \ + credit_with_stream" + ) + }) +} + +fn test_ready_streams() -> &'static std::sync::Mutex> { + static STREAMS: std::sync::OnceLock< + std::sync::Mutex>, + > = std::sync::OnceLock::new(); + STREAMS.get_or_init(|| std::sync::Mutex::new(std::collections::HashMap::new())) +} + +/// A flood whose completion is detectable by a marker that only the EXECUTED +/// printf emits (octal escapes keep the literal command text from echoing +/// the marker early — the same discipline as `term09_output_queue.rs`). +fn flood_command(lines: usize, marker: &str) -> String { + assert_eq!(marker, "FLOOD-DONE-MARKER"); + format!( + "yes 'STREAMDATA-XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX' | head -n {lines}; printf '\\106\\114\\117\\117\\104\\055\\104\\117\\116\\105\\055\\115\\101\\122\\113\\105\\122\\012'\n" + ) +} + +async fn send_input(ws: &mut WsClient, terminal_id: &str, data: &str) { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.input", + "terminalId": terminal_id, + "data": data, + }) + .to_string(), + )) + .await + .expect("send terminal.input"); +} + +/// Drain frames until `marker` appears in the accumulated output data. +/// Returns `(data, frames)` — the concatenated output data and every +/// terminal.output frame's seqStart (ascending as received). +async fn drain_until_marker( + ws: &mut WsClient, + marker: &str, + deadline: tokio::time::Instant, +) -> (String, Vec) { + let mut acc = String::new(); + let mut seqs = Vec::new(); + while tokio::time::Instant::now() < deadline { + let remaining = deadline.saturating_duration_since(tokio::time::Instant::now()); + match tokio::time::timeout(remaining.max(Duration::from_millis(1)), ws.next()).await { + Ok(Some(Ok(WsMessage::Text(text)))) => { + let Ok(value) = serde_json::from_str::(&text) else { + continue; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + if let Some(data) = value.get("data").and_then(|v| v.as_str()) { + acc.push_str(data); + } + seqs.push(value.get("seqStart").and_then(|v| v.as_i64()).unwrap_or(-1)); + } + if acc.contains(marker) { + return (acc, seqs); + } + } + _ => break, + } + } + (acc, seqs) +} + +/// Drive the terminal from a NON-NEGOTIATED driver connection until +/// `marker` is observed on it (the deterministic "the flood finished" +/// signal — the ring holds the full flood before the paced client attaches). +async fn flood_until_complete(url: &str, driver: &mut WsClient, terminal_id: &str, lines: usize) { + let mut observer = connect(url).await; + hello(&mut observer, false).await; + attach(&mut observer, terminal_id, "attach-observer").await; + let marker = "FLOOD-DONE-MARKER"; + send_input(driver, terminal_id, &flood_command(lines, marker)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let (acc, _) = drain_until_marker(&mut observer, marker, deadline).await; + assert!( + acc.contains(marker), + "the flood must complete on the observer" + ); + // Detach the observer so it stops receiving (and the terminal's + // subscriber set is clean for the paced client). + observer + .send(WsMessage::Text( + serde_json::json!({ "type": "terminal.detach", "terminalId": terminal_id }).to_string(), + )) + .await + .expect("observer detaches"); +} + +/// One attach's spontaneous delivery: the ready frame plus every output +/// frame the server sends WITHOUT any credit (the negotiated first page / +/// the non-negotiated full inline replay). Reading ends on the first QUIET +/// gap after the ready — the burst is one back-to-back admission, so a quiet +/// window means the server has nothing more to send unprompted. +async fn paced_attach_first_page( + ws: &mut WsClient, + terminal_id: &str, + arid: &str, +) -> (serde_json::Value, Vec) { + attach(ws, terminal_id, arid).await; + read_attach_burst(ws, arid).await +} + +/// [`paced_attach_first_page`] with a negotiated `replayPageBytes` bound +/// (E2R1 finding 2's sub-cap fixtures request a page budget BELOW the +/// production frame size, so every page the session produces is the page +/// builder's atomic single-frame result). +async fn paced_attach_first_page_with_budget( + ws: &mut WsClient, + terminal_id: &str, + arid: &str, + replay_page_bytes: serde_json::Value, +) -> (serde_json::Value, Vec) { + attach_with_page_budget(ws, terminal_id, arid, replay_page_bytes).await; + read_attach_burst(ws, arid).await +} + +/// The attach burst after the attach frame was sent: the ready (matched +/// by `attachRequestId`) plus every output frame of the first page, read +/// to the first quiet gap after the ready. +async fn read_attach_burst( + ws: &mut WsClient, + arid: &str, +) -> (serde_json::Value, Vec) { + read_attach_burst_with_quiet(ws, arid, Duration::from_millis(400)).await +} + +/// [`read_attach_burst`] with an explicit quiet window (the paced +/// sub-cap fixture's bursts are the only traffic on their connection, so +/// a short window keeps its per-pane attach dance tight). +async fn read_attach_burst_with_quiet( + ws: &mut WsClient, + arid: &str, + quiet: Duration, +) -> (serde_json::Value, Vec) { + let mut ready = None; + let mut outputs = Vec::new(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + match next_json_or_timeout(ws, quiet).await { + None => { + if ready.is_some() { + break; // the unprompted burst is complete + } + continue; + } + Some(value) => match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.attach.ready") + if value.get("attachRequestId").and_then(|v| v.as_str()) == Some(arid) => + { + ready = Some(value); + } + Some("terminal.output") => outputs.push(value), + _ => {} + }, + } + } + let ready = ready.expect("attach.ready never arrived"); + note_ready_stream(&ready); + (ready, outputs) +} + +/// The LEGACY (non-paced) attach burst: the ready frame (matched by +/// terminalId — this attach carries no `attachRequestId` to match) plus +/// every output frame the server sends spontaneously (the full inline +/// replay), read to the first quiet gap after the ready. +async fn legacy_attach_inline_replay( + ws: &mut WsClient, + terminal_id: &str, +) -> (serde_json::Value, Vec) { + let mut ready = None; + let mut outputs = Vec::new(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + match next_json_or_timeout(ws, Duration::from_millis(400)).await { + None => { + if ready.is_some() { + break; // the inline burst is complete + } + continue; + } + Some(value) => match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.attach.ready") + if value.get("terminalId").and_then(|v| v.as_str()) == Some(terminal_id) => + { + ready = Some(value); + } + Some("terminal.output") | Some("terminal.output.batch") => outputs.push(value), + _ => {} + }, + } + } + let ready = ready.expect("attach.ready never arrived"); + note_ready_stream(&ready); + (ready, outputs) +} + +/// One attach's full spontaneous burst, collecting EVERY frame type: the +/// ready (matched by `attachRequestId`), all `terminal.output` frames, and +/// any `terminal.output.gap` frames separately. The gap bucket is the +/// mixed-version matrix pin: a NON-NEGOTIATED attach must never produce a +/// `replay_window_exceeded` gap — today's silent retained-tail behavior is +/// the old-client contract. +async fn attach_burst_collecting_gaps( + ws: &mut WsClient, + terminal_id: &str, + arid: &str, +) -> ( + serde_json::Value, + Vec, + Vec, +) { + attach(ws, terminal_id, arid).await; + let mut ready = None; + let mut outputs = Vec::new(); + let mut gaps = Vec::new(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + match next_json_or_timeout(ws, Duration::from_millis(400)).await { + None => { + if ready.is_some() { + break; // the unprompted burst is complete + } + continue; + } + Some(value) => match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.attach.ready") + if value.get("attachRequestId").and_then(|v| v.as_str()) == Some(arid) => + { + ready = Some(value); + } + Some("terminal.output") => outputs.push(value), + Some("terminal.output.gap") => gaps.push(value), + _ => {} + }, + } + } + let ready = ready.expect("attach.ready never arrived"); + note_ready_stream(&ready); + (ready, outputs, gaps) +} + +/// The covered seq set of a frame list: every seq in [seqStart, seqEnd]. +fn covered_seqs(frames: &[serde_json::Value]) -> std::collections::BTreeSet { + frames + .iter() + .flat_map(|f| { + let start = f["seqStart"].as_i64().unwrap_or(0); + let end = f["seqEnd"].as_i64().unwrap_or(0); + start..=end + }) + .collect() +} + +/// The concatenated output data of a frame list, in receive (seq) order. +fn concatenated_data(frames: &[serde_json::Value]) -> String { + frames + .iter() + .map(|f| f["data"].as_str().unwrap_or("")) + .collect() +} + +/// Negotiated attach with real scrollback: ready carries the retention +/// bounds, the first page is a BOUNDED prefix, and NO further replay frames +/// arrive while the client withholds credit. A raw-socket credit then +/// produces the next page; an out-of-window credit produces nothing. +#[tokio::test] +async fn negotiated_attach_gets_first_page_only_and_credit_gates_the_rest() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-paced").await; + // ~64KB of scrollback: many 4KB pages, far under the 512KB ring. + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-paced").await; + + // Ready bounds: the negotiated restore contract fields. + assert!( + ready["headSeq"].as_i64().unwrap_or(0) >= 1, + "honest head: {ready}" + ); + let oldest = ready["oldestRetainedSeq"] + .as_i64() + .expect("negotiated ready carries oldestRetainedSeq"); + let head = ready["headSeq"].as_i64().expect("headSeq"); + assert!( + oldest >= 1 && oldest <= head + 1, + "honest retention bound: {ready}" + ); + assert!( + ready.get("replayResetReason").is_none(), + "no retention loss in this fixture: {ready}" + ); + // The paced window description. + assert_eq!( + ready["replayToSeq"].as_i64(), + Some(head), + "replayToSeq is the fixed target: {ready}" + ); + assert_eq!( + ready["replayFromSeq"].as_i64(), + Some(1), + "the window starts at the baseline+1: {ready}" + ); + + // The first page is a bounded prefix of the window. + assert!(!page1.is_empty(), "there is replay to page"); + let last_seq = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("page frames carry seqEnd"); + assert!( + last_seq < head, + "the first page must not cover the whole window (head {head})" + ); + let page_bytes: usize = page1.iter().map(|f| f.to_string().len()).sum(); + assert!( + page_bytes as i64 <= PAGE_BUDGET, + "the first page honors the serialized budget: {page_bytes}" + ); + for frame in &page1 { + assert_eq!( + frame["source"], "replay", + "pages are stamped source:'replay'" + ); + assert_eq!(frame["attachRequestId"], "attach-paced"); + assert!(frame["seqStart"].as_i64().unwrap_or(0) >= 1); + } + + // WITHHOLD: no credit => no further pages (a window with no frames). + let withheld = next_json_or_timeout(&mut paced, Duration::from_millis(1500)).await; + assert!( + withheld.is_none() + || withheld + .as_ref() + .unwrap() + .get("type") + .and_then(|v| v.as_str()) + != Some("terminal.output"), + "no further replay frames may arrive without credit, got {withheld:?}" + ); + + // Beyond-window credit: ignored, produces nothing, and does not consume + // the grant (the next VALID credit still works). + credit(&mut paced, &terminal_id, "attach-paced", last_seq + 100_000).await; + let bad = next_json_or_timeout(&mut paced, Duration::from_millis(1200)).await; + assert!( + bad.is_none() + || bad.as_ref().unwrap().get("type").and_then(|v| v.as_str()) + != Some("terminal.output"), + "a beyond-window credit must produce nothing, got {bad:?}" + ); + + // Valid credit => the next page arrives. + credit(&mut paced, &terminal_id, "attach-paced", last_seq).await; + let next = next_json(&mut paced).await; + assert_eq!( + next["type"], "terminal.output", + "the credit produces the next page: {next}" + ); + let next_end = next["seqEnd"].as_i64().expect("seqEnd"); + assert!( + next_end > last_seq, + "pages ascend: page1 ends {last_seq}, next starts at {}", + next["seqStart"] + ); + assert_eq!(next["attachRequestId"], "attach-paced"); + assert_eq!(next["source"], "replay"); +} + +/// Send `terminal.attach` with a negotiated forward-page upper bound +/// (`replayPageBytes`, round-2 finding F3). +async fn attach_with_page_budget( + ws: &mut WsClient, + terminal_id: &str, + attach_request_id: &str, + replay_page_bytes: serde_json::Value, +) { + attach_with_page_budget_and_since(ws, terminal_id, attach_request_id, replay_page_bytes, 0) + .await +} + +/// [`attach_with_page_budget`] with an explicit `sinceSeq` baseline (the +/// paced restore contract's coverage cursor — E2R1 finding 2's sub-cap +/// fixture attaches each pane from just below its observed head, so the +/// credited window is a couple of frames and the session HOLDS on its +/// first page until the client credits). +async fn attach_with_page_budget_and_since( + ws: &mut WsClient, + terminal_id: &str, + attach_request_id: &str, + replay_page_bytes: serde_json::Value, + since_seq: i64, +) { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.attach", + "terminalId": terminal_id, + "intent": "viewport_hydrate", + "cols": 80, + "rows": 24, + "attachRequestId": attach_request_id, + "sinceSeq": since_seq, + "replayPageBytes": replay_page_bytes, + }) + .to_string(), + )) + .await + .expect("send terminal.attach with replayPageBytes"); +} + +/// Round-2 finding F3: the negotiated `replayPageBytes` wire field is +/// honored as the session's page upper bound END-TO-END — the field used +/// to be emitted by the client and stripped by the server's +/// accept-and-strip deserializer, so the server always paged at its own +/// cap. The bound is proven at the wire three ways: the session's +/// `ws.restore.paced_start` event carries the CLAMPED effective budget +/// (the requested value when under the cap), every packable page stays +/// within it while the session still converges, and malformed or absent +/// values fall back to the server's default exactly like the old strip +/// behavior instead of failing the attach frame. (The fixture's flood +/// frames are ~8.5 KiB — larger than any sub-cap bound — so a page that +/// cannot pack even one frame carries exactly ONE frame: the page +/// builder's explicit atomic over-budget result, the plan's bounded +/// behavior for a frame larger than the budget.) +#[tokio::test] +async fn negotiated_attach_honors_the_requested_replay_page_bytes() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-paced-budget").await; + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + // The requested bound sits BELOW the server's 4096-byte cap. + let requested = 2048usize; + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + attach_with_page_budget( + &mut paced, + &terminal_id, + "attach-budget", + serde_json::json!(requested), + ) + .await; + let mut ready = None; + let mut page: Vec = Vec::new(); + let mut head = 0i64; + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + while tokio::time::Instant::now() < deadline { + match next_json_or_timeout(&mut paced, Duration::from_millis(400)).await { + None => { + if ready.is_some() { + break; + } + continue; + } + Some(value) => match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.attach.ready") => { + head = value["headSeq"].as_i64().expect("headSeq"); + note_ready_stream(&value); + ready = Some(value); + } + Some("terminal.output") => page.push(value), + _ => {} + }, + } + } + ready.expect("attach.ready never arrived"); + assert!(!page.is_empty(), "there is replay to page"); + + // THE WIRE PROOF: the session's effective page budget is the REQUESTED + // bound clamped to itself (under the cap) — the stripped-field bug + // would report the server cap (4096) instead. + let start_ev = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.paced_start", + "attach_request_id", + "attach-budget", + ) + .await + .expect("the bounded attach emits its paced_start event"); + assert_eq!( + start_ev.fields.get("page_budget").map(String::as_str), + Some("2048"), + "the requested replayPageBytes becomes the session's page budget: {:?}", + start_ev.fields + ); + + // Credit-drive the rest of the window at the SAME bound: every page + // either packs strictly within it or carries exactly ONE frame (the + // atomic over-budget result for a frame larger than the bound — the + // flood's ~8.5 KiB frames), and the session converges. + // + // E2R1 finding 2 — the single-frame assertions, added honestly: the + // atomic exception is DOCUMENTED, not hidden. A single-frame page MAY + // exceed the request (the builder always includes the window's first + // frame), but it reaches ONLY the atomic page ceiling — the fragment + // cap plus the page-envelope slack, the same bound the drain + // admission reserves for sub-cap budgets. + let atomic_ceiling = freshell_terminal::paced_atomic_page_serialized_ceiling(); + let mut pages = 0usize; + let mut last_seq = 0i64; + let mut saw_atomic_over_budget = false; + loop { + let page_bytes: usize = page.iter().map(|f| f.to_string().len()).sum(); + if page.len() > 1 { + assert!( + page_bytes <= requested, + "a multi-frame page must pack within the {requested}-byte bound, got {page_bytes}" + ); + } else if page.len() == 1 { + assert!( + page_bytes <= atomic_ceiling, + "a single-frame page may reach only the documented atomic ceiling \ + ({atomic_ceiling}), got {page_bytes}" + ); + if page_bytes > requested { + saw_atomic_over_budget = true; + } + } + pages += 1; + last_seq = page + .last() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .unwrap_or(last_seq); + if last_seq >= head { + break; // the fixed target is covered + } + credit(&mut paced, &terminal_id, "attach-budget", last_seq).await; + page.clear(); + let mut saw_frame = false; + while tokio::time::Instant::now() < deadline { + match next_json_or_timeout(&mut paced, Duration::from_millis(400)).await { + None => break, // the page is complete + Some(value) => { + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + saw_frame = true; + page.push(value); + } + } + } + } + assert!( + saw_frame || !page.is_empty(), + "the credit must produce the next page" + ); + assert!( + tokio::time::Instant::now() < deadline, + "the session must converge at the requested bound" + ); + } + assert!(pages > 1, "the window must page repeatedly ({pages})"); + assert!( + saw_atomic_over_budget, + "the fixture exercises the documented atomic exception: at least one \ + single-frame page exceeds the {requested}-byte request (the flood's \ + ~8.5 KiB frames against the 2048-byte bound)" + ); + + // A MALFORMED bound (wrong-typed) keeps the server default and never + // fails the attach frame — the pre-contract accept-and-strip + // tolerance, preserved at the wire level. + let mut malformed = connect(&url).await; + hello(&mut malformed, true).await; + attach_with_page_budget( + &mut malformed, + &terminal_id, + "attach-bad-budget", + serde_json::json!("2048"), + ) + .await; + let bad_ev = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.paced_start", + "attach_request_id", + "attach-bad-budget", + ) + .await + .expect("a wrong-typed replayPageBytes still attaches"); + assert_eq!( + bad_ev.fields.get("page_budget").map(String::as_str), + Some("4096"), + "a malformed bound falls back to the server cap" + ); + + // An ABSENT bound keeps the server default (the field is optional and + // additive). + let mut default_attach = connect(&url).await; + hello(&mut default_attach, true).await; + paced_attach_first_page(&mut default_attach, &terminal_id, "attach-default-budget").await; + let default_ev = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.paced_start", + "attach_request_id", + "attach-default-budget", + ) + .await + .expect("the absent-bound attach emits its paced_start event"); + assert_eq!( + default_ev.fields.get("page_budget").map(String::as_str), + Some("4096"), + "an absent bound keeps the server cap" + ); +} + +/// A PARTIAL-PAGE credit (a consumedSeq inside the outstanding page, below +/// its end) grants NOTHING on the wire: the server observes it as +/// `partial_consumption` and produces no page — only the page-end value +/// produces the next page (the one-unacknowledged-page bound). +#[tokio::test] +async fn partial_page_credit_on_the_wire_grants_nothing_until_the_page_end() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-partial-credit").await; + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-partial").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let last_seq = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("the first page covers something"); + assert!(last_seq < head, "the session starts mid-replay"); + + // Two partial consumption reports inside the outstanding page: observed, + // inert — no page may arrive for either. + credit(&mut paced, &terminal_id, "attach-partial", last_seq - 2).await; + let after_partial1 = next_json_or_timeout(&mut paced, Duration::from_millis(1200)).await; + assert!( + after_partial1 + .as_ref() + .map(|v| v.get("type").and_then(|t| t.as_str()) != Some("terminal.output")) + .unwrap_or(true), + "a partial-page credit must produce no page, got {after_partial1:?}" + ); + credit(&mut paced, &terminal_id, "attach-partial", last_seq - 1).await; + let after_partial2 = next_json_or_timeout(&mut paced, Duration::from_millis(1200)).await; + assert!( + after_partial2 + .as_ref() + .map(|v| v.get("type").and_then(|t| t.as_str()) != Some("terminal.output")) + .unwrap_or(true), + "repeated partial credits must stay inert, got {after_partial2:?}" + ); + let partial_event = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "partial_consumption", + ) + .await + .expect("the partial credit is observed as partial_consumption"); + + // The page-end value grants: the next page arrives. + credit(&mut paced, &terminal_id, "attach-partial", last_seq).await; + let next = next_json(&mut paced).await; + assert_eq!( + next["type"], "terminal.output", + "the page-end credit produces the next page: {next}" + ); + assert_eq!(next["attachRequestId"], "attach-partial"); + assert!( + next["seqEnd"].as_i64().unwrap_or(0) > last_seq, + "pages ascend after the grant" + ); + assert_eq!( + partial_event.fields.get("terminal_id").map(String::as_str), + Some(terminal_id.as_str()) + ); +} + +/// A CONTINUOUSLY-PRODUCING terminal's paced session must COMPLETE (the +/// fixed-completion requirement): the tail drain never chases the moving +/// head — it pages to the tail-start head and hands the staged remainder +/// to the live path — so `ws.restore.paced_complete` fires while +/// production is still running, and the connection keeps processing +/// input afterwards. The old moving-head drain chased forever, wedging +/// the connection's dispatch task inline. +#[tokio::test] +async fn paced_session_completes_while_the_terminal_keeps_producing() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-producing").await; + + // The sustained producer: `yes` with no head — production never pauses. + // A primer attach precedes the input (the suite's proven driver + // pattern: the attach applies the PTY's geometry and the shell then + // executes the buffered line). The primer then observes the flood + // demonstrably flowing before detaching, so the paced attach happens + // against a terminal that is KNOWN to be mid-production. + let mut primer = connect(&url).await; + hello(&mut primer, false).await; + attach(&mut primer, &terminal_id, "attach-producing-primer").await; + send_input( + &mut driver, + &terminal_id, + "yes 'STREAMDATA-XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX'\n", + ) + .await; + let primer_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let mut flood_observed = false; + while tokio::time::Instant::now() < primer_deadline { + let Some(value) = next_json_or_timeout(&mut primer, Duration::from_secs(2)).await else { + continue; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let data = value.get("data").and_then(|d| d.as_str()).unwrap_or(""); + // Flood OUTPUT rows, never the kernel's echo of the typed + // command line itself (`yes 'STREAMDATA-…'`). + if data.contains("STREAMDATA") && !data.contains("yes '") { + flood_observed = true; + break; + } + } + } + assert!( + flood_observed, + "the sustained producer must be flowing before the paced attach" + ); + primer + .send(WsMessage::Text( + serde_json::json!({ "type": "terminal.detach", "terminalId": terminal_id }).to_string(), + )) + .await + .expect("primer detaches"); + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-producing").await; + let attach_head = ready["headSeq"].as_i64().expect("headSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + credited > 0, + "the producer staged content before the attach" + ); + + // Drive the session with page-end credits (frame-end credits include + // every page's end). Retention expiry rounds arrive as negotiated + // gaps; the session continues within the same credit. + credit(&mut paced, &terminal_id, "attach-producing", credited).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let mut produced_past_target = false; + let mut completed = false; + while tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + break; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let end = value["seqEnd"].as_i64().unwrap_or(0); + produced_past_target |= end > attach_head; + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-producing", end).await; + } + } + Some("terminal.output.gap") => { + // The negotiated retention gap: continuation is within the + // already-granted credit; keep consuming frames. + } + _ => {} + } + if !completed { + completed = wait_for_restore_event_of_terminal( + &events, + &terminal_id, + "ws.restore.paced_complete", + ) + .await + .is_some(); + } + // The handoff's in-flight frames flow after the completion event: + // drain them (they carry seqs past the attach target) before + // asserting. + if completed && produced_past_target { + break; + } + } + + // THE completion: the session closed while `yes` is still running. + let complete = + wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_complete") + .await + .expect("the paced session completes against a continuously-producing terminal"); + assert_eq!( + complete.fields.get("attach_request_id").map(String::as_str), + Some("attach-producing") + ); + let last_seq = complete + .fields + .get("last_seq") + .and_then(|v| v.parse::().ok()) + .expect("last_seq recorded"); + assert!( + last_seq > attach_head, + "production continued past the attach-time head (attach {attach_head}, completed {last_seq}) — the session completed WITHOUT waiting for production to pause" + ); + assert!( + produced_past_target, + "the fixture's producer demonstrably outran the attach target" + ); + + // The connection is not wedged: stop the flood, then a fresh echo + // command round-trips through THE SAME (paced) socket — the + // connection that ran the drain must itself service input, not only + // a bystander connection. + send_input(&mut paced, &terminal_id, "\u{3}").await; + let echo_marker = "PRODUCING-ECHO-MARKER"; + send_input(&mut paced, &terminal_id, &format!("echo '{echo_marker}'\n")).await; + let drain_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let (acc, _) = drain_until_marker(&mut paced, echo_marker, drain_deadline).await; + assert!( + acc.contains(echo_marker), + "the connection still delivers output after the producing-terminal session completed" + ); + + // The session is gone: a further credit is a stale generation. + credit(&mut paced, &terminal_id, "attach-producing", last_seq).await; + let stale = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "stale_generation", + ) + .await + .expect("post-completion credits are stale generations"); + assert_eq!( + stale.fields.get("terminal_id").map(String::as_str), + Some(terminal_id.as_str()) + ); +} + +/// A helper for the producing-drain tests: the sustained `yes` primer — +/// attach a non-negotiated observer, start the flood through the driver, +/// and wait until flood OUTPUT is demonstrably flowing before detaching. +/// Returns the primer socket for the caller to detach. +async fn start_sustained_flood( + url: &str, + driver: &mut WsClient, + terminal_id: &str, + primer_arid: &str, +) -> WsClient { + let mut primer = connect(url).await; + hello(&mut primer, false).await; + attach(&mut primer, terminal_id, primer_arid).await; + send_input( + driver, + terminal_id, + "yes 'STREAMDATA-XXXXXXXXXXXXXXXXXXXXXXXXXXXXXXXX'\n", + ) + .await; + let primer_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + while tokio::time::Instant::now() < primer_deadline { + let Some(value) = next_json_or_timeout(&mut primer, Duration::from_secs(2)).await else { + continue; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let data = value.get("data").and_then(|d| d.as_str()).unwrap_or(""); + // Flood OUTPUT rows, never the kernel's echo of the typed + // command line itself (`yes 'STREAMDATA-…'`). + if data.contains("STREAMDATA") && !data.contains("yes '") { + break; + } + } + } + primer +} + +/// A producing drain's SAME-CONNECTION input service (the round-2 +/// recapture-loophole fix's dispatcher-bound requirement): while the +/// un-credited tail drain is STILL RUNNING — provably held mid-flight by +/// the connection's real output backpressure — an input frame sent on +/// THE SAME negotiated socket must be serviced by the connection +/// dispatcher BEFORE the drain completes. The inline-drain structure this +/// test replaces monopolized the dispatcher for the drain's whole +/// duration, so same-connection input could only ever be answered after +/// completion; the drain task + backpressure gate makes the answer arrive +/// first. The input is a bogus-terminal probe (the deterministic, +/// content-free `terminal.input.blocked` answer — the kata dtfn lane). +#[tokio::test] +async fn input_on_the_same_connection_is_serviced_while_a_producing_drain_is_still_running() { + let events = global_capture(); + let ring = 512 * 1024; + // A small TERM-09 output queue: the drain's own pages blow past the + // queue's backlog watermark almost immediately, so the backpressure- + // gated drain stays provably mid-flight (the completion cannot fire + // until the client consumes the backlog) across a wide, deterministic + // window in which to probe the dispatcher. + let url = spawn_server_with(ring, PAGE_BUDGET, Some(16 * 1024)).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-drain-input").await; + + // A large BOUNDED flood (`yes | head -n 100000`): ~5 MB — many ring + // capacities — so the producer demonstrably overlaps the attach and + // the credited replay (production keeps advancing the head while the + // client pages toward the attach target, leaving the un-credited + // drain a large staged tail), AND it self-terminates via SIGPIPE with + // the shell left at a prompt — no tty-signal dependence anywhere (a + // mid-drain Ctrl-C proved environment-flaky: whether the SIGINT + // reaches only `yes` or the shell itself depends on the spawned + // shell's job-control shape). The round-3 bounded handoff completes + // once the producer stops and the client consumes; the pre-fix drain + // only "completed" against a live producer by evicting its own pages + // past the queue cap in one ungated dump. + let mut primer = connect(&url).await; + hello(&mut primer, false).await; + attach(&mut primer, &terminal_id, "attach-drain-input-primer").await; + send_input( + &mut driver, + &terminal_id, + &flood_command(100_000, "FLOOD-DONE-MARKER"), + ) + .await; + let primer_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + loop { + let Some(value) = next_json_or_timeout(&mut primer, Duration::from_secs(2)).await else { + panic!("the bounded flood must start producing before the attach"); + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let data = value.get("data").and_then(|d| d.as_str()).unwrap_or(""); + // Flood OUTPUT rows, never the kernel's echo of the typed + // command line itself. + if data.contains("STREAMDATA") && !data.contains("yes '") { + break; + } + } + if tokio::time::Instant::now() >= primer_deadline { + panic!("the bounded flood must start producing before the attach"); + } + } + primer + .send(WsMessage::Text( + serde_json::json!({ "type": "terminal.detach", "terminalId": terminal_id }).to_string(), + )) + .await + .expect("primer detaches"); + + // Attach IMMEDIATELY (no fill delay — maximize the producer's live + // overlap with the credited replay): the drain's probe window is held + // open by the client's own slow reads against the staged tail, so a + // fill delay buys nothing. + + let mut paced = connect(&url).await; + // REAL backpressure needs the KERNEL to stop absorbing the drain: a + // loopback socket's default buffers swallow the whole drain before + // the writer queue ever fills, so the gate (and the drain's + // mid-flight window) would never engage. Shrink THIS client's + // receive buffer: the TCP window closes at kilobytes, the server's + // writer sends park, its queue fills to the watermark, and the drain + // provably waits on the connection's real consumption. + if let tokio_tungstenite::MaybeTlsStream::Plain(tcp) = paced.get_ref() { + socket2::SockRef::from(tcp) + .set_recv_buffer_size(8 * 1024) + .expect("shrink the test client's receive buffer"); + } + hello(&mut paced, true).await; + let (ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-drain-input").await; + let target = ready["replayToSeq"].as_i64().expect("replayToSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + credited > 0, + "the producer staged content before the attach" + ); + + // Credit the replay to its fixed target. The drain (the un-credited + // tail) begins when the final page reaches the target. + credit(&mut paced, &terminal_id, "attach-drain-input", credited).await; + let mut drain_started = false; + let mut frames_after_input_probe = 0u64; + let mut input_probe_answered = false; + let mut input_probe_sent = false; + // Generous wall-clock windows: the suite runs these PTY+socket tests + // in parallel, and the assertions are behavioral — the deadlines only + // need to dominate scheduling noise, never the observed behavior. + let deadline = tokio::time::Instant::now() + Duration::from_secs(60); + 'drain_watch: while tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + break 'drain_watch; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + if end < target { + // Mid-replay pages stay credit-driven. + credit(&mut paced, &terminal_id, "attach-drain-input", end).await; + } else if !drain_started { + // The first page at/past the fixed target: the + // un-credited drain is now demonstrably running. + drain_started = true; + } + } + if input_probe_sent { + frames_after_input_probe += 1; + } + // Deliberately consume the DRAIN slowly — but only once it + // is running: the credited replay phase reads at full + // speed (the probe window opens inside the flood's fixed + // wall-clock life), and the slow reads then make the + // drain's own pages outrun the client so the backpressure + // gate holds the drain open across the probe window. + if drain_started { + tokio::time::sleep(Duration::from_millis(25)).await; + } + } + Some("terminal.input.blocked") => { + // The probe's answer arrived on THIS socket: the dispatcher + // serviced same-connection input mid-drain. Deliberately NOT + // compared against the server-side ws.restore.paced_complete + // event: completion is emitted server-side when the drain's + // production ends, while this frame's ARRIVAL is serialized + // behind whatever drain pages already hold the queue (this + // client reads them slowly on purpose), so a correctly + // serviced probe can still land after the completion event. + // The mid-drain proof rides on the frames-after observable + // below — the pre-fix synchronous-drain stall delivered the + // answer only AFTER the drain's last page. + assert_eq!( + value["reason"], "unknown_terminal", + "the probe is answered by the input-blocked frame: {value}" + ); + input_probe_answered = true; + if !input_probe_sent { + panic!("probe answer observed before the probe was sent"); + } + } + _ => {} + } + if drain_started && !input_probe_sent { + input_probe_sent = true; + send_input(&mut paced, "no-such-terminal-drain-input", "x").await; + } + if input_probe_answered && frames_after_input_probe > 0 { + break 'drain_watch; + } + } + + assert!( + input_probe_sent, + "the drain must have demonstrably started for the probe window to open" + ); + assert!( + input_probe_answered, + "the dispatcher must service same-connection input even while a \ + producing drain is running (the probe answer never arrived)" + ); + assert!( + frames_after_input_probe > 0, + "the drain still had pages in flight after the probe answer (the probe was mid-drain, not after it)" + ); + + // The drain then completes once the bounded flood has self-terminated + // and the client consumes. The round-3 bounded handoff pages the + // staged post-target remainder through the same backpressure gate — + // one page budget per gate-pass — so the completion wait CONSUMES at + // full speed while it polls (the pre-fix drain only "completed" + // against the live flood by evicting its own pages past the queue + // cap in one ungated dump). + let paced_complete_fired = || { + events.lock().unwrap().iter().any(|e| { + e.message == "ws.restore.paced_complete" + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id.as_str()) + }) + }; + let mut complete = None; + let complete_deadline = tokio::time::Instant::now() + Duration::from_secs(60); + let mut reads_since_check = 0u32; + while tokio::time::Instant::now() < complete_deadline { + if next_json_or_timeout(&mut paced, Duration::from_millis(50)) + .await + .is_some() + { + reads_since_check += 1; + if reads_since_check >= 200 { + reads_since_check = 0; + if paced_complete_fired() { + complete = Some(()); + break; + } + } + continue; + } + // Quiet socket: check for the completion before the next read + // window. + if paced_complete_fired() { + complete = Some(()); + break; + } + } + assert!( + complete.is_some(), + "the producing drain completes after the gate releases" + ); + let complete = + wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_complete") + .await + .expect("the completion event is captured"); + assert_eq!( + complete.fields.get("attach_request_id").map(String::as_str), + Some("attach-drain-input") + ); + + // The pane-health postscript (a post-drain Ctrl-C + echo round-trip) + // lives in `paced_session_completes_while_the_terminal_keeps_producing` + // under the default queue/socket conditions, where it is stable; this + // test's uniquely choked fixture (the 16 KiB queue + shrunken RCVBUF + // that make the gate observable) is what its three assertions need, + // and they end here. +} + +/// Round-5 finding 1 (degenerate settings): the MINIMUM supported queue — +/// the TERM-09 64 KiB floor — is smaller than one default 128 KiB paced +/// page. The server boot clamps the page budget to the queue's admission +/// ceiling (32 KiB; this harness mirrors the boot's clamp), and the +/// drain's reserve-then-admit gate then walks the restore page by page +/// into the tiny queue. The paced session must COMPLETE within the +/// window (never deadlock — the reservation's oversize arm admits into +/// a fully drained queue even if the clamp were bypassed) and the +/// connection must NEVER self-spill its own pages: ZERO client-visible +/// `queue_overflow` gaps and ZERO server-side spill events across the +/// whole restore. +#[tokio::test] +async fn a_minimum_queue_restore_never_deadlocks_or_self_spills() { + let events = global_capture(); + let ring = 512 * 1024; + // The degenerate pairing, as configured: the default 128 KiB page + // budget request against the 64 KiB queue floor (the harness mirrors + // the boot clamp, so the effective budget is 32 KiB). + let url = spawn_server_with(ring, 128 * 1024, Some(64 * 1024)).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-min-queue").await; + + // A large BOUNDED flood (`yes | head -n 100000`, ~5 MB): the producer + // demonstrably outruns the attach target, so the un-credited drain + // walks a real staged remainder through the tiny queue — the + // degenerate paging this test exists to cover — and it + // self-terminates (SIGPIPE leaves the shell at its prompt). + let mut primer = connect(&url).await; + hello(&mut primer, false).await; + attach(&mut primer, &terminal_id, "attach-min-queue-primer").await; + send_input( + &mut driver, + &terminal_id, + &flood_command(100_000, "FLOOD-DONE-MARKER"), + ) + .await; + tokio::time::timeout(Duration::from_secs(20), async { + loop { + let Some(value) = next_json_or_timeout(&mut primer, Duration::from_secs(2)).await + else { + continue; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let data = value.get("data").and_then(|d| d.as_str()).unwrap_or(""); + if data.contains("STREAMDATA") && !data.contains("yes '") { + break; + } + } + } + }) + .await + .expect("the bounded flood must start producing before the attach"); + primer + .send(WsMessage::Text( + serde_json::json!({ "type": "terminal.detach", "terminalId": terminal_id }).to_string(), + )) + .await + .expect("primer detaches"); + + // The negotiated restore over the minimum-sized queue. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let arid = "attach-min-queue"; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, arid).await; + let target = ready["replayToSeq"].as_i64().expect("replayToSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + credited > 0, + "the producer staged content before the attach" + ); + // The first page is consumed: credit its end so the session can page + // on toward the fixed target (the credited phase's one-page-per-credit + // contract). + if credited < target { + credit(&mut paced, &terminal_id, arid, credited).await; + } + + // THE CLAMP IS VISIBLE: the session's first page is bounded by the + // clamped 32 KiB budget (the boot-mirrored ceiling for this queue), + // never the requested 128 KiB default — a full-size unclamped page + // would already have self-spilled the 64 KiB queue at attach. + let start = wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_start") + .await + .expect("the paced start event is captured"); + let page_bytes: i64 = start + .fields + .get("page_bytes") + .and_then(|v| v.parse().ok()) + .expect("paced_start carries page_bytes"); + assert!( + page_bytes > 0 && page_bytes <= 32 * 1024, + "the clamp caps pages at the queue's admission ceiling (got {page_bytes}B)" + ); + + // Consume at full speed, crediting toward the fixed target, then + // through the un-credited drain. THE BOUND: no queue_overflow gap may + // EVER arrive — the clamped, reserved admissions keep the aggregate + // inside the 64 KiB queue. Retention gaps (`replay_window_exceeded`) + // are the ring churn this fixture's 5 MB flood produces; they are the + // honest declared loss and legal. + let spill_events = || { + events + .lock() + .unwrap() + .iter() + .filter(|e| { + e.message.contains("queue_overflow_spill") + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id.as_str()) + }) + .count() + }; + let drain_completed = || { + events.lock().unwrap().iter().any(|e| { + e.message == "ws.restore.paced_complete" + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id.as_str()) + }) + }; + let deadline = tokio::time::Instant::now() + Duration::from_secs(90); + while !drain_completed() && tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + if end < target { + credit(&mut paced, &terminal_id, arid, end).await; + } + } + } + Some("terminal.output.gap") => { + assert_ne!( + value.get("reason").and_then(|v| v.as_str()).unwrap_or(""), + "queue_overflow", + "THE BOUND: a 64 KiB queue must never spill its own clamped \ + restore pages: {value}" + ); + } + _ => {} + } + } + if !drain_completed() { + let terminal_events = { + let events = events.lock().unwrap(); + events + .iter() + .filter(|e| { + e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id.as_str()) + }) + .map(|e| (e.message.clone(), e.fields.clone())) + .collect::>() + }; + panic!( + "NO DEADLOCK: the minimum-queue restore must complete within the window \ + (the reservation's oversize arm admits into a fully drained queue); \ + terminal events: {terminal_events:?}" + ); + } + assert_eq!( + spill_events(), + 0, + "THE BOUND: zero server-side spill events across the whole restore" + ); +} + +/// Read ONE frame (400ms quiet window) and route it to its pane's bucket +/// (0 = terminal A, 1 = terminal B of the caller's pair). Returns false +/// on a quiet read. Gap frames assert the no-self-spill bound inline. +async fn read_and_route_by_pane( + paced: &mut WsClient, + terminals: &[String; 2], + readies: &mut [Option; 2], + delivered: &mut [Vec<(i64, i64)>; 2], + declared_gaps: &mut Vec<(usize, i64, i64)>, +) -> bool { + let Some(value) = next_json_or_timeout(paced, Duration::from_millis(400)).await else { + return false; // quiet + }; + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or("") + .to_string(); + let pane = if term == terminals[0] { + 0 + } else if term == terminals[1] { + 1 + } else { + return true; // not ours (handshake stragglers): keep reading + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.attach.ready") => { + note_ready_stream(&value); + readies[pane] = Some(value); + } + Some("terminal.output") => delivered[pane].push(( + value["seqStart"].as_i64().unwrap_or(0), + value["seqEnd"].as_i64().unwrap_or(0), + )), + Some("terminal.output.gap") => { + assert_ne!( + value.get("reason").and_then(|v| v.as_str()).unwrap_or(""), + "queue_overflow", + "THE BOUND: no drain-induced spill: {value}" + ); + declared_gaps.push(( + pane, + value["fromSeq"].as_i64().unwrap_or(0), + value["toSeq"].as_i64().unwrap_or(0), + )); + } + _ => {} + } + true +} + +/// Round-3 finding 1, the per-connection bound: when MULTIPLE panes +/// restore over ONE connection, every completion-path admission stays +/// page-budget-sized and gated by the connection queue's real +/// consumption, so the panes' aggregate admitted-but-unconsumed backlog +/// stays under the queue's byte cap BY CONSTRUCTION — observed as ZERO +/// drain-induced `queue_overflow` spills (the queue's eviction is the +/// only mechanism that can push delivery past the cap, and it never +/// fires). The pre-fix completing call bulk-admitted the ENTIRE staged +/// post-target remainder — bounded only by each pane's ring — in ONE +/// ungated lock hold per pane: N panes aggregated up to N rings past the +/// cap in a single gate-pass each, and the queue evicted the drains' own +/// pages (client-visible `queue_overflow` gaps and undeclared holes). +/// +/// Round-5 finding 1: the fixture now uses REALISTIC page sizes — the +/// production-default 128 KiB budget and the production-default 16 MiB +/// queue (the pre-round-5 fixture shrank pages to 4 KiB, avoiding the +/// full-size concurrency this test exists to cover). The tight-watermark +/// aggregate bound itself (concurrent full-size pages against a +/// just-under-watermark backlog) is pinned deterministically at the +/// writer unit lane +/// (`concurrent_full_size_drain_admissions_never_self_spill_the_queue`); +/// THIS test proves the end-to-end shape: two concurrently-draining +/// panes with production-size pages never spill the connection. +#[tokio::test] +async fn multiple_restoring_panes_on_one_connection_stay_within_the_connection_queue_bound() { + let events = global_capture(); + let ring = 512 * 1024; + // The bound this test asserts: the gated drains' aggregate backlog + // stays below the queue cap, so the connection NEVER self-spills. + // Full-size pages: the production-default 128 KiB budget under the + // production-default 16 MiB queue cap. + let url = spawn_server_with(ring, 128 * 1024, None).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_a = create_shell_terminal(&mut driver, "create-multi-a").await; + let terminal_b = create_shell_terminal(&mut driver, "create-multi-b").await; + + // Both producers flood concurrently: bounded (`yes | head -n 100000` + // ≈ 5 MB each — many ring capacities, so production demonstrably + // outruns each attach target and the drains face real post-target + // remainders) and self-terminating (SIGPIPE leaves the shells at + // prompts; the drains then converge deterministically). + let mut observer = connect(&url).await; + hello(&mut observer, false).await; + attach(&mut observer, &terminal_a, "attach-multi-observer").await; + attach(&mut observer, &terminal_b, "attach-multi-observer").await; + send_input( + &mut driver, + &terminal_a, + &flood_command(100_000, "FLOOD-DONE-MARKER"), + ) + .await; + send_input( + &mut driver, + &terminal_b, + &flood_command(100_000, "FLOOD-DONE-MARKER"), + ) + .await; + let mut flood_seen = [false, false]; + let flood_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + while !(flood_seen[0] && flood_seen[1]) && tokio::time::Instant::now() < flood_deadline { + let Some(value) = next_json_or_timeout(&mut observer, Duration::from_secs(2)).await else { + continue; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let data = value.get("data").and_then(|d| d.as_str()).unwrap_or(""); + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or(""); + // Flood OUTPUT rows, never the kernel's echo of the typed + // command line itself. + if data.contains("STREAMDATA") && !data.contains("yes '") { + if term == terminal_a { + flood_seen[0] = true; + } else if term == terminal_b { + flood_seen[1] = true; + } + } + } + } + assert!( + flood_seen[0] && flood_seen[1], + "both floods must be flowing" + ); + observer + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.detach", + "terminalId": terminal_a, + }) + .to_string(), + )) + .await + .expect("observer detaches A"); + observer + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.detach", + "terminalId": terminal_b, + }) + .to_string(), + )) + .await + .expect("observer detaches B"); + + // ONE negotiated connection restores BOTH panes. Every frame read + // during the attach sequence is routed by terminal — a pane's + // retention-gap frame can arrive while the OTHER pane's attach burst + // is being read, and the contiguity check must not lose it. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let mut readies: [Option; 2] = [None, None]; + let mut delivered: [Vec<(i64, i64)>; 2] = [Vec::new(), Vec::new()]; + let mut declared_gaps: Vec<(usize, i64, i64)> = Vec::new(); + let terminals = [terminal_a.clone(), terminal_b.clone()]; + let read_and_route = read_and_route_by_pane; + attach(&mut paced, &terminal_a, "attach-multi-a").await; + while readies[0].is_none() { + assert!( + read_and_route( + &mut paced, + &terminals, + &mut readies, + &mut delivered, + &mut declared_gaps + ) + .await, + "pane A's attach.ready must arrive" + ); + } + attach(&mut paced, &terminal_b, "attach-multi-b").await; + // Read until BOTH readies are in and the socket goes quiet — pane A's + // first page (and any retention gap) can trail pane B's burst. + while readies[1].is_none() { + assert!( + read_and_route( + &mut paced, + &terminals, + &mut readies, + &mut delivered, + &mut declared_gaps + ) + .await, + "pane B's attach.ready must arrive" + ); + } + for _ in 0..8 { + if !read_and_route( + &mut paced, + &terminals, + &mut readies, + &mut delivered, + &mut declared_gaps, + ) + .await + { + break; // quiet: both attach bursts are complete + } + } + let target_a = readies[0].as_ref().expect("ready A")["replayToSeq"] + .as_i64() + .expect("replayToSeq A"); + let target_b = readies[1].as_ref().expect("ready B")["replayToSeq"] + .as_i64() + .expect("replayToSeq B"); + let targets = [target_a, target_b]; + let arids = ["attach-multi-a", "attach-multi-b"]; + let mut credited = [ + delivered[0].iter().map(|(_, e)| *e).max().unwrap_or(0), + delivered[1].iter().map(|(_, e)| *e).max().unwrap_or(0), + ]; + assert!(credited[0] > 0, "pane A staged content before the attach"); + assert!(credited[1] > 0, "pane B staged content before the attach"); + credit(&mut paced, &terminal_a, "attach-multi-a", credited[0]).await; + credit(&mut paced, &terminal_b, "attach-multi-b", credited[1]).await; + + // Consume at full speed, crediting each pane toward its fixed target, + // until BOTH drains complete. THE BOUND: no `queue_overflow` gap may + // EVER arrive — the gated, page-budget-sized admissions keep the + // aggregate backlog under the queue cap for the whole concurrent + // restore. Retention gaps (`replay_window_exceeded`) are the honest + // declared loss this fixture's ring churn produces; they are legal. + let mut max_seq = [credited[0], credited[1]]; + let mut completions = [false, false]; + let drain_completed = |events: &Arc>>, terminal_id: &str| -> bool { + events.lock().unwrap().iter().any(|e| { + e.message == "ws.restore.paced_complete" + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id) + }) + }; + let deadline = tokio::time::Instant::now() + Duration::from_secs(60); + while !(completions[0] && completions[1]) && tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + completions[0] |= drain_completed(&events, &terminal_a); + completions[1] |= drain_completed(&events, &terminal_b); + continue; + }; + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or("") + .to_string(); + let pane = if term == terminal_a { + 0 + } else if term == terminal_b { + 1 + } else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let end = value["seqEnd"].as_i64().unwrap_or(0); + let start = value["seqStart"].as_i64().unwrap_or(0); + max_seq[pane] = max_seq[pane].max(end); + // Delivered ranges must be recorded for the hole check; + // credit mid-replay pages toward the fixed target. + delivered[pane].push((start, end)); + if end > credited[pane] { + credited[pane] = end; + if end < targets[pane] { + credit(&mut paced, &terminals[pane], arids[pane], end).await; + } + } + } + Some("terminal.output.gap") => { + let reason = value.get("reason").and_then(|v| v.as_str()).unwrap_or(""); + assert_ne!( + reason, "queue_overflow", + "THE BOUND: the gated, page-budget-sized handoff keeps the \ + connection's aggregate admitted bytes under the queue cap — \ + a queue_overflow gap means a drain spilled the connection's \ + own queued pages: {value}" + ); + declared_gaps.push(( + pane, + value["fromSeq"].as_i64().unwrap_or(0), + value["toSeq"].as_i64().unwrap_or(0), + )); + } + _ => {} + } + completions[0] |= drain_completed(&events, &terminal_a); + completions[1] |= drain_completed(&events, &terminal_b); + } + assert!(completions[0], "pane A's drain completes within the window"); + assert!(completions[1], "pane B's drain completes within the window"); + + // The completion events fire at ADMISSION: the final pages/chunks may + // still be in flight on the socket. Drain until quiet before the + // contiguity check. + let quiet_drain = tokio::time::Instant::now() + Duration::from_secs(10); + let mut quiet_streak = 0u8; + while tokio::time::Instant::now() < quiet_drain && quiet_streak < 3 { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_millis(300)).await else { + quiet_streak += 1; + continue; + }; + quiet_streak = 0; + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or("") + .to_string(); + let pane = if term == terminal_a { + 0 + } else if term == terminal_b { + 1 + } else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let end = value["seqEnd"].as_i64().unwrap_or(0); + let start = value["seqStart"].as_i64().unwrap_or(0); + max_seq[pane] = max_seq[pane].max(end); + delivered[pane].push((start, end)); + } + Some("terminal.output.gap") => { + let reason = value.get("reason").and_then(|v| v.as_str()).unwrap_or(""); + assert_ne!( + reason, "queue_overflow", + "THE BOUND: no drain-induced spill, even in the in-flight tail: {value}" + ); + declared_gaps.push(( + pane, + value["fromSeq"].as_i64().unwrap_or(0), + value["toSeq"].as_i64().unwrap_or(0), + )); + } + _ => {} + } + } + + // Production demonstrably outran both attach targets (the completing + // calls faced real post-target remainders). + assert!( + max_seq[0] > target_a, + "pane A's producer outran its attach target" + ); + assert!( + max_seq[1] > target_b, + "pane B's producer outran its attach target" + ); + + // No undeclared holes: every delivered hole lies inside the pane's + // DECLARED LOSS — the merged union of its retention-gap intervals + // (adjacent declared gaps tile, exactly like the client's + // knownLostRanges merging). + for pane in [0usize, 1usize] { + let mut gap_ranges: Vec<(i64, i64)> = declared_gaps + .iter() + .filter(|(gap_pane, _, _)| *gap_pane == pane) + .map(|(_, from, to)| (*from, *to)) + .collect(); + gap_ranges.sort_unstable(); + let mut merged: Vec<(i64, i64)> = Vec::new(); + for (from, to) in gap_ranges { + match merged.last_mut() { + Some((_, last_to)) if from <= *last_to + 1 => { + *last_to = (*last_to).max(to); + } + _ => merged.push((from, to)), + } + } + let mut ranges = delivered[pane].clone(); + ranges.sort_unstable(); + let mut prev_end: Option = None; + for (start, end) in &ranges { + if *start <= 0 { + continue; + } + if let Some(prev) = prev_end { + if *start > prev + 1 { + let hole = (prev + 1, *start - 1); + let declared = merged + .iter() + .any(|(from, to)| hole.0 >= *from && hole.1 <= *to); + assert!( + declared, + "pane {pane} has an undeclared seq hole {hole:?} (declared loss {merged:?})" + ); + } + } + prev_end = Some(prev_end.map_or(*end, |p| p.max(*end))); + } + } +} + +/// E2R1 finding 2(b) — the multi-pane zero-spill canary extended with +/// SUB-CAP page requests: the end-to-end shape. Multiple panes restore +/// over ONE connection with `replayPageBytes` far below the production +/// frame size (128 bytes), so every page the session produces is the +/// builder's ATOMIC single-frame result (~4-8.5 KiB, the documented +/// exception to the requested bound). At the canary's own production +/// sizing this proves the whole chain: the sub-cap request is honored +/// as every session's budget, the atomic pages flow and converge, the +/// concurrent drains complete against real post-target remainders, and +/// the connection never self-spills (`queue_overflow` never appears). +/// +/// The TIGHT-WATERMARK aggregate bound — the reservation must account +/// max(requested budget, the atomic page ceiling) so concurrent +/// sub-cap drains cannot overbook the admission gate — is pinned +/// deterministically at the writer unit lane +/// (`concurrent_sub_cap_drain_admissions_never_under_reserve_the_atomic_page`), +/// exactly the split the round-5 canary itself uses for its full-size +/// twin (an end-to-end test at a queue small enough to weaponize every +/// legal in-flight shape times out on fixture noise, not on the bound). +#[tokio::test] +async fn sub_cap_page_requests_restore_end_to_end_at_the_atomic_page_bound() { + const PANES: usize = 2; + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server_with(ring, PAGE_BUDGET, None).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let mut terminals: Vec = Vec::new(); + for pane in 0..PANES { + terminals.push(create_shell_terminal(&mut driver, &format!("create-subcap-{pane}")).await); + } + let arids: Vec = (0..PANES) + .map(|pane| format!("attach-subcap-{pane}")) + .collect(); + + // The canary's flood discipline: the observer attaches before any + // flood exists, both panes flood concurrently (bounded, + // self-terminating), production demonstrably outruns every attach + // target, and the observer detaches before the restore. + let mut observer = connect(&url).await; + hello(&mut observer, false).await; + for terminal in &terminals { + attach(&mut observer, terminal, "attach-subcap-observer").await; + } + for terminal in &terminals { + send_input( + &mut driver, + terminal, + &flood_command(100_000, "FLOOD-DONE-MARKER"), + ) + .await; + } + let mut flood_seen = [false; PANES]; + let flood_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + while !flood_seen.iter().all(|seen| *seen) && tokio::time::Instant::now() < flood_deadline { + let Some(value) = next_json_or_timeout(&mut observer, Duration::from_secs(2)).await else { + continue; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let data = value.get("data").and_then(|d| d.as_str()).unwrap_or(""); + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or(""); + if data.contains("STREAMDATA") && !data.contains("yes '") { + if let Some(pane) = terminals + .iter() + .position(|terminal| terminal.as_str() == term) + { + flood_seen[pane] = true; + } + } + } + } + assert!( + flood_seen.iter().all(|seen| *seen), + "both floods must be flowing before the restore" + ); + for terminal in &terminals { + observer + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.detach", + "terminalId": terminal, + }) + .to_string(), + )) + .await + .expect("observer detaches"); + } + drop(observer); + + // ONE negotiated connection restores BOTH panes with the sub-cap + // budget. Each attach's first page is read before the next attach. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let mut targets = [0i64; PANES]; + let mut delivered: Vec> = vec![Vec::new(); PANES]; + let mut credited = [0i64; PANES]; + for pane in 0..PANES { + let (ready, page) = paced_attach_first_page_with_budget( + &mut paced, + &terminals[pane], + &arids[pane], + serde_json::json!(128), + ) + .await; + targets[pane] = ready["replayToSeq"].as_i64().expect("replayToSeq"); + for frame in &page { + delivered[pane].push(( + frame["seqStart"].as_i64().unwrap_or(0), + frame["seqEnd"].as_i64().unwrap_or(0), + )); + } + credited[pane] = page + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + credited[pane] > 0, + "pane {pane}'s first page carries content" + ); + } + for pane in 0..PANES { + credit(&mut paced, &terminals[pane], &arids[pane], credited[pane]).await; + } + + // Consume at full speed, crediting each pane toward its fixed target + // and then through the concurrent drains, until BOTH complete. THE + // BOUND (the canary's): no `queue_overflow` gap may EVER arrive. + // Retention gaps (`replay_window_exceeded`) are the honest declared + // loss this fixture's ring churn produces; they are legal. + let mut declared_gaps: Vec<(usize, i64, i64)> = Vec::new(); + let mut max_seq = [0i64; PANES]; + // The sub-cap contract's load-bearing evidence: the sessions page + // single frames whose serialized size EXCEEDS the 128-byte request — + // the documented atomic exception, real on the wire. + let mut saw_frame_over_budget = false; + let drain_completed = |terminal_id: &str| -> bool { + events.lock().unwrap().iter().any(|e| { + e.message == "ws.restore.paced_complete" + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id) + }) + }; + let pane_of = |term: &str| -> Option { + terminals + .iter() + .position(|terminal| terminal.as_str() == term) + }; + let deadline = tokio::time::Instant::now() + Duration::from_secs(90); + loop { + assert!( + tokio::time::Instant::now() < deadline, + "both sub-cap drains must complete within the window" + ); + // Completion is checked only when the socket goes QUIET: the + // shared capture vec is process-global, and scanning it per + // frame would starve this client's reads. + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(2)).await else { + if terminals.iter().all(|terminal| drain_completed(terminal)) { + break; + } + continue; + }; + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or("") + .to_string(); + let Some(pane) = pane_of(&term) else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let start = value["seqStart"].as_i64().unwrap_or(0); + let end = value["seqEnd"].as_i64().unwrap_or(0); + if value.to_string().len() > 128 { + saw_frame_over_budget = true; + } + max_seq[pane] = max_seq[pane].max(end); + delivered[pane].push((start, end)); + // Credit each page toward the pane's fixed target; the + // FINAL page (end >= target) is not credited — the credit + // that produced it already started the pane's drain. + if end > credited[pane] && end < targets[pane] { + credited[pane] = end; + credit(&mut paced, &terminals[pane], &arids[pane], end).await; + } else if end > credited[pane] { + credited[pane] = end; + } + } + Some("terminal.output.gap") => { + let reason = value.get("reason").and_then(|v| v.as_str()).unwrap_or(""); + assert_ne!( + reason, "queue_overflow", + "THE BOUND: the gated sub-cap drains must never spill the \ + connection's own queued pages: {value}" + ); + declared_gaps.push(( + pane, + value["fromSeq"].as_i64().unwrap_or(0), + value["toSeq"].as_i64().unwrap_or(0), + )); + } + _ => {} + } + } + + // Drain until quiet before the contiguity check (the completion events + // fire at admission; the final pages may still be in flight). + let quiet_drain = tokio::time::Instant::now() + Duration::from_secs(10); + let mut quiet_streak = 0u8; + while tokio::time::Instant::now() < quiet_drain && quiet_streak < 3 { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_millis(300)).await else { + quiet_streak += 1; + continue; + }; + quiet_streak = 0; + let term = value + .get("terminalId") + .and_then(|v| v.as_str()) + .unwrap_or("") + .to_string(); + let Some(pane) = pane_of(&term) else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => delivered[pane].push(( + value["seqStart"].as_i64().unwrap_or(0), + value["seqEnd"].as_i64().unwrap_or(0), + )), + Some("terminal.output.gap") => { + let reason = value.get("reason").and_then(|v| v.as_str()).unwrap_or(""); + assert_ne!( + reason, "queue_overflow", + "THE BOUND: no spill, even in the in-flight tail: {value}" + ); + declared_gaps.push(( + pane, + value["fromSeq"].as_i64().unwrap_or(0), + value["toSeq"].as_i64().unwrap_or(0), + )); + } + _ => {} + } + } + + assert!( + saw_frame_over_budget, + "the sub-cap sessions page atomic frames that exceed the 128-byte request \ + (the documented exception, real on the wire)" + ); + for pane in 0..PANES { + let start_ev = wait_for_restore_event( + &events, + &terminals[pane], + "ws.restore.paced_start", + "attach_request_id", + &arids[pane], + ) + .await + .expect("the sub-cap attach emits its paced_start event"); + assert_eq!( + start_ev.fields.get("page_budget").map(String::as_str), + Some("128"), + "the 128-byte replayPageBytes request is pane {pane}'s session budget" + ); + assert!( + max_seq[pane] > targets[pane], + "pane {pane}'s producer outran its attach target (the drain faced a real remainder)" + ); + // No undeclared holes: every delivered hole lies inside the pane's + // declared retention loss (the canary's contiguity contract). + let mut gap_ranges: Vec<(i64, i64)> = declared_gaps + .iter() + .filter(|(gap_pane, _, _)| *gap_pane == pane) + .map(|(_, from, to)| (*from, *to)) + .collect(); + gap_ranges.sort_unstable(); + let mut merged: Vec<(i64, i64)> = Vec::new(); + for (from, to) in gap_ranges { + match merged.last_mut() { + Some((_, last_to)) if from <= *last_to + 1 => { + *last_to = (*last_to).max(to); + } + _ => merged.push((from, to)), + } + } + let mut ranges = delivered[pane].clone(); + ranges.sort_unstable(); + let mut prev_end: Option = None; + for (start, end) in &ranges { + if *start <= 0 { + continue; + } + if let Some(prev) = prev_end { + if *start > prev + 1 { + let hole = (prev + 1, *start - 1); + let declared = merged + .iter() + .any(|(from, to)| hole.0 >= *from && hole.1 <= *to); + assert!( + declared, + "pane {pane} has an undeclared seq hole {hole:?} (declared loss {merged:?})" + ); + } + } + prev_end = Some(prev_end.map_or(*end, |p| p.max(*end))); + } + } +} +/// A producing drain's completion must be BOUNDED and COUNTABLE (the +/// round-2 recapture-loophole fix): the drain completes against a +/// sustained producer with a page count bounded by the ring capacity +/// (the drain target is captured once; no moving-head recapture can grow +/// the count), every delivered seq range is contiguous except across +/// explicitly declared bounds-carrying retention gaps, and production +/// demonstrably continued past the attach target at completion time. +#[tokio::test] +async fn producing_drain_completes_within_a_countable_page_bound_with_no_undeclared_jumps() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-bounded-drain").await; + + let mut primer = start_sustained_flood( + &url, + &mut driver, + &terminal_id, + "attach-bounded-drain-primer", + ) + .await; + primer + .send(WsMessage::Text( + serde_json::json!({ "type": "terminal.detach", "terminalId": terminal_id }).to_string(), + )) + .await + .expect("primer detaches"); + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-bounded-drain").await; + let target = ready["replayToSeq"].as_i64().expect("replayToSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + credited > 0, + "the producer staged content before the attach" + ); + + // Everything the session delivers, in receive order: (seqStart, seqEnd) + // and every explicitly declared gap interval. + let mut delivered: Vec<(i64, i64)> = Vec::new(); + let mut declared_gaps: Vec<(i64, i64)> = Vec::new(); + let mut produced_past_target = false; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + credit(&mut paced, &terminal_id, "attach-bounded-drain", credited).await; + let mut completed = false; + while tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + break; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + let start = value["seqStart"].as_i64().unwrap_or(0); + let end = value["seqEnd"].as_i64().unwrap_or(0); + delivered.push((start, end)); + produced_past_target |= end > target; + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-bounded-drain", end).await; + } + } + Some("terminal.output.gap") => { + assert!( + value["reason"] == "replay_window_exceeded" + || value["reason"] == "handoff_boundary_reached", + "the producing-drain fixture only ever declares retention gaps and \ + the round-4 fixed-boundary delivery gaps: {value}" + ); + declared_gaps.push(( + value["fromSeq"].as_i64().unwrap_or(0), + value["toSeq"].as_i64().unwrap_or(0), + )); + // A declared gap is also the negotiated continuation + // cursor: keep crediting the resumed front's pages. + } + _ => {} + } + if !completed { + completed = events.lock().unwrap().iter().any(|e| { + e.message == "ws.restore.paced_complete" + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id.as_str()) + }); + } + if completed && produced_past_target { + // Drain the in-flight handoff frames before asserting. + if delivered.last().map(|(_, e)| *e).unwrap_or(0) > target { + break; + } + } + } + + let complete = + wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_complete") + .await + .expect("the producing drain completes within the deadline"); + let pages: u64 = complete + .fields + .get("pages") + .and_then(|v| v.parse::().ok()) + .expect("pages recorded"); + // THE countable bound: the session's pages are bounded by the ring + // capacity scaled by the page budget (each page carries real content + // toward a FIXED target; retention expiry shrinks the drain range, + // never grows it). A recapturing chase grows the page count without + // bound while production runs. + let bound: u64 = 8 * (ring as u64 / PAGE_BUDGET as u64) + 64; + assert!( + pages <= bound, + "the session's page count is bounded by ring capacity ({pages} pages, bound {bound})" + ); + assert!( + produced_past_target, + "the fixture's producer demonstrably outran the attach target" + ); + + // No silent forward jumps: the delivered ranges tile contiguously + // from the first delivered seq, and every hole between consecutive + // ranges lies inside an explicitly declared gap interval. + let mut ranges = delivered.clone(); + ranges.sort_unstable(); + let mut prev_end: Option = None; + for (start, end) in &ranges { + if *start <= 0 { + continue; + } + if let Some(prev) = prev_end { + if *start > prev + 1 { + let hole = (prev + 1, *start - 1); + let declared = declared_gaps + .iter() + .any(|(from, to)| hole.0 >= *from && hole.1 <= *to); + assert!( + declared, + "undeclared seq hole {hole:?} between delivered ranges (gaps: {declared_gaps:?})" + ); + } + } + prev_end = Some(prev_end.map_or(*end, |p| p.max(*end))); + } + + // Cleanup: stop the flood. + send_input(&mut paced, &terminal_id, "\u{3}").await; +} + +/// Live output produced DURING the paced replay is delivered strictly after +/// the pages covering it, in seq order, with no loss or duplication — the +/// hard "no overtaking" invariant, end to end. +#[tokio::test] +async fn live_output_during_paced_replay_never_overtakes_the_pages() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-live-race").await; + flood_until_complete(&url, &mut driver, &terminal_id, 400).await; + + let marker1 = "FLOOD-DONE-MARKER"; + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-live").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let first_page_last = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + let last_seq = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + last_seq < head, + "the session starts mid-replay (head {head}, first page ends {last_seq}, frames {})", + page1.len() + ); + + // MORE live output while the replay is in flight (deferred — staged). + // A non-negotiated observer attaches (inline replay + direct live) and + // drives the flood to completion — the deterministic "it finished" signal. + let mut midflood = connect(&url).await; + hello(&mut midflood, false).await; + attach(&mut midflood, &terminal_id, "attach-midflood").await; + let marker2 = "FLOOD-DONE-MARKER"; + send_input(&mut midflood, &terminal_id, &flood_command(300, marker2)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let (mid_acc, _) = drain_until_marker(&mut midflood, marker2, deadline).await; + assert!( + mid_acc.contains(marker2), + "the mid flood completes on its own observer" + ); + drop(midflood); + + // Credit through to completion; the tail delivers the staged range + // spontaneously once the cursor reaches the target. + let mut received: Vec<(i64, String)> = page1 + .iter() + .map(|f| { + ( + f["seqStart"].as_i64().unwrap_or(0), + f["data"].as_str().unwrap_or("").to_string(), + ) + }) + .collect(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let mut credited = last_seq; + // Un-pause the session: the first page is the one outstanding page, so + // credit it before waiting for the next. + credit(&mut paced, &terminal_id, "attach-live", first_page_last).await; + while tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + break; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + received.push(( + value["seqStart"].as_i64().unwrap_or(0), + value["data"].as_str().unwrap_or("").to_string(), + )); + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-live", end).await; + } + let acc: String = received.iter().map(|(_, d)| d.as_str()).collect(); + if acc.contains(marker2) { + break; + } + } + } + let acc: String = received.iter().map(|(_, d)| d.as_str()).collect(); + assert!( + acc.contains(marker1) && acc.contains(marker2), + "the session delivers both floods" + ); + + // The hard invariant: strictly ascending seqs, no duplicates, and every + // delivered frame lies at or after the FIRST page's range (nothing the + // pages cover is re-delivered, and no live frame jumped the pages). + let seqs: Vec = received.iter().map(|(s, _)| *s).collect(); + let mut sorted = seqs.clone(); + sorted.sort_unstable(); + sorted.dedup(); + assert_eq!(seqs.len(), sorted.len(), "no duplicate seqStarts"); + assert_eq!(seqs, sorted, "delivery is strictly in seq order"); + assert!(seqs[0] >= 1, "the first page starts at the baseline+1"); + // Every seq from the attach baseline to the end is covered exactly once. + for expected in seqs[0]..=*seqs.last().unwrap() { + assert!( + seqs.contains(&expected), + "no seq holes: {expected} missing from {:?}", + &seqs[..seqs.len().min(20)] + ); + } +} + +/// Retention expiry MID-REPLAY: the negotiated `terminal.output.gap` with +/// reason `replay_window_exceeded`, the exact lost interval, the task-2 +/// bounds fields, and continuation from the new baseline through to the +/// session's target. +#[tokio::test] +async fn mid_replay_retention_expiry_emits_the_exact_negotiated_gap_and_continues() { + // Small ring so a withheld session loses its middle to eviction. + let ring = 12 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-expiry").await; + flood_until_complete(&url, &mut driver, &terminal_id, 100).await; + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-expiry").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let last_seq = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("the first page covers something"); + assert!(last_seq < head); + + // Evict the middle of the replay window while the client withholds. + // The evictor attaches (non-negotiated — inline replay + live) to + // observe its own flood's completion deterministically. + let mut evictor = connect(&url).await; + hello(&mut evictor, false).await; + attach(&mut evictor, &terminal_id, "attach-evictor").await; + let marker2 = "FLOOD-DONE-MARKER"; + send_input(&mut evictor, &terminal_id, &flood_command(400, marker2)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let (acc, _) = drain_until_marker(&mut evictor, marker2, deadline).await; + assert!(acc.contains(marker2), "the evicting flood completes"); + drop(evictor); + + // Credit: the next needed frames were evicted => the negotiated gap. + credit(&mut paced, &terminal_id, "attach-expiry", last_seq).await; + let gap = loop { + let value = next_json(&mut paced).await; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output.gap") => break value, + Some("terminal.output") => continue, // tail leakage of pre-eviction pages + other => panic!("expected the retention gap, got {other:?} ({value})"), + } + }; + assert_eq!( + gap["reason"], "replay_window_exceeded", + "the gap names retention loss: {gap}" + ); + assert_eq!( + gap["fromSeq"].as_i64(), + Some(last_seq + 1), + "the lost interval starts at the credited cursor+1: {gap}" + ); + let gap_oldest = gap["oldestRetainedSeq"] + .as_i64() + .expect("negotiated gap carries oldestRetainedSeq"); + let gap_head = gap["headSeq"] + .as_i64() + .expect("negotiated gap carries headSeq"); + assert_eq!( + gap["toSeq"].as_i64(), + Some(gap_oldest - 1), + "the lost interval ends just before the new ring front: {gap}" + ); + assert!(gap_head >= head, "the bounds are current: {gap}"); + assert_eq!(gap["attachRequestId"], "attach-expiry"); + + // Continuation from the new baseline: the next page starts at the ring + // front and the session still reaches its original target + the live + // tail (both markers). + let mut received: Vec = Vec::new(); + let mut data = String::new(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let mut credited = gap["toSeq"].as_i64().unwrap(); + while tokio::time::Instant::now() < deadline { + let value = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await; + let Some(value) = value else { break }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let seq = value["seqStart"].as_i64().unwrap_or(0); + assert!( + seq >= gap_oldest, + "continuation starts at the new baseline (ring front {gap_oldest}), got seq {seq}" + ); + received.push(seq); + data.push_str(value["data"].as_str().unwrap_or("")); + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-expiry", end).await; + } + if data.contains(marker2) { + break; + } + } + } + assert!( + data.contains(marker2), + "the session continues through the retained range to the live tail" + ); + let mut sorted = received.clone(); + sorted.sort_unstable(); + assert_eq!(received, sorted, "continuation pages ascend"); +} + +/// A re-attach supersedes the old session: the old generation's credit is +/// ignored (stale), the new generation's credit drives its own pages. +#[tokio::test] +async fn reattach_supersedes_the_old_paced_session() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-supersede").await; + flood_until_complete(&url, &mut driver, &terminal_id, 500).await; + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready1, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-gen-1").await; + let last1 = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("gen-1 first page"); + assert!(last1 < ready1["headSeq"].as_i64().unwrap()); + + // Re-attach (generation 2): a fresh ready + fresh first page; the + // attach.ready supersede also discards gen-1's un-leased frames. + let (ready2, page2) = paced_attach_first_page(&mut paced, &terminal_id, "attach-gen-2").await; + assert_eq!(ready2["attachRequestId"], "attach-gen-2"); + let last2 = page2 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("gen-2 first page"); + assert!( + page2.iter().all(|f| f["attachRequestId"] == "attach-gen-2"), + "gen-2's pages are stamped with gen-2's id" + ); + + // The OLD generation's credit is stale: nothing arrives. + credit(&mut paced, &terminal_id, "attach-gen-1", last1).await; + let stale = next_json_or_timeout(&mut paced, Duration::from_millis(1200)).await; + assert!( + stale.is_none() + || stale.as_ref().unwrap().get("type").and_then(|v| v.as_str()) + != Some("terminal.output"), + "a stale-generation credit must be ignored, got {stale:?}" + ); + + // The NEW generation's credit drives its own session. + credit(&mut paced, &terminal_id, "attach-gen-2", last2).await; + let next = next_json(&mut paced).await; + assert_eq!(next["type"], "terminal.output"); + assert_eq!(next["attachRequestId"], "attach-gen-2"); + assert!( + next["seqEnd"].as_i64().unwrap_or(0) > last2, + "gen-2's pages continue from its own cursor" + ); +} + +/// The binding supersede rule covers EVERY successful re-attach, not just +/// the paced one: an arid-less LEGACY re-attach (the negotiated fallback +/// shape) must also cancel the connection's previous paced session for the +/// terminal — a credit for the superseded generation must produce NOTHING +/// (no `terminal.output`, no `terminal.output.batch`), and the connection +/// must keep working afterwards. +#[tokio::test] +async fn legacy_reattach_cancels_the_stale_paced_session() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-legacy-supersede").await; + flood_until_complete(&url, &mut driver, &terminal_id, 400).await; + + // Generation 1: the paced attach starts a session whose first page is + // a bounded prefix (the session is ACTIVE, mid-replay). + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready1, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-gen-1").await; + let last1 = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("gen-1 first page"); + assert!( + last1 < ready1["headSeq"].as_i64().unwrap(), + "the session is mid-replay (supersede-able)" + ); + + // The arid-less LEGACY re-attach: the whole window replays inline. + attach_without_arid(&mut paced, &terminal_id).await; + let (ready2, outputs) = legacy_attach_inline_replay(&mut paced, &terminal_id).await; + assert!( + ready2 + .get("attachRequestId") + .and_then(|v| v.as_str()) + .is_none(), + "the re-attach carried no attachRequestId: {ready2}" + ); + let head = ready2["headSeq"].as_i64().unwrap(); + let max_seq = outputs + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert_eq!( + max_seq, head, + "the legacy re-attach replays the whole window inline" + ); + + // The superseded generation's credit (its arid + in-window consumedSeq): + // it must be ignored — no output frame of ANY kind may be produced. + credit(&mut paced, &terminal_id, "attach-gen-1", last1).await; + let stale = next_json_or_timeout(&mut paced, Duration::from_millis(1500)).await; + let is_output_frame = |v: &serde_json::Value| { + matches!( + v.get("type").and_then(|t| t.as_str()), + Some("terminal.output") | Some("terminal.output.batch") + ) + }; + assert!( + stale.as_ref().map(|v| !is_output_frame(v)).unwrap_or(true), + "a credit for the session superseded by the legacy re-attach must \ + produce nothing, got {stale:?}" + ); + + // The connection is not wedged by the cancel: live output keeps + // flowing to the re-attached (legacy) subscriber. + let marker = "FLOOD-DONE-MARKER"; + send_input(&mut driver, &terminal_id, &flood_command(30, marker)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let (acc, _) = drain_until_marker(&mut paced, marker, deadline).await; + assert!( + acc.contains(marker), + "live output flows normally after the superseded credit" + ); +} + +/// A credit on a NON-NEGOTIATED connection is inert: no pacing machinery +/// engages, inline delivery continues to work exactly as before. +#[tokio::test] +async fn credit_on_a_non_negotiated_connection_is_inert() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-plain-credit").await; + flood_until_complete(&url, &mut driver, &terminal_id, 100).await; + + let mut plain = connect(&url).await; + hello(&mut plain, false).await; + let (ready, outputs) = paced_attach_first_page(&mut plain, &terminal_id, "attach-plain").await; + // Non-negotiated: the FULL inline replay arrived immediately. + let head = ready["headSeq"].as_i64().unwrap(); + let max_seq = outputs + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert_eq!( + max_seq, head, + "a non-negotiated attach gets the whole replay inline" + ); + assert!( + ready.get("oldestRetainedSeq").is_none(), + "no contract fields for a plain connection" + ); + + // The inert credit: live output keeps flowing inline afterwards. + credit(&mut plain, &terminal_id, "attach-plain", max_seq).await; + let marker = "FLOOD-DONE-MARKER"; + send_input(&mut driver, &terminal_id, &flood_command(30, marker)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let (acc, _) = drain_until_marker(&mut plain, marker, deadline).await; + assert!( + acc.contains(marker), + "live output flows normally after an inert credit" + ); +} + +/// A disconnect mid-replay leaves the terminal RUNNING and clean: a +/// subsequent (non-negotiated) attach gets the full replay and live output — +/// no leak, no wedge. +#[tokio::test] +async fn disconnect_mid_replay_leaves_the_terminal_running_and_reattachable() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-disconnect").await; + flood_until_complete(&url, &mut driver, &terminal_id, 400).await; + + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-doomed").await; + let last_seq = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert!( + last_seq < ready["headSeq"].as_i64().unwrap(), + "mid-replay before the drop" + ); + drop(paced); // the socket dies mid-replay (no detach, no credit) + + tokio::time::sleep(Duration::from_millis(300)).await; + + // A fresh non-negotiated connection attaches and gets the full replay. + let mut fresh = connect(&url).await; + hello(&mut fresh, false).await; + let (ready2, outputs) = paced_attach_first_page(&mut fresh, &terminal_id, "attach-after").await; + let head = ready2["headSeq"].as_i64().unwrap(); + let max_seq = outputs + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .unwrap_or(0); + assert_eq!( + max_seq, head, + "the full replay is intact after the dropped paced session" + ); + + // And live output still flows. + let marker = "FLOOD-DONE-MARKER"; + send_input(&mut driver, &terminal_id, &flood_command(30, marker)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let (acc, _) = drain_until_marker(&mut fresh, marker, deadline).await; + assert!( + acc.contains(marker), + "live output flows after the re-attach" + ); +} + +/// The `ws.restore.*` observability contract, pinned end-to-end on the real +/// dispatch: `ws.restore.paced_start` carries its identifiers/measurements +/// (including `maxReplayBytes`), `ws.restore.paced_complete` closes the +/// session, and all FOUR `ws.restore.credit` verdicts (accepted / +/// stale_generation / beyond_window / non_negotiated) are emitted with their +/// status field — and NO event ever carries terminal CONTENT (identifiers +/// and measurements only). +#[tokio::test] +async fn restore_observability_events_are_emitted_content_free() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-observability").await; + flood_until_complete(&url, &mut driver, &terminal_id, 100).await; + + // The negotiated attach carries a maxReplayBytes request (the TERM-07 + // seam) — it must ride the paced_start event. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + paced + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.attach", + "terminalId": terminal_id, + "intent": "viewport_hydrate", + "cols": 80, + "rows": 24, + "attachRequestId": "attach-ev", + "sinceSeq": 0, + "maxReplayBytes": 262144, + }) + .to_string(), + )) + .await + .expect("send attach with maxReplayBytes"); + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-ev").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let last_seq = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("first page"); + let page_bytes: u64 = page1.iter().map(|f| f.to_string().len() as u64).sum(); + + // paced_start: identifiers + measurements, maxReplayBytes included. + let start_ev = + wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_start") + .await + .expect("ws.restore.paced_start is emitted on the negotiated attach"); + assert_eq!( + start_ev.fields.get("attach_request_id").map(String::as_str), + Some("attach-ev") + ); + assert_eq!( + start_ev.fields.get("requested_since").map(String::as_str), + Some("0") + ); + assert_eq!( + start_ev.fields.get("effective_since").map(String::as_str), + Some("0") + ); + assert_eq!( + start_ev.fields.get("target").map(String::as_str), + Some(head.to_string().as_str()), + "target is the attach-time head" + ); + assert_eq!( + start_ev.fields.get("max_replay_bytes").map(String::as_str), + Some("Some(262144)"), + "the TERM-07 seam value rides the event (Debug of Option)" + ); + assert_eq!( + start_ev.fields.get("page_bytes").map(String::as_str), + Some(page_bytes.to_string().as_str()), + "page_bytes is the first page's real serialized size" + ); + + // beyond_window: a consumedSeq past the last-sent page's end is ignored. + credit(&mut paced, &terminal_id, "attach-ev", last_seq + 100_000).await; + let beyond = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "beyond_window", + ) + .await + .expect("the beyond-window credit is observed"); + assert_eq!( + beyond.fields.get("terminal_id").map(String::as_str), + Some(terminal_id.as_str()) + ); + assert_eq!( + beyond + .fields + .get("consumed_seq") + .and_then(|v| v.parse::().ok()), + Some(last_seq + 100_000) + ); + + // stale_generation: a credit for an attachRequestId no active session + // holds is a stale generation. + credit(&mut paced, &terminal_id, "attach-bogus", last_seq).await; + let stale = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "stale_generation", + ) + .await + .expect("the stale-generation credit is observed"); + assert_eq!( + stale.fields.get("terminal_id").map(String::as_str), + Some(terminal_id.as_str()) + ); + + // accepted: a valid credit produces the next page... + credit(&mut paced, &terminal_id, "attach-ev", last_seq).await; + let accepted = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "accepted", + ) + .await + .expect("the valid credit is observed as accepted"); + assert_eq!( + accepted + .fields + .get("consumed_seq") + .and_then(|v| v.parse::().ok()), + Some(last_seq) + ); + // ...and the page actually arrives (the event describes real behavior). + let next = next_json(&mut paced).await; + assert_eq!(next["type"], "terminal.output"); + + // Drive the session to completion; the tail drains un-credited and the + // registry's atomic clear completes the session. + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let mut credited = next["seqEnd"].as_i64().unwrap_or(last_seq); + credit(&mut paced, &terminal_id, "attach-ev", credited).await; + while tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + break; + }; + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output") { + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-ev", end).await; + } + } + } + let complete = + wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_complete") + .await + .expect("ws.restore.paced_complete closes the session"); + assert_eq!( + complete.fields.get("attach_request_id").map(String::as_str), + Some("attach-ev") + ); + assert_eq!( + complete.fields.get("last_seq").map(String::as_str), + Some(credited.to_string().as_str()), + "last_seq is the session's final cursor" + ); + assert!( + complete + .fields + .get("pages") + .and_then(|v| v.parse::().ok()) + .is_some_and(|pages| pages >= 2), + "pages counts every page the session produced" + ); + + // non_negotiated: a credit from a connection that never negotiated the + // paced capability is inert and observed as such. + credit(&mut driver, &terminal_id, "attach-ev", 0).await; + let non_negotiated = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "non_negotiated", + ) + .await + .expect("the non-negotiated connection's credit is observed as inert"); + assert_eq!( + non_negotiated.fields.get("terminal_id").map(String::as_str), + Some(terminal_id.as_str()) + ); + + // Identifiers/measurements only: NO terminal content ever leaks into a + // ws.restore.* event for this terminal (the flood payload and its + // marker must be absent from every event field). + let captured = events.lock().unwrap(); + for event in captured.iter().filter(|e| { + e.message.starts_with("ws.restore.") + && e.fields.get("terminal_id").map(String::as_str) == Some(terminal_id.as_str()) + }) { + for (name, value) in &event.fields { + assert!( + !value.contains("STREAMDATA") && !value.contains("FLOOD-DONE-MARKER"), + "terminal content leaked into ws.restore.{name}={value}" + ); + } + } +} + +// ── Mixed-version compatibility matrix (responsive-terminal-restore, +// task-008): old client → new server on the real socket ────────────────── + +/// Old client → new server, ATTACH on large scrollback (matrix cell 1): a +/// non-negotiated attach sees the FULL inline replay in one unprompted +/// burst — far beyond one paced page budget, so pacing provably never +/// engaged — with NO new ready fields, NO gap frames, contiguous coverage, +/// and a raw credit send that is inert (observed as +/// `ws.restore.credit status=non_negotiated`, producing nothing). +#[tokio::test] +async fn non_negotiated_attach_on_large_scrollback_stays_legacy_inline_and_credits_are_inert() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-legacy-large").await; + // ~64KB of scrollback: many 4KB pages had pacing (wrongly) engaged. + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + let mut plain = connect(&url).await; + hello(&mut plain, false).await; + let (ready, outputs, gaps) = + attach_burst_collecting_gaps(&mut plain, &terminal_id, "attach-legacy-large").await; + + // NO new ready fields for the non-negotiated connection. + assert!( + ready.get("oldestRetainedSeq").is_none(), + "an old client's attach.ready must not gain contract fields: {ready}" + ); + assert!( + ready.get("replayResetReason").is_none(), + "an old client's attach.ready must not carry a reset reason: {ready}" + ); + let head = ready["headSeq"].as_i64().expect("headSeq"); + + // The whole window arrived unprompted: contiguous coverage to the head, + // and the burst is far beyond one page budget (no pages, no credit gate). + let burst_bytes: usize = outputs.iter().map(|f| f.to_string().len()).sum(); + assert!( + burst_bytes as i64 > PAGE_BUDGET, + "the unprompted burst ({burst_bytes}B) must exceed one page budget \ + ({PAGE_BUDGET}B) — a paced first page would have been bounded" + ); + let seqs = covered_seqs(&outputs); + assert_eq!( + seqs.iter().min(), + Some(&1), + "the inline replay starts at the window baseline" + ); + assert_eq!( + seqs.iter().max(), + Some(&head), + "the inline replay reaches the head ({head})" + ); + let mut contiguous: Vec = seqs.iter().copied().collect(); + contiguous.dedup(); + assert_eq!( + contiguous.len() as i64, + head, + "the inline replay is contiguous with no holes" + ); + assert!( + gaps.is_empty(), + "an old client's attach burst must contain no gap frames: {gaps:?}" + ); + + // The raw credit send: inert. Observed as the non_negotiated verdict... + credit(&mut plain, &terminal_id, "attach-legacy-large", head).await; + let _inert = wait_for_restore_event( + &events, + &terminal_id, + "ws.restore.credit", + "status", + "non_negotiated", + ) + .await + .expect("a non-negotiated connection's credit must be classified non_negotiated"); + + // ...and it produces nothing. + let after = next_json_or_timeout(&mut plain, Duration::from_millis(1200)).await; + assert!( + after + .as_ref() + .map(|v| v.get("type").and_then(|t| t.as_str()) != Some("terminal.output")) + .unwrap_or(true), + "an inert credit must produce no output, got {after:?}" + ); + + // Live output keeps flowing after the inert credit (the connection is + // not wedged by the refused pacing machinery). + let marker = "FLOOD-DONE-MARKER"; + send_input(&mut driver, &terminal_id, &flood_command(30, marker)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let (acc, _) = drain_until_marker(&mut plain, marker, deadline).await; + assert!( + acc.contains(marker), + "live output flows normally after an inert credit" + ); +} + +/// Old client + retention loss → NO new gap (matrix cell 2): a +/// non-negotiated attach whose `sinceSeq` predates the retained history +/// receives the retained tail SILENTLY — today's behavior, never the +/// negotiated `replay_window_exceeded` gap — on a real server with an +/// evicted ring. +#[tokio::test] +async fn non_negotiated_attach_after_retention_loss_gets_the_retained_tail_silently() { + let ring = 12 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-legacy-evict").await; + flood_until_complete(&url, &mut driver, &terminal_id, 100).await; + + // Evict the ring front: a non-negotiated evictor drives a bigger flood + // to completion (its own inline replay + live tail observe the marker). + let mut evictor = connect(&url).await; + hello(&mut evictor, false).await; + attach(&mut evictor, &terminal_id, "attach-evictor-legacy").await; + let marker2 = "FLOOD-DONE-MARKER"; + send_input(&mut driver, &terminal_id, &flood_command(400, marker2)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let (acc, _) = drain_until_marker(&mut evictor, marker2, deadline).await; + assert!(acc.contains(marker2), "the evicting flood completes"); + drop(evictor); + + // The old client attaches with sinceSeq 0 — predating the retained ring. + let mut plain = connect(&url).await; + hello(&mut plain, false).await; + let (ready, outputs, gaps) = + attach_burst_collecting_gaps(&mut plain, &terminal_id, "attach-legacy-evict").await; + + // THE matrix pin: no `replay_window_exceeded` may reach an old client — + // the retention loss is reported the way today's server does: silently. + assert!( + gaps.is_empty(), + "an old client must never receive the newly introduced retention gap: {gaps:?}" + ); + assert!( + ready.get("oldestRetainedSeq").is_none(), + "an old client's attach.ready must not gain contract fields: {ready}" + ); + let head = ready["headSeq"].as_i64().expect("headSeq"); + let replay_from = ready["replayFromSeq"].as_i64().expect("replayFromSeq"); + assert!( + replay_from > 1, + "the fixture must have evicted the ring front: {ready}" + ); + + // The retained tail arrives silently: exactly [replay_from, head], + // starting at the ring front, contiguous, no phantom pre-eviction bytes. + let seqs = covered_seqs(&outputs); + assert!(!seqs.is_empty(), "there is a retained tail to deliver"); + assert_eq!( + seqs.iter().min(), + Some(&replay_from), + "the silent tail starts at the ring front: {ready}" + ); + assert_eq!( + seqs.iter().max(), + Some(&head), + "the silent tail reaches the head" + ); + let mut contiguous: Vec = seqs.iter().copied().collect(); + contiguous.dedup(); + assert_eq!( + contiguous.len() as i64, + head - replay_from + 1, + "the silent tail is contiguous with no holes" + ); + + // Live output continues after the silent retained tail (no stall). + let marker3 = "FLOOD-DONE-MARKER"; + send_input(&mut driver, &terminal_id, &flood_command(30, marker3)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(20); + let (acc, _) = drain_until_marker(&mut plain, marker3, deadline).await; + assert!( + acc.contains(marker3), + "live output flows normally after the silent retained tail" + ); +} + +/// Mixed connections on ONE terminal (matrix cell 4): a negotiated client and +/// a non-negotiated client attached to the SAME terminal simultaneously — +/// the negotiated one gets paced pages (credit-gated), the non-negotiated +/// one gets the legacy inline replay; no cross-talk in either direction, and +/// both converge to the SAME frame content (identical seq coverage and +/// identical concatenated data — equality modulo pacing/batching shape). +#[tokio::test] +async fn mixed_negotiated_and_legacy_attachments_converge_without_cross_talk() { + let events = global_capture(); + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-mixed").await; + flood_until_complete(&url, &mut driver, &terminal_id, 400).await; + + // The negotiated client attaches first: one bounded page, mid-replay. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (paced_ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-mix-paced").await; + let head = paced_ready["headSeq"].as_i64().expect("headSeq"); + let last1 = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("the first page covers something"); + assert!( + last1 < head, + "the negotiated session is mid-replay (head {head})" + ); + + // The legacy twin attaches to the SAME terminal while the paced session + // is mid-replay: the FULL window arrives inline, immediately. + let mut legacy = connect(&url).await; + hello(&mut legacy, false).await; + let (legacy_ready, legacy_outputs, legacy_gaps) = + attach_burst_collecting_gaps(&mut legacy, &terminal_id, "attach-mix-legacy").await; + assert!( + legacy_gaps.is_empty(), + "the legacy twin must see no gap frames: {legacy_gaps:?}" + ); + assert!( + legacy_ready.get("oldestRetainedSeq").is_none(), + "the legacy twin's ready must stay pre-contract: {legacy_ready}" + ); + let legacy_seqs = covered_seqs(&legacy_outputs); + assert_eq!( + legacy_seqs.iter().max(), + Some(&head), + "the legacy twin gets the whole window inline" + ); + + // No cross-talk, direction 1: the other connection's attach must NOT + // cancel the negotiated session — its credit still produces its page... + credit(&mut paced, &terminal_id, "attach-mix-paced", last1).await; + let next = next_json(&mut paced).await; + assert_eq!( + next["type"], "terminal.output", + "the negotiated session survives the legacy attach: {next}" + ); + assert_eq!(next["attachRequestId"], "attach-mix-paced"); + + // No cross-talk, direction 2: while the negotiated session withholds, + // the legacy attach must not leak pages to it (the next unprompted + // frame on the negotiated side is only the one its own credit bought). + let mut paced_frames = page1.clone(); + let next_end = next["seqEnd"].as_i64().unwrap_or(last1); + paced_frames.push(next); + let mut credited = next_end; + // Credit every consumed frame's end — a page may span several frames + // (or one atomic over-budget frame per page), and the NEXT page is + // produced only on a credit inside the delivered window. + credit(&mut paced, &terminal_id, "attach-mix-paced", credited).await; + let mut saw_head = credited >= head; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + while tokio::time::Instant::now() < deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_secs(5)).await else { + assert!( + saw_head, + "the paced replay stalled before its target (head {head}, credited {credited})" + ); + break; + }; + if value.get("type").and_then(|v| v.as_str()) != Some("terminal.output") { + continue; + } + assert_eq!( + value["attachRequestId"], "attach-mix-paced", + "only the negotiated session's own pages arrive: {value}" + ); + let end = value["seqEnd"].as_i64().unwrap_or(0); + saw_head |= end >= head; + paced_frames.push(value); + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-mix-paced", end).await; + } + } + let complete = + wait_for_restore_event_of_terminal(&events, &terminal_id, "ws.restore.paced_complete") + .await + .expect("the negotiated session reaches paced_complete"); + assert_eq!( + complete.fields.get("attach_request_id").map(String::as_str), + Some("attach-mix-paced") + ); + + // Convergence: identical seq coverage and identical concatenated data — + // the paced side (pages, credit-gated) and the legacy side (one inline + // burst) delivered the same terminal content, modulo pacing shape. + let paced_seqs = covered_seqs(&paced_frames); + assert_eq!(paced_seqs, legacy_seqs, "both sides cover the same seqs"); + assert_eq!( + paced_seqs.iter().min(), + Some(&1), + "both sides cover from the window baseline" + ); + assert_eq!( + paced_seqs.iter().max(), + Some(&head), + "both sides reach the head" + ); + let paced_data = concatenated_data(&paced_frames); + let legacy_data = concatenated_data(&legacy_outputs); + assert_eq!( + paced_data, legacy_data, + "both sides delivered byte-identical content" + ); +} + +/// E2R1 finding 1 (a) + finding 4: a natural exit mid-restore sequences +/// behind the session's normal CREDITED completion — and the ordering is +/// asserted over the FULL frame stream. A CREDITING client drives the +/// whole deferred range one page per credit in ascending order, and +/// terminal.exit is the LAST frame: exactly one exit, every output frame +/// (and the final marker produced past the attach target) precedes it, +/// and NOTHING follows it — the test keeps reading for a deterministic +/// quiet window after the exit and any frame that arrived would fail the +/// asserts. (The old test broke on the FIRST exit, so its output arm +/// could never observe a post-exit frame — the stated claim was +/// unproven. The pre-fix behavior this replaces: the exit-drain removed +/// the still-credited session and dumped its whole window uncredited.) +#[tokio::test] +async fn natural_exit_mid_restore_sequences_final_output_before_terminal_exit() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-exit-mid-restore").await; + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + // Negotiate a paced attach and do NOT credit yet: the session sits + // in the credited phase, its deferral armed, with the replay window + // still open when the shell exits. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-exit").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("first page frames"); + assert!( + credited < head, + "the session is mid-restore when the shell exits (page end {credited}, head {head})" + ); + + // The shell produces one FINAL marker line and exits naturally. The + // octal escapes keep the echoed command from printing the marker early. + send_input( + &mut paced, + &terminal_id, + "printf '\\106\\111\\116\\101\\114\\055\\115\\101\\122\\113\\105\\122\\012'; exit\n", + ) + .await; + + // The client already consumed the first page before the exit — its + // continuation credit is due NOW. From here the flow is pure + // credit-pacing: each credit grants the next page through the whole + // deferred range, and the exit sequences behind the credited + // completion. + credit(&mut paced, &terminal_id, "attach-exit", credited).await; + + // Collect EVERY frame until a deterministic quiet window closes AFTER + // the exit (finding 4: the ordering is asserted over the FULL frame + // stream — the old test broke on the first exit, so its output arm + // could never observe a post-exit frame). + let mut saw_exit = false; + let mut exits = 0usize; + let mut acc = String::new(); + let mut seqs: Vec = Vec::new(); + let mut output_frames = 0usize; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + while tokio::time::Instant::now() < deadline { + let window = if saw_exit { + Duration::from_millis(750) + } else { + Duration::from_secs(5) + }; + let Some(value) = next_json_or_timeout(&mut paced, window).await else { + if saw_exit { + break; // quiet AFTER the exit: the stream is provably complete + } + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + assert!(!saw_exit, "no output may follow terminal.exit: {value}"); + seqs.push(value["seqStart"].as_i64().unwrap_or(0)); + if let Some(data) = value.get("data").and_then(|v| v.as_str()) { + acc.push_str(data); + } + output_frames += 1; + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-exit", end).await; + } + } + Some("terminal.exit") => { + saw_exit = true; + exits += 1; + assert_eq!( + exits, 1, + "exactly one terminal.exit may ever arrive: {value}" + ); + } + _ => {} + } + } + assert!( + saw_exit, + "terminal.exit must arrive — the staged exit delivers at the session's \ + CREDITED completion (pages flow only on credits)" + ); + assert!( + acc.contains("FINAL-MARKER"), + "the deferred final output is delivered before the exit (got {output_frames} frames)" + ); + assert!( + output_frames > 1, + "the deferred range paged before the exit ({output_frames} frames)" + ); + let mut sorted = seqs.clone(); + sorted.sort_unstable(); + assert_eq!( + seqs, sorted, + "every delivered frame precedes the exit in ascending seq order" + ); + // exit is the LAST frame: the loop only breaks on a quiet window + // AFTER the exit — any frame that arrived in it failed the asserts + // above, so reaching here proves nothing followed the exit. +} + +/// E2R1 finding 1 (b): the pacing contract PINNED at the exit boundary. +/// A client that WITHHOLDS credits after a natural exit mid-restore +/// receives NO further pages and NO exit — asserted as absence over a +/// deterministic window — and the flow RESUMES on the next credit (the +/// deferred final output pages only on credits, and the exit sequences +/// behind the session's credited completion). The pre-fix behavior this +/// guards against: the exit-drain removed the still-credited session and +/// dumped its whole window (plus the exit) uncredited, as fast as the +/// socket accepted it. +#[tokio::test] +async fn natural_exit_mid_restore_withholding_client_gets_no_pages_and_no_exit_until_it_credits() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-exit-withhold").await; + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + // Negotiate a paced attach and do NOT credit: the session sits in + // the credited phase, its deferral armed, mid-restore. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-exit-hold").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("first page frames"); + assert!( + credited < head, + "the session is mid-restore when the shell exits" + ); + + // The shell produces one FINAL marker line and exits naturally. + send_input( + &mut paced, + &terminal_id, + "printf '\\106\\111\\116\\101\\114\\055\\115\\101\\122\\113\\105\\122\\012'; exit\n", + ) + .await; + + // THE WITHHOLD: no credits. Over a deterministic window, NOTHING may + // arrive — no page, no gap, no exit. (Give the exit a moment to + // stage, then hold: the window covers both.) + let hold_deadline = tokio::time::Instant::now() + Duration::from_millis(2_000); + while tokio::time::Instant::now() < hold_deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_millis(250)).await else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => panic!( + "no page may flow while the client withholds credits \ + (the deferred final output pages ONLY on credits): {value}" + ), + Some("terminal.output.gap") => { + panic!("no gap may flow while the client withholds credits: {value}") + } + Some("terminal.exit") => panic!( + "no exit may arrive while pages remain undelivered — the exit \ + sequences behind the session's CREDITED completion: {value}" + ), + _ => {} + } + } + + // RESUME ON CREDIT: the first credit pages the deferred range, and + // the flow converges — the marker, then the exit LAST (the same + // full-stream ordering as the crediting case). + credit(&mut paced, &terminal_id, "attach-exit-hold", credited).await; + let mut saw_exit = false; + let mut exits = 0usize; + let mut acc = String::new(); + let mut seqs: Vec = Vec::new(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + while tokio::time::Instant::now() < deadline { + let window = if saw_exit { + Duration::from_millis(750) + } else { + Duration::from_secs(5) + }; + let Some(value) = next_json_or_timeout(&mut paced, window).await else { + if saw_exit { + break; + } + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + assert!(!saw_exit, "no output may follow terminal.exit: {value}"); + seqs.push(value["seqStart"].as_i64().unwrap_or(0)); + if let Some(data) = value.get("data").and_then(|v| v.as_str()) { + acc.push_str(data); + } + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > credited { + credited = end; + credit(&mut paced, &terminal_id, "attach-exit-hold", end).await; + } + } + Some("terminal.exit") => { + saw_exit = true; + exits += 1; + assert_eq!(exits, 1, "exactly one terminal.exit: {value}"); + } + _ => {} + } + } + assert!( + saw_exit, + "the flow resumes on credit and the exit arrives last" + ); + assert!( + acc.contains("FINAL-MARKER"), + "the deferred final output resumed on credit and was delivered" + ); + let mut sorted = seqs.clone(); + sorted.sort_unstable(); + assert_eq!(seqs, sorted, "the resumed pages ascend"); +} + +/// E2R1 finding 1 (c): retention expiry while the exit-staged session +/// WAITS on credits. The ring's front evicts the credited window's next +/// needed frames while the client withholds; the client's next credit +/// reports the EXACT bounds-carrying gap, the continuation pages only on +/// credits through the retained window, and the exit is the LAST frame. +#[tokio::test] +async fn natural_exit_mid_restore_retention_expiry_reports_the_exact_gap_then_exit() { + // Small ring so a withheld session loses its middle to eviction. + let ring = 12 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-exit-expiry").await; + flood_until_complete(&url, &mut driver, &terminal_id, 100).await; + + // Negotiate a paced attach and do NOT credit: the session sits in + // the credited phase, mid-restore. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = paced_attach_first_page(&mut paced, &terminal_id, "attach-exit-exp").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("first page frames"); + assert!( + credited < head, + "the session is mid-restore when the ring churns" + ); + + // Evict the middle of the replay window while the client withholds + // (the evictor attaches non-negotiated and observes its own flood's + // completion deterministically). + let mut evictor = connect(&url).await; + hello(&mut evictor, false).await; + attach(&mut evictor, &terminal_id, "attach-exit-evictor").await; + let marker2 = "FLOOD-DONE-MARKER"; + send_input(&mut evictor, &terminal_id, &flood_command(400, marker2)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let (evictor_acc, _) = drain_until_marker(&mut evictor, marker2, deadline).await; + assert!( + evictor_acc.contains(marker2), + "the evicting flood completes" + ); + drop(evictor); + + // The shell now produces one FINAL marker line and exits naturally, + // staging the exit behind the still-armed session. The exit must + // NOT deliver anything while the client withholds. + send_input( + &mut paced, + &terminal_id, + "printf '\\106\\111\\116\\101\\114\\062\\055\\115\\101\\122\\113\\105\\122\\012'; exit\n", + ) + .await; + let hold_deadline = tokio::time::Instant::now() + Duration::from_millis(1_500); + while tokio::time::Instant::now() < hold_deadline { + let Some(value) = next_json_or_timeout(&mut paced, Duration::from_millis(250)).await else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") | Some("terminal.output.gap") | Some("terminal.exit") => { + panic!( + "nothing may flow while the client withholds credits — the exact \ + gap reports only on the next credit: {value}" + ) + } + _ => {} + } + } + + // THE CREDIT: the next needed frames were evicted, so the drive's + // FIRST product is the EXACT negotiated gap — never a silent + // forward jump — and the continuation then pages from the ring + // front on credits, to the marker and the exit LAST. + credit(&mut paced, &terminal_id, "attach-exit-exp", credited).await; + let gap = next_json_or_timeout(&mut paced, Duration::from_secs(5)) + .await + .expect("the credit reports the retention gap"); + assert_eq!( + gap.get("type").and_then(|v| v.as_str()), + Some("terminal.output.gap"), + "the credit's first product is the exact retention gap: {gap}" + ); + assert_eq!( + gap["reason"], "replay_window_exceeded", + "the gap names retention loss: {gap}" + ); + assert_eq!( + gap["fromSeq"].as_i64(), + Some(credited + 1), + "the lost interval starts at the credited cursor+1: {gap}" + ); + let gap_oldest = gap["oldestRetainedSeq"] + .as_i64() + .expect("oldestRetainedSeq"); + assert_eq!( + gap["toSeq"].as_i64(), + Some(gap_oldest - 1), + "the lost interval ends just before the new ring front: {gap}" + ); + assert_eq!(gap["attachRequestId"], "attach-exit-exp"); + + // Continuation on credits: the pages start at the ring front, the + // final marker is delivered, and the exit is the LAST frame (the + // full-stream ordering). + let mut saw_exit = false; + let mut exits = 0usize; + let mut acc = String::new(); + let mut seqs: Vec = Vec::new(); + let mut continued = credited.max(gap_oldest - 1); + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + while tokio::time::Instant::now() < deadline { + let window = if saw_exit { + Duration::from_millis(750) + } else { + Duration::from_secs(5) + }; + let Some(value) = next_json_or_timeout(&mut paced, window).await else { + if saw_exit { + break; + } + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + assert!(!saw_exit, "no output may follow terminal.exit: {value}"); + let start = value["seqStart"].as_i64().unwrap_or(0); + assert!( + start >= gap_oldest, + "the continuation starts at the new baseline (ring front \ + {gap_oldest}), got seq {start}" + ); + seqs.push(start); + if let Some(data) = value.get("data").and_then(|v| v.as_str()) { + acc.push_str(data); + } + let end = value["seqEnd"].as_i64().unwrap_or(0); + if end > continued { + continued = end; + credit(&mut paced, &terminal_id, "attach-exit-exp", end).await; + } + } + Some("terminal.exit") => { + saw_exit = true; + exits += 1; + assert_eq!(exits, 1, "exactly one terminal.exit: {value}"); + } + _ => {} + } + } + assert!( + saw_exit, + "the exact gap is followed by the continuation and the exit arrives last" + ); + assert!( + acc.contains("FINAL2-MARKER"), + "the retained window pages through the final marker on credits" + ); + let mut sorted = seqs.clone(); + sorted.sort_unstable(); + assert_eq!(seqs, sorted, "the continuation pages ascend"); +} + +/// E2R2 finding (the exit-arming race), test-side page reader: read the +/// ONE page a credit grants (its frames arrive back-to-back; a quiet gap +/// closes the page — nothing further flows without a credit). Carries +/// INVARIANT 2's read-side guard: a terminal.exit observed while the page +/// that reaches the armed exit head is still uncredited IS the +/// premature-exit violation under test — panic with the frame. Gaps are +/// collected, not fatal (the retention fixture's first credit reports +/// the exact gap before its continuation frames). +async fn read_credited_page( + ws: &mut WsClient, + quiet: Duration, +) -> (String, i64, Vec) { + let mut acc = String::new(); + let mut max_seq_end = 0i64; + let mut gaps = Vec::new(); + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + assert!( + tokio::time::Instant::now() < deadline, + "a credit must grant its page" + ); + let window = if acc.is_empty() && gaps.is_empty() { + Duration::from_secs(5) + } else { + quiet + }; + match next_json_or_timeout(ws, window).await { + Some(value) => match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => { + if let Some(data) = value.get("data").and_then(|v| v.as_str()) { + acc.push_str(data); + } + max_seq_end = max_seq_end.max(value["seqEnd"].as_i64().unwrap_or(0)); + } + Some("terminal.output.gap") => gaps.push(value), + Some("terminal.exit") => panic!( + "invariant 2 violated: terminal.exit delivered while the page \ + reaching the armed exit head was still uncredited (the exit \ + must ride the CREDIT that acknowledges that page): {value}" + ), + _ => {} + }, + None => { + assert!( + !acc.is_empty() || !gaps.is_empty(), + "the credit must produce a page or a gap" + ); + return (acc, max_seq_end, gaps); + } + } + } +} + +/// The quiet-window hold: while the client withholds the credit for a page +/// it has READ, no PACING-CONTRACT frame may arrive — no page, no gap, and +/// in particular no terminal.exit. Unrelated control-plane broadcasts +/// (the exit's terminal.meta.updated retire, sessions.changed, ...) are +/// not pacing frames and pass through, exactly as in the crediting- +/// promptly tests. +async fn assert_quiet_hold(ws: &mut WsClient, ms: u64, what: &str) { + let hold = tokio::time::Instant::now() + Duration::from_millis(ms); + while tokio::time::Instant::now() < hold { + if let Some(value) = next_json_or_timeout(ws, Duration::from_millis(250)).await { + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") => panic!( + "no page may flow while credits are withheld ({what}): {value}" + ), + Some("terminal.output.gap") => panic!( + "no gap may flow while credits are withheld ({what}): {value}" + ), + Some("terminal.exit") => panic!( + "terminal.exit must ride the credit that acknowledges the page reaching the armed exit head ({what}): {value}" + ), + _ => {} + } + } + } +} + +/// E2R2 finding, required test (b) — THE PREMATURE-EXIT ORDERING, over the +/// real socket with EXPLICITLY controlled credit timing (the +/// parser-consumption boundary is the thing under test; the crediting- +/// promptly tests cannot see this race). The exit stages mid-restore, the +/// credits walk the pages to the one that reaches the frozen exit head, +/// and THAT PAGE IS READ BUT NOT CREDITED: no exit may deliver on the +/// drive that emitted it. Only the credit acknowledging that page may +/// deliver the exit — and it must be the client's last frame. +#[tokio::test] +async fn natural_exit_exit_page_read_uncredited_holds_the_exit_for_its_credit() { + let ring = 512 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-exit-hold-credit").await; + flood_until_complete(&url, &mut driver, &terminal_id, 700).await; + + // Paced attach mid-restore; the first page is outstanding and NOT + // credited yet. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-exit-holdc").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let mut credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("first page frames"); + assert!( + credited < head, + "the session is mid-restore when the shell exits" + ); + + // The shell produces one FINAL marker line and exits naturally while + // the first page is uncredited (octal escapes keep the echoed command + // from containing the literal marker). + send_input( + &mut paced, + &terminal_id, + "printf '\\106\\111\\116\\101\\114\\063\\055\\115\\101\\122\\113\\105\\122\\012'; exit\n", + ) + .await; + // The withhold: nothing flows without credits (no page, no exit). + assert_quiet_hold(&mut paced, 1_500, "after the exit staged").await; + + // Walk the pages ONE CREDIT AT A TIME: credit the outstanding first + // page, read the page that credit grants, credit it in turn, until the + // page carrying the literal marker — the page that reached the frozen + // exit head. That page is then HELD uncredited: the parser-consumption + // boundary. + let marker = "FINAL3-MARKER"; + let mut pages = 0usize; + credit(&mut paced, &terminal_id, "attach-exit-holdc", credited).await; + let marker_page_end = loop { + let (acc, page_end, gaps) = + read_credited_page(&mut paced, Duration::from_millis(400)).await; + assert!( + gaps.is_empty(), + "no retention loss in this fixture: {gaps:?}" + ); + pages += 1; + assert!(pages < 500, "the page walk must converge"); + if acc.contains(marker) { + break page_end; + } + credit(&mut paced, &terminal_id, "attach-exit-holdc", page_end).await; + credited = credited.max(page_end); + }; + assert!( + marker_page_end >= head, + "the marker page reached the frozen head (page end {marker_page_end}, head {head})" + ); + assert!( + credited < marker_page_end, + "the marker page is UN-CREDITED at the hold (credited {credited})" + ); + + // THE HOLD: the page reaching the armed exit head is read but not + // credited — NO exit may deliver. (The pre-fix removal delivered it + // here, straight off the drive that emitted the page.) + assert_quiet_hold( + &mut paced, + 1_500, + "the exit page's parser-consumption boundary", + ) + .await; + + // The acknowledging credit: the exit arrives NOW, and it is the + // client's LAST frame (the full-stream ordering — a post-exit quiet + // window proves nothing follows). + credit( + &mut paced, + &terminal_id, + "attach-exit-holdc", + marker_page_end, + ) + .await; + let exit = next_json_or_timeout(&mut paced, Duration::from_secs(5)) + .await + .expect("the exit rides the credit acknowledging the marker page"); + assert_eq!( + exit.get("type").and_then(|v| v.as_str()), + Some("terminal.exit"), + "the acknowledging credit delivers the staged exit: {exit}" + ); + let post_exit = tokio::time::Instant::now() + Duration::from_millis(1_000); + while tokio::time::Instant::now() < post_exit { + if let Some(value) = next_json_or_timeout(&mut paced, Duration::from_millis(250)).await { + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.exit") => { + panic!("exactly one terminal.exit may ever arrive: {value}") + } + Some("terminal.output") | Some("terminal.output.gap") => panic!( + "no page or gap may follow terminal.exit (the exit is the terminal's last pacing frame): {value}" + ), + _ => {} + } + } + } +} + +/// E2R2 finding, required test (d) — RETENTION OVERRUN mid-wait: the ring +/// evicts the credited window's next-needed frames while the client +/// withholds, the exit then stages behind the armed deferral, and the next +/// credit reports the EXACT bounds-carrying gap before paging the retained +/// window — with the FINAL page (the one reaching the frozen exit head) +/// held uncredited across the parser-consumption boundary: the gap first, +/// the pages on credits, and the exit only on the credit that +/// acknowledges the final page. +#[tokio::test] +async fn natural_exit_retention_gap_then_held_final_page_delivers_exit_on_its_credit() { + let ring = 12 * 1024; + let url = spawn_server(ring).await; + let mut driver = connect(&url).await; + hello(&mut driver, false).await; + let terminal_id = create_shell_terminal(&mut driver, "create-exit-gap-hold").await; + flood_until_complete(&url, &mut driver, &terminal_id, 100).await; + + // Paced attach mid-restore; the first page is outstanding, uncredited. + let mut paced = connect(&url).await; + hello(&mut paced, true).await; + let (ready, page1) = + paced_attach_first_page(&mut paced, &terminal_id, "attach-exit-gaph").await; + let head = ready["headSeq"].as_i64().expect("headSeq"); + let credited = page1 + .iter() + .map(|f| f["seqEnd"].as_i64().unwrap_or(0)) + .max() + .expect("first page frames"); + assert!( + credited < head, + "the session is mid-restore when the ring churns" + ); + + // Evict the credited window's middle while the client withholds. + let mut evictor = connect(&url).await; + hello(&mut evictor, false).await; + attach(&mut evictor, &terminal_id, "attach-exit-gap-evictor").await; + let marker2 = "FLOOD-DONE-MARKER"; + send_input(&mut evictor, &terminal_id, &flood_command(400, marker2)).await; + let deadline = tokio::time::Instant::now() + Duration::from_secs(30); + let (evictor_acc, _) = drain_until_marker(&mut evictor, marker2, deadline).await; + assert!( + evictor_acc.contains(marker2), + "the evicting flood completes" + ); + drop(evictor); + + // The exit stages behind the still-armed deferral; the withhold holds. + send_input( + &mut paced, + &terminal_id, + "printf '\\106\\111\\116\\101\\114\\064\\055\\115\\101\\122\\113\\105\\122\\012'; exit\n", + ) + .await; + assert_quiet_hold(&mut paced, 1_500, "after the exit staged behind the gap").await; + + // THE CREDIT: the first product is the EXACT bounds-carrying gap. + credit(&mut paced, &terminal_id, "attach-exit-gaph", credited).await; + let marker = "FINAL4-MARKER"; + let mut gap_seen: Option = None; + let mut pages = 0usize; + let mut cursor = credited; + let marker_page_end = loop { + let (acc, page_end, gaps) = + read_credited_page(&mut paced, Duration::from_millis(400)).await; + if let Some(gap) = gaps.first() { + let first = gap_seen.replace(gap.clone()); + assert!( + first.is_none(), + "exactly one retention gap may report: {gap_seen:?}" + ); + assert_eq!( + gap.get("type").and_then(|v| v.as_str()), + Some("terminal.output.gap"), + "the overrun reports as the negotiated gap: {gap}" + ); + assert_eq!( + gap["reason"], "replay_window_exceeded", + "the gap names retention loss: {gap}" + ); + assert_eq!( + gap["fromSeq"].as_i64(), + Some(credited + 1), + "the lost interval starts at the credited cursor+1: {gap}" + ); + let gap_oldest = gap["oldestRetainedSeq"] + .as_i64() + .expect("oldestRetainedSeq"); + assert_eq!( + gap["toSeq"].as_i64(), + Some(gap_oldest - 1), + "the lost interval ends just before the new ring front: {gap}" + ); + assert_eq!(gap["attachRequestId"], "attach-exit-gaph"); + cursor = cursor.max(gap_oldest - 1); + } + pages += 1; + assert!(pages < 500, "the page walk must converge"); + let start_ok = acc.is_empty() || gap_seen.is_some(); + assert!( + start_ok, + "frames without a gap would be a silent forward jump" + ); + if acc.contains(marker) { + break page_end; + } + credit(&mut paced, &terminal_id, "attach-exit-gaph", page_end).await; + cursor = cursor.max(page_end); + }; + assert!( + gap_seen.is_some(), + "the retention gap reported before the continuation" + ); + assert!( + cursor < marker_page_end, + "the marker page is UN-CREDITED at the hold (cursor {cursor})" + ); + + // THE HOLD: the final page (reaching the frozen exit head) is read but + // not credited — no exit may deliver on the drive that emitted it. + assert_quiet_hold( + &mut paced, + 1_500, + "the final page's parser-consumption boundary", + ) + .await; + + // The acknowledging credit: the exit arrives and is the last frame. + credit( + &mut paced, + &terminal_id, + "attach-exit-gaph", + marker_page_end, + ) + .await; + let exit = next_json_or_timeout(&mut paced, Duration::from_secs(5)) + .await + .expect("the exit rides the credit acknowledging the final page"); + assert_eq!( + exit.get("type").and_then(|v| v.as_str()), + Some("terminal.exit"), + "the acknowledging credit delivers the staged exit: {exit}" + ); + let post_exit = tokio::time::Instant::now() + Duration::from_millis(1_000); + while tokio::time::Instant::now() < post_exit { + if let Some(value) = next_json_or_timeout(&mut paced, Duration::from_millis(250)).await { + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.exit") => panic!("exactly one terminal.exit may ever arrive: {value}"), + Some("terminal.output") | Some("terminal.output.gap") => panic!( + "no page or gap may follow terminal.exit (the exit is the terminal's last pacing frame): {value}" + ), + _ => {} + } + } + } +} diff --git a/crates/freshell-ws/tests/term09_output_queue.rs b/crates/freshell-ws/tests/term09_output_queue.rs index e8a94eb8d..3a7fd8ac7 100644 --- a/crates/freshell-ws/tests/term09_output_queue.rs +++ b/crates/freshell-ws/tests/term09_output_queue.rs @@ -22,6 +22,127 @@ use freshell_ws::WsState; const AUTH_TOKEN: &str = "s3cr3t-token-abcdef"; +/// PTY floods in this file are heavy (tens of MB of PTY traffic plus +/// per-frame JSON serialization). Serialize the flood tests within this +/// binary: concurrent floods starve each other's production and drain +/// rates, and a reader starved past a stall window legitimately trips the +/// very liveness decisions under test — the failures would be contention, +/// not behavior. An async mutex: the guard is held across `.await`s for +/// the whole test. +static FLOOD_TEST_LOCK: tokio::sync::Mutex<()> = tokio::sync::Mutex::const_new(()); + +async fn flood_test_serial() -> tokio::sync::MutexGuard<'static, ()> { + FLOOD_TEST_LOCK.lock().await +} + +// ── capturing tracing layer (dev-only test facility, the paced_replay.rs +// pattern). PROCESS-GLOBAL by deliberate choice: the spawned in-process +// axum server emits its diagnostics from the connection task, and a +// thread-local `set_default` capture is UNSOUND for callsites shared with +// sibling threads — tracing-core caches each callsite's Interest +// process-wide on first registration, so a subscriber-less sibling thread +// executing a shared emission site first caches `Interest::never` and the +// event! macro short-circuits before any dispatch. One global subscriber +// sees every thread's events; every read below MUST filter by message name +// because ALL tests in this binary share the vec. +// ───────────────────────────────────────────────────────────────────────── + +use std::sync::Mutex; +use tracing::field::{Field, Visit}; +use tracing::{Event, Subscriber}; +use tracing_subscriber::layer::{Context, SubscriberExt}; +use tracing_subscriber::Layer; + +#[derive(Debug, Clone, Default)] +struct CapturedEvent { + message: String, + fields: std::collections::BTreeMap, +} + +#[derive(Default)] +struct FieldVisitor { + message: String, + fields: std::collections::BTreeMap, +} + +impl Visit for FieldVisitor { + fn record_debug(&mut self, field: &Field, value: &dyn std::fmt::Debug) { + let rendered = format!("{value:?}"); + if field.name() == "message" { + self.message = rendered; + } else { + self.fields.insert(field.name().to_string(), rendered); + } + } + + fn record_str(&mut self, field: &Field, value: &str) { + if field.name() == "message" { + self.message = value.to_string(); + } else { + self.fields + .insert(field.name().to_string(), value.to_string()); + } + } + + fn record_i64(&mut self, field: &Field, value: i64) { + self.fields + .insert(field.name().to_string(), value.to_string()); + } + + fn record_u64(&mut self, field: &Field, value: u64) { + self.fields + .insert(field.name().to_string(), value.to_string()); + } +} + +struct CaptureLayer { + events: std::sync::Arc>>, +} + +impl Layer for CaptureLayer { + fn on_event(&self, event: &Event<'_>, _ctx: Context<'_, S>) { + let mut visitor = FieldVisitor::default(); + event.record(&mut visitor); + self.events + .lock() + .expect("capture lock") + .push(CapturedEvent { + message: visitor.message, + fields: visitor.fields, + }); + } +} + +/// Process-global capture for this test binary (first caller installs; +/// `get_or_init` is the synchronization). This binary installs no other +/// global subscriber; `.expect()` turns any future second installer into an +/// immediate diagnosable panic instead of a silently-empty capture. +fn global_capture() -> std::sync::Arc>> { + static EVENTS: std::sync::OnceLock>>> = + std::sync::OnceLock::new(); + std::sync::Arc::clone(EVENTS.get_or_init(|| { + let events = std::sync::Arc::new(Mutex::new(Vec::new())); + let layer = CaptureLayer { + events: std::sync::Arc::clone(&events), + }; + // Level-filter the process-global registry to WARN (task-010b + // hygiene): an unfiltered registry enables every callsite + // process-wide — including any future debug site — a latent perf + // and determinism footgun on the flood path. Every event this + // binary's assertions read is warn-level + // (`ws.terminal_stream.catastrophic_close`, and the sibling + // `ws.terminal_stream.queue_overflow_spill`), so the filter + // disables nothing the tests read while every sub-WARN callsite + // short-circuits before dispatch. + let subscriber = tracing_subscriber::registry() + .with(layer) + .with(tracing_subscriber::filter::LevelFilter::WARN); + tracing::subscriber::set_global_default(subscriber) + .expect("this test binary installs exactly one global subscriber"); + events + })) +} + fn test_settings_value() -> serde_json::Value { serde_json::json!({ "ai": {}, @@ -190,6 +311,31 @@ async fn complete_handshake(ws: &mut TestWs) { } } +/// Same as [`complete_handshake`], but the hello carries a capabilities +/// object (the restore-contract negotiation tests need it). +async fn complete_handshake_with_capabilities(ws: &mut TestWs, capabilities: serde_json::Value) { + ws.send(WsMessage::Text( + serde_json::json!({ + "type": "hello", + "token": AUTH_TOKEN, + "protocolVersion": freshell_protocol::WS_PROTOCOL_VERSION, + "capabilities": capabilities, + }) + .to_string(), + )) + .await + .expect("send hello"); + + for _ in 0..4u8 { + let msg = tokio::time::timeout(Duration::from_secs(5), ws.next()) + .await + .expect("handshake message within timeout") + .expect("stream not ended") + .expect("no ws error"); + assert!(matches!(msg, WsMessage::Text(_))); + } +} + async fn create_shell_terminal(ws: &mut TestWs, request_id: &str) -> String { ws.send(WsMessage::Text( serde_json::json!({ @@ -324,6 +470,7 @@ async fn drain_until_marker_or_deadline( /// -- legacy's "slow-client gap/recovery or documented close". #[tokio::test] async fn slow_client_does_not_block_fast_client_and_is_bounded() { + let _flood_guard = flood_test_serial().await; let term09 = Term09Config { queue_max_bytes: 8 * 1024, catastrophic_buffered_bytes: 32 * 1024, @@ -368,8 +515,13 @@ async fn slow_client_does_not_block_fast_client_and_is_bounded() { .expect("send flood input"); // The FAST client must see the flood complete promptly, regardless of - // the slow client never draining anything. - let fast_deadline = tokio::time::Instant::now() + Duration::from_secs(20); + // the slow client never draining anything. 60 s wall-clock margin + // (task-010b, the task-10 retune precedent): structurally identical to + // the fast-client drain that starved past its 20 s deadline when the + // full-workspace gate ran on a loaded shared box (1.9 s isolated). + // The deadline only bounds failure diagnosis — the marker breaks the + // loop the moment the flood completes — never the pass-path wall time. + let fast_deadline = tokio::time::Instant::now() + Duration::from_secs(60); let (fast_acc, _fast_gap, fast_closed) = drain_until_marker_or_deadline(&mut fast, marker, fast_deadline).await; assert!( @@ -385,8 +537,12 @@ async fn slow_client_does_not_block_fast_client_and_is_bounded() { // NOW resume the slow client and observe the TERM-09 policy in effect: // either it received a queue-overflow gap, or it was already closed // (catastrophic backpressure). Both are acceptable per the acceptance - // text ("slow-client gap/recovery or documented close"). - let slow_deadline = tokio::time::Instant::now() + Duration::from_secs(10); + // text ("slow-client gap/recovery or documented close"). 30 s (3x) + // resume margin (task-010b): a stuck-client resume wait structurally + // identical to the gap-negotiation test's, whose class starved past + // its original bound under full-workspace gate load; the gap-or-close + // it waits for already exists server-side. + let slow_deadline = tokio::time::Instant::now() + Duration::from_secs(30); let (_slow_acc, slow_gap, slow_closed) = drain_until_marker_or_deadline(&mut slow, marker, slow_deadline).await; assert!( @@ -396,3 +552,651 @@ async fn slow_client_does_not_block_fast_client_and_is_bounded() { (catastrophic backpressure fired); observed neither" ); } + +/// What a throttled drain observed: accumulated output text, total output +/// payload bytes, every `terminal.output.gap` frame (full JSON, in delivery +/// order), every delivered output frame's `[seqStart..seqEnd]` range (in +/// delivery order), and whether the connection ended. +struct ThrottledDrain { + text: String, + output_bytes: usize, + gaps: Vec, + frame_seqs: Vec<(i64, i64)>, + closed: bool, +} + +/// Read terminal output at a bounded BYTE rate — a slow-but-PROGRESSING +/// consumer (the incident's shape: a real browser on a slow path that keeps +/// draining, never a dead socket). After every `chunk_bytes` of output +/// payload received, pause `chunk_pause`, so the sustained drain rate is +/// ~`chunk_bytes / chunk_pause`. Stops at `marker`, connection end, or the +/// deadline (whichever first). +async fn drain_throttled_until_marker( + ws: &mut TestWs, + marker: &str, + chunk_bytes: usize, + chunk_pause: Duration, + deadline: tokio::time::Instant, +) -> ThrottledDrain { + let mut report = ThrottledDrain { + text: String::new(), + output_bytes: 0, + gaps: Vec::new(), + frame_seqs: Vec::new(), + closed: false, + }; + let mut received_since_pause = 0usize; + // Rolling tail for marker detection: a marker can only arrive in fresh + // data (possibly straddling a frame boundary), so scanning the whole + // accumulated text per message would be quadratic — at incident-scale + // flood sizes that alone throttles the drain and masquerades as server + // slowness. The tail spans several marker lengths; a few hundred bytes + // bounds the scan at O(1) per message. + let mut tail = String::new(); + while tokio::time::Instant::now() < deadline { + let remaining = deadline.saturating_duration_since(tokio::time::Instant::now()); + match tokio::time::timeout(remaining.max(Duration::from_millis(1)), ws.next()).await { + Ok(Some(Ok(WsMessage::Text(text)))) => { + let Ok(value) = serde_json::from_str::(&text) else { + continue; + }; + match value.get("type").and_then(|v| v.as_str()) { + Some("terminal.output") | Some("terminal.output.batch") => { + let data_len = value + .get("data") + .and_then(|v| v.as_str()) + .map(str::len) + .unwrap_or(0); + if let Some(data) = value.get("data").and_then(|v| v.as_str()) { + report.text.push_str(data); + tail.push_str(data); + } + let seq_start = value.get("seqStart").and_then(|v| v.as_i64()); + let seq_end = value.get("seqEnd").and_then(|v| v.as_i64()); + if let (Some(start), Some(end)) = (seq_start, seq_end) { + report.frame_seqs.push((start, end)); + } + report.output_bytes += data_len; + received_since_pause += data_len; + if received_since_pause >= chunk_bytes { + received_since_pause = 0; + tokio::time::sleep(chunk_pause).await; + } + } + Some("terminal.output.gap") => report.gaps.push(value), + _ => {} + } + if tail.len() > 4 * marker.len() { + let keep = tail.len() - 2 * marker.len(); + tail.drain(..keep); + } + if tail.contains(marker) { + break; + } + } + Ok(Some(Ok(WsMessage::Close(_)))) | Ok(None) | Ok(Some(Err(_))) => { + report.closed = true; + break; + } + Ok(Some(Ok(_))) => {} + Err(_) => break, // timed out + } + } + report +} + +/// Responsive-terminal-restore Workstream 3, the incident shape: a burst far +/// larger than the SPILL bound, drained by a slow-but-progressing client. +/// The production incident (~21-25 MB backlog) disconnected the client under +/// the old defaults (disconnect 16 MiB fired strictly before spill 32 MiB +/// could evict); under the new defaults the same pressure must SPILL oldest +/// output with an exact generation-scoped gap and KEEP THE CONNECTION OPEN, +/// while the client keeps making successful sends and receives well over the +/// old disconnect threshold (>16 MiB) across a >10 s pressure window. +#[tokio::test] +async fn incident_backlog_spills_instead_of_disconnecting() { + let _flood_guard = flood_test_serial().await; + // The real default configuration — this test pins the shipped defaults, + // not an injected variant. + let term09 = Term09Config::default(); + let url = spawn_server(term09).await; + + let mut creator = connect_and_complete_handshake(&url).await; + let terminal_id = create_shell_terminal(&mut creator, "create-incident").await; + + // The victim: a normal client that reads continuously but slowly + // (~1 MiB/s), like a browser on a congested path. It is never stuck — + // its socket keeps accepting writes the whole time. + let mut victim = connect_and_complete_handshake(&url).await; + attach(&mut victim, &terminal_id, "attach-incident").await; + tokio::time::sleep(Duration::from_millis(200)).await; + + let marker = "FLOOD-DONE-MARKER"; + // ~90 bytes/line * 400_000 lines =~ 40 MB — comfortably past the spill + // bound (16 MiB) so eviction genuinely fires, while the ~1 MiB/s drain + // keeps the client sending successfully across a >10 s pressure window. + let flood = flood_command(400_000, marker); + let started = tokio::time::Instant::now(); + creator + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.input", + "terminalId": terminal_id, + "data": flood, + }) + .to_string(), + )) + .await + .expect("send flood input"); + + let deadline = started + Duration::from_secs(60); + let report = drain_throttled_until_marker( + &mut victim, + marker, + 256 * 1024, + Duration::from_millis(250), + deadline, + ) + .await; + let elapsed = started.elapsed(); + + assert!( + !report.closed, + "a slow-but-progressing client must survive incident-scale pressure: \ + the spill bound (eviction + gap) relieves it strictly before any \ + pressure-related disconnect" + ); + assert!( + report.text.contains(marker), + "the flood tail must be delivered after the spill; got {} bytes without \ + the marker", + report.output_bytes + ); + assert!( + elapsed >= Duration::from_secs(10), + "the pressure window must span the full old stall window (10s); observed {elapsed:?}" + ); + assert!( + report.output_bytes > 16 * 1024 * 1024, + "the client must receive more than the OLD disconnect threshold \ + (>16 MiB) while under pressure; got {} bytes", + report.output_bytes + ); + assert!( + !report.gaps.is_empty(), + "a burst larger than the spill bound must produce an eviction gap" + ); + for gap in &report.gaps { + assert_eq!( + gap["reason"], "queue_overflow", + "the observed loss is the queue-overflow (spill) gap: {gap}" + ); + assert_eq!( + gap["terminalId"], terminal_id, + "gap addresses this terminal" + ); + let from = gap["fromSeq"].as_i64().expect("fromSeq"); + let to = gap["toSeq"].as_i64().expect("toSeq"); + assert!(from >= 1 && from <= to, "exact coalesced interval: {gap}"); + for (start, end) in &report.frame_seqs { + assert!( + *end < from || *start > to, + "evicted bytes are never delivered: frame [{start}..{end}] must \ + not intersect gap [{from}..{to}]" + ); + } + } + let mut seqs = report.frame_seqs.iter(); + if let Some(mut last) = seqs.next() { + for next in seqs { + assert!( + next.0 > last.1, + "delivered frames stay in strict sequence order across gaps: \ + {last:?} then {next:?}" + ); + last = next; + } + } +} + +/// Drain-progress liveness, end to end: a client that keeps making successful +/// socket sends must NEVER be closed by the catastrophic monitor, even while +/// its pending bytes sit continuously OVER the disconnect threshold. +/// +/// The monitor's byte dimension is only observable when the queue may hold +/// more than the threshold, which the boot validation forbids for real +/// configurations (spill must sit strictly below disconnect). This test +/// therefore deliberately injects the OLD inverted shape (queue bound above +/// the disconnect threshold) straight into `WsState` — the one shape in which +/// over-threshold pending bytes can persist — and proves the monitor keys on +/// SEND PROGRESS, not on the byte count: the old sustained-bytes-only +/// decision closed exactly this client. +/// +/// Flake hardening (task-010, task-007 review Minor 1 direction): the +/// original tuning (8 MiB queue against a ~2 MiB/s drain) needed bash +/// production to outpace the drain by a FULL 8 MiB before the first +/// eviction — a margin that disappears under CI load, where PTY production +/// drops to near the drain rate and the gap-evidence assertion fails with +/// the queue never overflowing. The queue bound now sits just above the +/// threshold (the minimal inversion this test exists to inject), the drain +/// matches the incident test's demonstrated ~1 MiB/s accumulation regime, +/// and the stall window is widened to 5 s so a loaded runner's scheduling +/// gaps cannot masquerade as a genuine >2 s send silence. Every assertion +/// is unchanged; only the reachability margins were re-derived. +#[tokio::test] +async fn slow_but_progressing_client_survives_over_threshold_backlog() { + let _flood_guard = flood_test_serial().await; + // Inverted ON PURPOSE (see doc comment): 2 MiB queue > 1 MiB threshold. + // The 5 s stall window keeps the decision observable while a loaded + // runner cannot fake a >2 s send silence; the harness's 30 s ping + // interval keeps the per-send write timeout at 60 s, far beyond the + // window, so only the monitor's decision can close this connection. + let term09 = Term09Config { + queue_max_bytes: 2 * 1024 * 1024, + catastrophic_buffered_bytes: 1024 * 1024, + catastrophic_stall_ms: 5_000, + }; + let url = spawn_server(term09).await; + + let mut creator = connect_and_complete_handshake(&url).await; + let terminal_id = create_shell_terminal(&mut creator, "create-progress").await; + + let mut victim = connect_and_complete_handshake(&url).await; + attach(&mut victim, &terminal_id, "attach-progress").await; + tokio::time::sleep(Duration::from_millis(200)).await; + + let marker = "FLOOD-DONE-MARKER"; + // ~27 MB at ~1 MiB/s drain: the queue pins at its 2 MiB bound — over the + // 1 MiB threshold continuously for many multiples of the 5 s stall + // window — while every frame the writer leases completes a successful + // send. The overflow (and its exact gap) is reachable even when a + // loaded runner slows PTY production to the drain rate. + let flood = flood_command(300_000, marker); + creator + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.input", + "terminalId": terminal_id, + "data": flood, + }) + .to_string(), + )) + .await + .expect("send flood input"); + + // Deadline sized for a ~630 KB/s effective drain under CI load (the + // per-frame JSON parsing contends with everything else on the runner): + // ~27 MB then needs ~45 s, so 75 s keeps >1.5x headroom. + let deadline = tokio::time::Instant::now() + Duration::from_secs(75); + let report = drain_throttled_until_marker( + &mut victim, + marker, + 128 * 1024, + Duration::from_millis(125), + deadline, + ) + .await; + + assert!( + !report.closed, + "a client with continuous successful sends must stay connected even \ + with pending bytes over the threshold for the whole stall window" + ); + assert!( + report.text.contains(marker), + "the progressing client must see the flood complete; got {} bytes", + report.output_bytes + ); + assert!( + !report.gaps.is_empty(), + "the 2 MiB queue must evict (gap) against a ~27 MB flood at this \ + drain rate; the overflow is repaired by the exact gap, not by hanging up" + ); +} + +/// Eviction and supersede are NOT send progress: a connection whose queue +/// bytes shrink only via eviction and a superseding attach — with zero +/// completed socket sends across the stall window — must still be closed by +/// the catastrophic monitor. Byte reductions alone must never masquerade as +/// liveness. +#[tokio::test] +async fn eviction_and_supersede_without_sends_still_close() { + let _flood_guard = flood_test_serial().await; + // Install the process-global capture BEFORE the server spawns: the + // catastrophic_close event below is emitted by the connection task the + // moment the monitor fires, and an event emitted before the subscriber + // exists is lost for good (tracing dispatch is not replayed). + let events = global_capture(); + // Same deliberately-inverted injection as the progressing-client test: + // the queue may hold bytes over the threshold, making the monitor's + // window observable. The 30 s harness ping interval keeps the per-send + // write timeout at 60 s — far beyond this test's bounds — so a closure + // observed here is the monitor's, not the send timeout's. + let term09 = Term09Config { + queue_max_bytes: 8 * 1024 * 1024, + // 32 KiB, not the earlier 1 MiB: premise-neutral (the eviction floor + // stays at the 8 MiB queue cap, far above this, so "bytes shrink via + // eviction while over threshold" is preserved) but it shrinks the + // load-sensitive step — the post-supersede-discard refill — 32x. + // Under concurrent full-suite gates the flood shell's production + // rate collapses, and a 1 MiB refill (≈10k flood lines) can take + // many minutes-to-forever: the monitor honestly has not seen + // over-threshold yet. 32 KiB (≈350 lines) keeps the window + // observable on a contended box; the observation loop below is + // production-progress-driven, so the two together are load-immune. + catastrophic_buffered_bytes: 32 * 1024, + catastrophic_stall_ms: 2_000, + }; + let url = spawn_server(term09).await; + + let mut creator = connect_and_complete_handshake(&url).await; + let terminal_id = create_shell_terminal(&mut creator, "create-stuck").await; + + // The stuck client: tiny SO_RCVBUF and it NEVER reads. Its queue fills, + // then EVICTS continuously (bytes shrinking without any send), and its + // in-flight frame wedges once the kernel buffers fill (zero sends). + let mut stuck = connect_with_tiny_recv_buffer(&url, 4096).await; + complete_handshake(&mut stuck).await; + attach(&mut stuck, &terminal_id, "attach-stuck").await; + tokio::time::sleep(Duration::from_millis(200)).await; + + let marker = "FLOOD-DONE-MARKER"; + // ~60 MB keeps production alive for several seconds past the mid-stall + // supersede below (the queue must refill after the discard). + let flood = flood_command(600_000, marker); + + creator + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.input", + "terminalId": terminal_id, + "data": flood, + }) + .to_string(), + )) + .await + .expect("send flood input"); + + // Do NOT read from the stuck socket at all while the stall window runs — + // reading would drain the kernel buffers and count as send progress. + // First the queue fills and the in-flight frame wedges (eviction keeps + // shrinking queue bytes with zero completed sends). + tokio::time::sleep(Duration::from_millis(1_000)).await; + + // Mid-stall, re-attach the stuck client: the superseding attach DISCARDS + // its entire queued output (a byte reduction that is NOT a send). The + // monitor may honestly reset on the below-threshold fall, but the still- + // producing flood refills the queue and the window must close the + // connection — eviction/supersede alone must never keep it alive. + attach(&mut stuck, &terminal_id, "attach-stuck-supersede").await; + + // Keep not reading through the refill + a full stall window (+margin): + // zero successful sends the whole time. + tokio::time::sleep(Duration::from_millis(4_000)).await; + + // NOW resume reading: the connection should already be terminated (the + // catastrophic monitor fired while we were silent). Attribution does + // NOT come from observing a Close frame: this socket is deliberately + // backpressured to saturation, so the monitor's 4008 CloseFrame cannot + // be delivered — the teardown surfaces as a bare stream end or error. + // The decision is proven server-side by the capture assert below + // (exactly one ws.terminal_stream.catastrophic_close event); a write- + // timeout or keepalive close would produce none. + // + // The observation is PROGRESS-BASED (standing test discipline: a + // wall-clock budget must never fail working code). This test's waits + // starved out three times under concurrent full-suite gates before the + // design converged: a fixed 60 s window (twice), then a 300 s + // decision cap whose real victim was the FLOOD — under extreme + // starvation the shell's production rate collapses and the + // post-discard refill (which must re-cross the injected threshold) + // had not happened yet, so the monitor was honestly still waiting + // (not dead). The threshold injection is therefore sized for the + // contended-box reality (32 KiB re-crosses at even ~1% of focused + // production rate), and the wait below tracks the monitor's own + // decision record in the capture. The only fixed bound before the + // decision is a true-stall cap: 300 s with no decision event at all — + // a dead monitor or a flood starved below ~100 B/s of production, at + // which point no gate finishes anyway. A slow-but-producing box can + // never trip it. After the decision, a 120 s delivery cap catches a + // wedged teardown. + let decision_deadline = tokio::time::Instant::now() + Duration::from_secs(300); + let mut decided_at: Option = None; + loop { + let now = tokio::time::Instant::now(); + if decided_at.is_none() + && events + .lock() + .expect("capture lock") + .iter() + .any(|e| e.message == "ws.terminal_stream.catastrophic_close") + { + decided_at = Some(now); + } + if let Some(at) = decided_at { + assert!( + at.elapsed() < Duration::from_secs(120), + "the monitor decided but its close was never delivered to the \ + now-reading client within 120 s — a wedged teardown, not a \ + slow box" + ); + } else { + assert!( + now < decision_deadline, + "no monitor decision in the capture within 300 s — a dead \ + monitor or a flood starved below ~100 B/s of production (a \ + slow-but-producing box can never trip this)" + ); + } + let tick = if decided_at.is_some() { + Duration::from_millis(50) + } else { + Duration::from_millis(250) + }; + // The loop's only non-panic exit is the close observation itself: + // every other path is a stall-cap panic carrying the failure's + // exact context, so the connection-must-close requirement is + // enforced structurally. + match tokio::time::timeout(tick, stuck.next()).await { + Ok(Some(Ok(WsMessage::Close(_)))) | Ok(None) | Ok(Some(Err(_))) => break, + Ok(Some(Ok(_))) => {} + Err(_) => continue, // tick elapsed with no frame: re-check progress + } + } + + // Task-007 review M3 (landed by task-010): the catastrophic-close event + // must be diagnosable from the log line ALONE. `sends_in_window` is the + // per-occurrence evidence — completed sends DURING the deciding window + // for THIS close, structurally zero (any send resets the window) — while + // `total_sends` (the renamed lifetime `sends` counter) carries the + // connection's whole history. The stuck client here completed sends + // (its attach handshake) BEFORE the window, so the two fields must + // disagree: window evidence 0, lifetime total >= 1. + let close_events: Vec<_> = events + .lock() + .expect("capture lock") + .iter() + .filter(|e| e.message == "ws.terminal_stream.catastrophic_close") + .cloned() + .collect(); + assert_eq!( + close_events.len(), + 1, + "the monitor's close must emit exactly one catastrophic_close event: {close_events:?}" + ); + let close = &close_events[0]; + assert_eq!( + close.fields.get("sends_in_window").map(String::as_str), + Some("0"), + "the deciding window for THIS occurrence was send-silent — the \ + per-occurrence evidence must say so directly: {close:?}" + ); + let total_sends: u64 = close + .fields + .get("total_sends") + .expect("the lifetime counter is carried as total_sends") + .parse() + .expect("total_sends renders as a plain integer"); + assert!( + total_sends >= 1, + "the stuck client completed its attach handshake before the wedge, \ + so the lifetime counter must be nonzero — the field pair is what \ + distinguishes wedge-after-progress from never-sent: {close:?}" + ); + assert_eq!( + close.fields.get("window_ms").map(String::as_str), + Some("2000"), + "the injected stall window is reported" + ); + assert_eq!( + close.fields.get("threshold").map(String::as_str), + Some("32768"), + "the injected threshold is reported" + ); + let pending: usize = close + .fields + .get("pending_bytes") + .expect("pending bytes at fire time") + .parse() + .expect("pending_bytes renders as a plain integer"); + assert!( + pending > 32 * 1024, + "the monitor only fires over-threshold: {close:?}" + ); +} + +/// Read frames until the first `terminal.output.gap` arrives, returning its +/// JSON. Callers resume a previously-stuck client: the delivery queue serves +/// the terminal's pending gap ahead of its remaining frames, so the gap +/// surfaces within the stuck backlog. +async fn first_gap_frame(ws: &mut TestWs, deadline: tokio::time::Instant) -> serde_json::Value { + while tokio::time::Instant::now() < deadline { + let remaining = deadline.saturating_duration_since(tokio::time::Instant::now()); + match tokio::time::timeout(remaining.max(Duration::from_millis(1)), ws.next()).await { + Ok(Some(Ok(WsMessage::Text(text)))) => { + if let Ok(value) = serde_json::from_str::(&text) { + if value.get("type").and_then(|v| v.as_str()) == Some("terminal.output.gap") { + return value; + } + } + } + Ok(Some(Ok(_))) => {} + _ => break, + } + } + panic!("no terminal.output.gap arrived before the deadline"); +} + +/// Restore-contract negotiation gating, end to end on real sockets + a real +/// PTY (responsive-terminal-restore): two slow clients attach to the SAME +/// flooding terminal — one whose hello negotiated `pacedTerminalReplayV1`, +/// one not. The negotiated client's queue-overflow gap carries +/// `headSeq`/`oldestRetainedSeq` resolved from the registry at +/// gap-emission time; the non-negotiated client's gap omits BOTH keys — +/// byte-identical to the pre-contract wire. +#[tokio::test] +async fn queue_overflow_gap_bounds_follow_negotiation() { + let _flood_guard = flood_test_serial().await; + // Tiny queue so overflow fires almost immediately; catastrophic + // backpressure threshold far above it so the connection stays OPEN and + // the observed loss is the queue-overflow gap (not a 4008 close). + let term09 = Term09Config { + queue_max_bytes: 8 * 1024, + catastrophic_buffered_bytes: 8 * 1024 * 1024, + catastrophic_stall_ms: 60_000, + }; + let url = spawn_server(term09).await; + + let mut creator = connect_and_complete_handshake(&url).await; + let terminal_id = create_shell_terminal(&mut creator, "create-gap").await; + + // Negotiated slow client: tiny SO_RCVBUF. It reads NOTHING from flood + // start until the fast client below has seen the flood complete — the + // deterministic TERM-09 slow-reader shape (a client that reads along can + // keep the writer draining and no overflow ever fires). + let mut paced = connect_with_tiny_recv_buffer(&url, 4096).await; + complete_handshake_with_capabilities( + &mut paced, + serde_json::json!({ "pacedTerminalReplayV1": true }), + ) + .await; + attach(&mut paced, &terminal_id, "attach-paced").await; + + // Non-negotiated slow client on the SAME terminal, same stuck shape. + let mut plain = connect_with_tiny_recv_buffer(&url, 4096).await; + complete_handshake(&mut plain).await; + attach(&mut plain, &terminal_id, "attach-plain").await; + + // A fast, always-reading client proves when the flood has fully run. + let mut fast = connect_and_complete_handshake(&url).await; + attach(&mut fast, &terminal_id, "attach-fast").await; + + // Let the attach.ready frames settle before flooding. + tokio::time::sleep(Duration::from_millis(200)).await; + + let marker = "FLOOD-DONE-MARKER"; + // ~90 bytes/line * 20_000 lines =~ 1.8 MB: comfortably overruns both + // stuck clients' 8 KB writer queues long before the 60 s send timeout. + let flood = flood_command(20_000, marker); + creator + .send(WsMessage::Text( + serde_json::json!({ + "type": "terminal.input", + "terminalId": terminal_id, + "data": flood, + }) + .to_string(), + )) + .await + .expect("send flood input"); + + // Wait for the flood to COMPLETE on the fast client: by then both stuck + // clients' queues have overflowed and their queue-overflow gaps exist. + // 60 s wall-clock margin (task-010b, the task-10 retune precedent): + // this drain measures 1.9 s isolated but starved past its 20 s + // deadline when the full-workspace gate ran `cargo test --workspace` + // on a loaded shared box. The deadline only bounds failure diagnosis — + // the marker breaks the loop the moment the flood completes — never + // the pass-path wall time. + let fast_deadline = tokio::time::Instant::now() + Duration::from_secs(60); + let (fast_acc, _fast_gap, fast_closed) = + drain_until_marker_or_deadline(&mut fast, marker, fast_deadline).await; + assert!( + !fast_closed && fast_acc.contains(marker), + "the fast client must see the flood complete before the slow clients resume" + ); + + // NOW resume each stuck client and capture its first queue-overflow gap. + // Same loaded-gate margin as the fast-client drain above (the gaps + // already exist server-side once the flood completes; the deadline only + // bounds how long we wait to observe them through the resumed + // backlog). + let deadline = tokio::time::Instant::now() + Duration::from_secs(60); + let paced_gap = first_gap_frame(&mut paced, deadline).await; + assert_eq!( + paced_gap["reason"], "queue_overflow", + "the negotiated client's observed loss is the queue-overflow gap: {paced_gap}" + ); + let paced_head = paced_gap["headSeq"] + .as_i64() + .expect("negotiated gap carries headSeq"); + let paced_oldest = paced_gap["oldestRetainedSeq"] + .as_i64() + .expect("negotiated gap carries oldestRetainedSeq"); + assert!(paced_head >= 1, "honest current head: {paced_gap}"); + assert!( + paced_oldest >= 1 && paced_oldest <= paced_head + 1, + "oldestRetainedSeq is an honest retention bound (front of the ring, \ + head+1 when empty): {paced_gap}" + ); + + let plain_gap = first_gap_frame(&mut plain, deadline).await; + assert_eq!( + plain_gap["reason"], "queue_overflow", + "the non-negotiated client's observed loss is the queue-overflow gap: {plain_gap}" + ); + assert!( + plain_gap.get("headSeq").is_none() && plain_gap.get("oldestRetainedSeq").is_none(), + "a non-negotiated gap must not gain any new key: {plain_gap}" + ); +} diff --git a/crates/freshell-ws/tests/terminal_interest_wire.rs b/crates/freshell-ws/tests/terminal_interest_wire.rs index 42e9aa5d9..cbb70659e 100644 --- a/crates/freshell-ws/tests/terminal_interest_wire.rs +++ b/crates/freshell-ws/tests/terminal_interest_wire.rs @@ -44,6 +44,52 @@ async fn connect(url: &str, opt_in: bool) -> (TestWs, Value) { receive(&mut ws, "terminal.inventory").await; (ws, ready) } + +/// Connect with an explicit hello capability set (for the lifetime-claim +/// negotiation shapes). +async fn connect_with_capabilities(url: &str, capabilities: Value) -> (TestWs, Value) { + let (mut ws, _) = tokio_tungstenite::connect_async(url).await.unwrap(); + let hello = json!({"type":"hello","token":common::AUTH_TOKEN, + "protocolVersion":freshell_protocol::WS_PROTOCOL_VERSION, + "capabilities": capabilities}); + ws.send(Message::Text(hello.to_string())).await.unwrap(); + let ready = receive(&mut ws, "ready").await; + receive(&mut ws, "terminal.inventory").await; + (ws, ready) +} + +/// Send `terminal.create` (shell) and return the created terminalId. +async fn create_shell_terminal(ws: &mut TestWs, request_id: &str) -> String { + send( + ws, + json!({"type":"terminal.create","requestId":request_id, + "mode":"shell","shell":"system"}), + ) + .await; + let deadline = tokio::time::Instant::now() + common::FRAME_BUDGET; + while tokio::time::Instant::now() < deadline { + let value = tokio::time::timeout(common::FRAME_BUDGET, ws.next()) + .await + .expect("socket remains open") + .expect("valid frame") + .expect("frame ok"); + let Message::Text(text) = value else { continue }; + let parsed: Value = serde_json::from_str(&text).expect("JSON"); + if parsed["type"] == "terminal.created" && parsed["requestId"] == request_id { + return parsed["terminalId"] + .as_str() + .expect("terminalId") + .to_string(); + } + } + panic!("terminal.created never arrived"); +} + +/// Fence: the pong for a ping proves the preceding typed dispatch finished. +async fn fence_with_ping(ws: &mut TestWs) { + send(ws, json!({"type":"ping"})).await; + receive(ws, "pong").await; +} async fn send(ws: &mut TestWs, value: Value) { ws.send(Message::Text(value.to_string())).await.unwrap(); } @@ -112,3 +158,129 @@ async fn malformed_and_stale_snapshots_have_bounded_non_destructive_handling() { send(&mut ws, json!({"type":"ping"})).await; receive(&mut ws, "pong").await; } + +// ── Hidden-pane lifetime claims (responsive-terminal-restore WS1) ── + +/// A negotiated connection's `claimedTerminalIds` round-trips into the +/// registry: the claim clears `released_by_client` WITHOUT attaching (no +/// subscriber exists — no replay was ever granted). +#[tokio::test] +async fn claimed_ids_round_trip_and_apply_without_attaching() { + let (url, registry) = common::spawn_server_with_specs(vec![]).await; + let (mut ws, ready) = connect_with_capabilities( + &url, + json!({"terminalInterestV1":true,"terminalLifetimeClaimV1":true}), + ) + .await; + assert_eq!(ready["capabilities"]["terminalLifetimeClaimV1"], true); + + let tid = create_shell_terminal(&mut ws, "req-claim-1").await; + assert_eq!( + registry.claim_state(&tid), + Some(freshell_terminal::ClaimState { + claimers: 0, + released_by_client: true + }) + ); + + send( + &mut ws, + json!({"type":"terminal.interest","revision":1, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[&tid]}), + ) + .await; + fence_with_ping(&mut ws).await; + + assert_eq!( + registry.claim_state(&tid), + Some(freshell_terminal::ClaimState { + claimers: 1, + released_by_client: false + }), + "the negotiated claim must mark the terminal wanted" + ); + // No attach ever happened: zero subscribers. + assert!(!registry + .directory() + .iter() + .any(|d| d.terminal_id == tid && d.has_clients)); + + registry.kill(&tid); +} + +/// The latest snapshot's claim set wins per connection: a superseding +/// snapshot that omits the id is the explicit withdrawal (release restores +/// fast-reap eligibility). +#[tokio::test] +async fn superseded_snapshot_withdraws_the_previous_claim() { + let (url, registry) = common::spawn_server_with_specs(vec![]).await; + let (mut ws, _) = connect_with_capabilities( + &url, + json!({"terminalInterestV1":true,"terminalLifetimeClaimV1":true}), + ) + .await; + let tid = create_shell_terminal(&mut ws, "req-claim-2").await; + + send( + &mut ws, + json!({"type":"terminal.interest","revision":1, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[&tid]}), + ) + .await; + fence_with_ping(&mut ws).await; + assert_eq!( + registry + .claim_state(&tid) + .map(|s| (s.claimers, s.released_by_client)), + Some((1, false)) + ); + + // Supersede: revision 2 no longer claims the id. + send( + &mut ws, + json!({"type":"terminal.interest","revision":2, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[]}), + ) + .await; + fence_with_ping(&mut ws).await; + assert_eq!( + registry + .claim_state(&tid) + .map(|s| (s.claimers, s.released_by_client)), + Some((0, true)), + "withdrawal via the latest snapshot must restore fast-reap eligibility" + ); + + registry.kill(&tid); +} + +/// Claims from a connection that did NOT negotiate the capability are +/// ignored server-side (the client gates send-side; the server must be +/// robust on its own). +#[tokio::test] +async fn claims_from_an_unnegotiated_connection_are_ignored() { + let (url, registry) = common::spawn_server_with_specs(vec![]).await; + let (mut ws, _) = connect(&url, true).await; // terminalInterestV1 only + let tid = create_shell_terminal(&mut ws, "req-claim-3").await; + + send( + &mut ws, + json!({"type":"terminal.interest","revision":1, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[&tid]}), + ) + .await; + fence_with_ping(&mut ws).await; + assert_eq!( + registry + .claim_state(&tid) + .map(|s| (s.claimers, s.released_by_client)), + Some((0, true)), + "a non-negotiated connection's claim field must be ignored" + ); + + registry.kill(&tid); +} diff --git a/crates/freshell-ws/tests/terminal_lifetime_claim.rs b/crates/freshell-ws/tests/terminal_lifetime_claim.rs new file mode 100644 index 000000000..a07b90df7 --- /dev/null +++ b/crates/freshell-ws/tests/terminal_lifetime_claim.rs @@ -0,0 +1,337 @@ +//! Real-server idle-threshold survival for hidden-pane lifetime claims +//! (responsive-terminal-restore Workstream 1). +//! +//! Runs as its own INTEGRATION binary on purpose: the shared test clock +//! (`freshell_platform::clock`, HARNESS-14) is process-global, so overriding +//! it here keeps the virtual-time machinery scoped to this binary (same +//! discipline as `test_clock_routing.rs`). +//! +//! Proves, over the REAL axum server + REAL `tokio-tungstenite` client + a +//! REAL spawned shell PTY, with a configured `autoKillIdleMinutes`: +//! 1. a created-hidden terminal held only by the negotiated claim survives +//! the configured idle threshold, and an explicit claim withdrawal (a +//! later interest snapshot omitting the id) re-exposes it — the sweep +//! then reaps it at the threshold; +//! 2. a claim-connection DROP (transport loss) keeps the terminal wanted +//! past the configured threshold (24-hour hard cap only — never +//! threshold-reaped), and the hard cap stays the cleanup backstop. +//! +//! Zero wall-clock sleeps for the virtual waits; the sweep itself is driven +//! explicitly via the public `enforce_idle_kills`. The only processes this +//! binary kills are its own spawned server's terminals (its own PTYs), so +//! the destructive-test sandbox rule does not apply. + +use std::sync::{Mutex, MutexGuard}; +use std::time::Duration; + +use futures_util::{SinkExt, StreamExt}; +use serde_json::{json, Value}; +use tokio_tungstenite::tungstenite::Message; + +mod common; + +const MINUTE_MS: i64 = 60_000; + +/// Serialize + scope the process-global override within THIS binary. +static LOCK: Mutex<()> = Mutex::new(()); + +struct GateGuard { + _guard: MutexGuard<'static, ()>, +} + +impl GateGuard { + fn enable() -> Self { + let guard = LOCK.lock().unwrap_or_else(|p| p.into_inner()); + freshell_platform::clock::set_enabled_override_for_tests(Some(true)); + freshell_platform::clock::reset().expect("override enabled"); + freshell_platform::clock::freeze().expect("freeze at virtual T"); + Self { _guard: guard } + } +} + +impl Drop for GateGuard { + fn drop(&mut self) { + let _ = freshell_platform::clock::reset(); + freshell_platform::clock::set_enabled_override_for_tests(None); + } +} + +async fn receive(ws: &mut common::TestWs, kind: &str) -> Value { + tokio::time::timeout(common::FRAME_BUDGET, async { + loop { + match ws + .next() + .await + .expect("socket remains open") + .expect("valid frame") + { + Message::Text(text) => { + let value: Value = serde_json::from_str(&text).expect("JSON"); + if value["type"] == kind { + return value; + } + assert_ne!(value["type"], "error", "unexpected error: {value}"); + } + Message::Ping(bytes) => { + ws.send(Message::Pong(bytes)).await.unwrap(); + } + Message::Close(_) => panic!("unexpected close"), + _ => {} + } + } + }) + .await + .expect("bounded frame receive") +} + +async fn send(ws: &mut common::TestWs, value: Value) { + ws.send(Message::Text(value.to_string())).await.unwrap(); +} + +/// Connect with the lifetime-claim negotiation and read past the handshake. +async fn connect_claim_negotiated(url: &str) -> (common::TestWs, Value) { + let (mut ws, _) = tokio_tungstenite::connect_async(url).await.unwrap(); + send( + &mut ws, + json!({"type":"hello","token":common::AUTH_TOKEN, + "protocolVersion":freshell_protocol::WS_PROTOCOL_VERSION, + "capabilities":{"terminalInterestV1":true,"terminalLifetimeClaimV1":true}}), + ) + .await; + let ready = receive(&mut ws, "ready").await; + receive(&mut ws, "terminal.inventory").await; + (ws, ready) +} + +/// Connect negotiating `terminalInterestV1` ONLY — the claim-robustness +/// shape: the connection may send `terminal.interest` snapshots, but its +/// `claimedTerminalIds` field must be ignored server-side because it never +/// negotiated `terminalLifetimeClaimV1`. +async fn connect_interest_only(url: &str) -> (common::TestWs, Value) { + let (mut ws, _) = tokio_tungstenite::connect_async(url).await.unwrap(); + send( + &mut ws, + json!({"type":"hello","token":common::AUTH_TOKEN, + "protocolVersion":freshell_protocol::WS_PROTOCOL_VERSION, + "capabilities":{"terminalInterestV1":true}}), + ) + .await; + let ready = receive(&mut ws, "ready").await; + receive(&mut ws, "terminal.inventory").await; + (ws, ready) +} + +/// Send `terminal.create` (shell) and return the created terminalId. +async fn create_shell_terminal(ws: &mut common::TestWs, request_id: &str) -> String { + send( + ws, + json!({"type":"terminal.create","requestId":request_id, + "mode":"shell","shell":"system"}), + ) + .await; + let deadline = tokio::time::Instant::now() + common::FRAME_BUDGET; + while tokio::time::Instant::now() < deadline { + let value = tokio::time::timeout(common::FRAME_BUDGET, ws.next()) + .await + .expect("socket remains open") + .expect("valid frame") + .expect("frame ok"); + let Message::Text(text) = value else { continue }; + let parsed: Value = serde_json::from_str(&text).expect("JSON"); + if parsed["type"] == "terminal.created" && parsed["requestId"] == request_id { + return parsed["terminalId"] + .as_str() + .expect("terminalId") + .to_string(); + } + } + panic!("terminal.created never arrived"); +} + +/// Fence: the pong for a ping proves the preceding typed dispatch finished. +async fn fence_with_ping(ws: &mut common::TestWs) { + send(ws, json!({"type":"ping"})).await; + receive(ws, "pong").await; +} + +/// Poll until `pred` holds (bounded) — used to await the server's +/// asynchronous socket-close cleanup (the claim sweep in remove_connection). +async fn eventually(mut probe: impl FnMut() -> Option) -> T { + let deadline = tokio::time::Instant::now() + Duration::from_secs(10); + loop { + if let Some(value) = probe() { + return value; + } + assert!( + tokio::time::Instant::now() < deadline, + "condition never became true" + ); + tokio::time::sleep(Duration::from_millis(10)).await; + } +} + +/// Created-hidden + claim-only survival past the configured threshold, then +/// explicit withdrawal re-exposes the terminal to the threshold and the +/// sweep reaps it. +#[tokio::test] +async fn created_hidden_claim_survives_threshold_and_withdrawal_reaps() { + let _gate = GateGuard::enable(); + let (url, registry) = common::spawn_server_with_specs(vec![]).await; + let (mut ws, ready) = connect_claim_negotiated(&url).await; + assert_eq!( + ready["capabilities"]["terminalLifetimeClaimV1"], true, + "the negotiation rail must echo the claim capability: {ready}" + ); + + // Created-hidden: the pane never attaches. All stamps land at virtual T + // (the clock is frozen). + let tid = create_shell_terminal(&mut ws, "req-survival-1").await; + registry.set_auto_kill_idle_minutes(1); + + // The interest snapshot claims the hidden terminal (no attach). + send( + &mut ws, + json!({"type":"terminal.interest","revision":1, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[&tid]}), + ) + .await; + fence_with_ping(&mut ws).await; + assert_eq!( + registry + .claim_state(&tid) + .map(|s| (s.claimers, s.released_by_client)), + Some((1, false)) + ); + + // Two virtual minutes idle — past the 1-minute configured threshold. + freshell_platform::clock::advance_ms(2 * MINUTE_MS).unwrap(); + let killed = registry.enforce_idle_kills(); + assert!( + !killed.contains(&tid), + "a claim-only hidden terminal must survive the configured idle threshold, got {killed:?}" + ); + + // Explicit withdrawal (a later snapshot no longer claims the id): the + // terminal is re-exposed to the configured threshold... + send( + &mut ws, + json!({"type":"terminal.interest","revision":2, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[]}), + ) + .await; + fence_with_ping(&mut ws).await; + assert_eq!( + registry + .claim_state(&tid) + .map(|s| (s.claimers, s.released_by_client)), + Some((0, true)) + ); + + // ...and once past the DEV-0009 withdrawal grace, the sweep reaps it. + freshell_platform::clock::advance_ms(2 * MINUTE_MS).unwrap(); + let killed = registry.enforce_idle_kills(); + assert_eq!( + killed, + vec![tid.clone()], + "after explicit withdrawal the terminal must be reaped at the configured threshold" + ); +} + +/// A claim-connection DROP (transport loss) is NOT release: the terminal +/// stays wanted past the configured threshold (24-hour hard cap only), and +/// the hard cap stays the cleanup backstop for an abandoned claim. +#[tokio::test] +async fn claim_connection_drop_keeps_terminal_wanted_until_hard_cap() { + let _gate = GateGuard::enable(); + let (url, registry) = common::spawn_server_with_specs(vec![]).await; + let (mut ws, _) = connect_claim_negotiated(&url).await; + + let tid = create_shell_terminal(&mut ws, "req-survival-2").await; + registry.set_auto_kill_idle_minutes(1); + send( + &mut ws, + json!({"type":"terminal.interest","revision":1, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[&tid]}), + ) + .await; + fence_with_ping(&mut ws).await; + + // Transport loss: close the socket (no terminal.detach, no withdrawal). + ws.send(Message::Close(None)).await.expect("clean close"); + drop(ws); + eventually(|| match registry.claim_state(&tid) { + Some(state) if state.claimers == 0 => Some(()), + _ => None, + }) + .await; + + // Past the configured threshold, still wanted (24h hard cap only). + freshell_platform::clock::advance_ms(2 * MINUTE_MS).unwrap(); + let killed = registry.enforce_idle_kills(); + assert!( + !killed.contains(&tid), + "transport loss must not re-expose the terminal to the configured threshold, got {killed:?}" + ); + + // 25 virtual hours after the drop: the hard cap reaps the abandoned row. + freshell_platform::clock::advance_ms(25 * 60 * MINUTE_MS).unwrap(); + let killed = registry.enforce_idle_kills(); + assert_eq!( + killed, + vec![tid], + "the 24h hard cap must stay the cleanup backstop for an abandoned claim" + ); +} + +/// Claim robustness (mixed-version matrix cell 5): an interest snapshot +/// carrying `claimedTerminalIds` from a connection that did NOT negotiate +/// `terminalLifetimeClaimV1` is ignored — no claim is recorded, and the +/// terminal stays RELEASED and REAPABLE: the idle sweep reaps it at the +/// configured threshold exactly as if no claim had ever been sent. The +/// wire-level claim_state pin lives in `terminal_interest_wire.rs` +/// (`claims_from_an_unnegotiated_connection_are_ignored`); THIS test adds +/// the sweep outcome — the protection the old client never gets. +#[tokio::test] +async fn unnegotiated_claim_is_ignored_and_the_terminal_stays_threshold_reapable() { + let _gate = GateGuard::enable(); + let (url, registry) = common::spawn_server_with_specs(vec![]).await; + let (mut ws, ready) = connect_interest_only(&url).await; + assert!( + ready["capabilities"]["terminalLifetimeClaimV1"].is_null(), + "the interest-only connection must not get the claim echo: {ready}" + ); + + let tid = create_shell_terminal(&mut ws, "req-claim-robust").await; + registry.set_auto_kill_idle_minutes(1); + + // The old-shape snapshot: accepted (interest is negotiated) but its + // claim field must be ignored server-side. + send( + &mut ws, + json!({"type":"terminal.interest","revision":1, + "focusedTerminalId":null,"visibleTerminalIds":[], + "claimedTerminalIds":[&tid]}), + ) + .await; + fence_with_ping(&mut ws).await; + assert_eq!( + registry + .claim_state(&tid) + .map(|s| (s.claimers, s.released_by_client)), + Some((0, true)), + "a non-negotiated connection's claim field must be ignored" + ); + + // The ignored claim provides NO protection: the sweep reaps the hidden + // terminal at the configured threshold — it stays released/reapable. + freshell_platform::clock::advance_ms(2 * MINUTE_MS).unwrap(); + let killed = registry.enforce_idle_kills(); + assert_eq!( + killed, + vec![tid.clone()], + "an ignored claim must leave the terminal reapable at the configured threshold" + ); +} diff --git a/docs/plans/2026-09-19-responsive-terminal-restore.md b/docs/plans/2026-09-19-responsive-terminal-restore.md new file mode 100644 index 000000000..c87362a09 --- /dev/null +++ b/docs/plans/2026-09-19-responsive-terminal-restore.md @@ -0,0 +1,289 @@ +# Reliable, Responsive Pane Restore — Diagnosis and Plan + +Status: revised proposal after source review; screen reconstruction and history representation require design validation before implementation (no code changes yet) +Date: 2026-09-19 +Incident: live self-hosted server on garageserver (`192.168.3.150:3001`), Electron client on DANDESKTOP +Related: `docs/plans/2026-03-30-paginated-terminal-replay.md`, `docs/plans/2026-07-24-rust-attach-viewport.md` (TERM-07 open items) + +## User Request + +### Requested result +- Implement the committed plan `docs/plans/2026-09-19-responsive-terminal-restore.md` (reliable, responsive pane restore) starting with its own Sequencing: the shared restore contract plus the bounded recovery increment (workstreams 1 paced path, 2, and 3 together), fully tested and prepared for PR approval. + +### Explicit constraints +- Follow the plan's Sequencing section: the shared restore contract and bounded recovery increment come first; the screen-first snapshot path, on-demand history, and conversation refresh/pagination are later increments gated on their design validations and are not in this run. +- Use red/green/refactor TDD; extend existing behavior tests rather than tests that assert plan text. +- No PR creation and no live deployment; never restart the live self-hosted server without the user's explicit "APPROVED". +- Broad repo-supported test runs go through the shared coordinator gate; the worktree base must be green via `scripts/base-gate.sh`. +- Do not experiment on live user terminals to manufacture loss; use disposable test instances and fixtures. +- Older clients must not receive newly introduced retention gaps; capability negotiation gates all new restore semantics, and no healthy process is killed or replaced because replay is missing. + +### Accepted tradeoffs and residuals +- The bounded recovery increment may not meet the one-second fresh-screen latency target when no valid baseline exists; screen-first snapshots, on-demand history loading, and conversation refresh/pagination correctness remain follow-on increments per the plan's Sequencing. + +## Summary + +When a client attaches to a terminal, the server sends its entire retained output newer than the requested position and ignores the requested replay budget. In the incident, initial and repeated full hydrations amounted to roughly 21 MB across seven terminals. The server's sustained-backlog disconnect threshold is lower than its output-spill threshold, creating a reconnect loop. The client already records parser-applied progress, but an incomplete fresh-surface hydrate and other safety checks can force the next attempt back to zero. + +The fix must bound delivery without treating arbitrary terminal output as a screen snapshot. First establish a correct reconstruction and resume contract, then implement bounded delivery, resumable hydration, and backpressure handling together. Terminal history browsing and conversation pagination are separate follow-on work. Removing the conversation refresh cover also requires fixing the same-revision refresh retry condition. + +This revision incorporates source review of the existing client checkpoints, terminal queues, OpenCode recovery, and conversation refresh paths. Incident measurements below are retained from the original investigation; the review did not repeat live probes or establish how much of the network backlog was caused specifically by renderer work. + +## What the user experienced + +- The pane for `opencode --session ses_f5434cdf6ffeZYNTjLVLre77W8` never finished resuming; it appeared hung behind a "Refreshing conversation…" cover. +- Restarting the Electron client did not help: the reconnect/disconnect loop resumed immediately. +- Other panes in the same app showed similar restoring/refreshing symptoms because they share one connection. + +The "Refreshing conversation…" cover is implemented by `FreshAgentView`, whereas terminal replay is handled by `TerminalView`. Treat these as separate recovery paths observed in the same incident; record pane IDs and content types in the reproduction rather than attributing the conversation overlay directly to terminal replay. + +## Diagnosis + +### The failure loop + +1. Initial connection and subsequent hydration schedule terminal attachments across the layout (7 terminals plus 7 agent-conversation panes in the incident). Existing background hydration/rebind queues already stage some work; this is not one unconditional synchronous attach of every pane. +2. For each zero-position terminal hydrate, the server resends the terminal's full retained past output ("replay"). Measured ~3 MB per pane. +3. The client asks for only the newest 128 KB per normal terminal pane. The server ignores that request: the field (`maxReplayBytes`) exists in the protocol but no server code reads it. Verified by attaching with a 128 KB request and still receiving ~3 MB. +4. Full-screen terminal apps (opencode) request no limit at all, by design. +5. The observed server output backlog is ~21–25 MB and remains above 16 MB for the required 10 seconds. The writer measures queued plus in-flight terminal output, not xterm's parsing progress. Color-heavy replay and large HTTP conversation snapshots are additional client work, but HTTP snapshot bytes are not part of that terminal queue measurement. +6. The server disconnects with code 4008 ("catastrophic backpressure"). Reconnect can force another fresh-surface hydrate, which resets the usable replay position despite some output having reached the parser; the observed attempts do not converge. +7. The server's graceful fallback — spill the oldest unsent output — only runs at 32 MB, above the 16 MB disconnect threshold, so it never gets a chance. + +### Evidence + +| Measurement | Value | +| --- | --- | +| Resend for terminal `c1d8efc…` (`ses_f5434…`) | 2,996,228 bytes / 7,491 frames per attach | +| Resend for the other OpenCode terminal (`8ff6bee0…`) | 3,017,089 bytes / 18,805 frames | +| Resend for a Codex terminal (`a6af3bdf…`) | 2,999,964 bytes / 14,487 frames | +| Resend with `maxReplayBytes: 131072` requested | 3,000,000 bytes (cap ignored) | +| Bytes written by the opencode TUI processes over 3 s | ~0 (no live output during the incident) | +| Server queue at disconnect | 21–25 MB vs 16,777,216-byte threshold | +| Disconnect cadence | every 15–20 s, repeating; 28 closures on Sep 18 and continuing Sep 19 | +| Agent-conversation snapshot | 6,015,324 bytes; 6.5 s to fetch for the active pane | + +Supporting source locations: + +- Attach handling: `crates/freshell-ws/src/terminal.rs:6916` (`handle_attach`) → `crates/freshell-terminal/src/registry.rs:1594` (`attach_to_shared`); the full-replay snapshot is taken at `registry.rs:1616-1621`. +- Ignored budget field: `crates/freshell-protocol/src/client_messages.rs:371` (`max_replay_bytes`); no reader anywhere in `crates/freshell-ws` or `crates/freshell-terminal`. +- Client budget request: `src/components/TerminalView.tsx:221` (`TRUNCATED_REPLAY_BYTES = 128 * 1024`) and `:278` (`viewportHydrateReplayOptions`, which returns no limit for opencode). +- Client handling of a budget gap: `src/components/TerminalView.tsx:4335` (`replay_budget_exceeded` → "load more" affordance). +- Thresholds: `crates/freshell-terminal/src/output_queue.rs:22` (queue spill at 32 MB), `crates/freshell-ws/src/backpressure.rs:68-75` (disconnect at 16 MB sustained 10 s), monitor at `crates/freshell-ws/src/terminal.rs:400-410` and close at `:549-565`. +- Graceful gap emission today: `crates/freshell-ws/src/connection_writer.rs:490-498` (queue overflow only; no gap is emitted when the requested position predates the retained history). +- Agent-pane cover: `src/components/fresh-agent/FreshAgentView.tsx:3360` and `:3742` (full-pane overlay while `snapshotDirty`). +- Legacy budget behavior: `server/terminal-stream/broker.ts:454-481` at commit `a7d36d5f8^` (budget walk newest-to-oldest, `replay_budget_exceeded` gap); the Rust port leaves `maxReplayBytes` unimplemented per the TERM-07 note in `docs/plans/2026-07-24-rust-attach-viewport.md`. Restoring the tail selection alone is insufficient for screen correctness and resumable progress. + +### Why it cannot recover + +Three design choices combine into a permanent loop: + +1. A full hydrate sends the entire retained ring because its budget is ignored; an already-valid delta attach does honor `since_seq`. +2. An incomplete fresh-surface hydrate can force another full hydrate. Furthermore, reporting a skipped prefix through the existing gap handler prevents the parser-applied checkpoint from advancing past that prefix. +3. Sustained terminal backlog can trigger disconnect before spilling begins, even if socket sends are still making progress. + +Fixing only the ignored budget shrinks the burst but leaves (2) and (3) able to reproduce the loop under other load. + +### Not the cause + +- The OpenCode session is healthy: idle, last turn completed 17:21 UTC, no error state. +- The terminal programs are not flooding: their own writes were ~0 bytes during the incident. The traffic is entirely resent history. +- The server is up and serving; this is not a crash or restart problem. +- The client/server build mismatch is real but handled correctly (the one-shot reload guard suppresses further reloads). It is unrelated. + +### How to reproduce + +For future reproduction, use a disposable test instance populated with equivalent scrollback, not live user terminals. Open a WebSocket to `/ws`, complete the `hello` handshake (protocol version 10, token), and send `terminal.attach` for a terminal with a full scrollback, with and without `maxReplayBytes`. Count the payload bytes of the `terminal.output.batch` messages for five seconds. The original incident probe observed: + +- Without the field: ~3 MB arrives in under a second, then silence. +- With `maxReplayBytes: 131072`: the same ~3 MB arrives — proof the server ignores the request. + +To reproduce the user-visible loop, open the app with several full-scrollback terminal panes and watch the server log for `ws.terminal_stream.catastrophic_close` followed by `ws.connection.closed reason=catastrophic_backpressure code=4008`, then a fresh `ws.connection.established` — repeating. + +## Goals + +What the person at the keyboard should get: + +1. Opening Freshell with many panes: target an interactive, correct active screen in about a second on the incident hardware. Measure this; a bounded network payload alone does not establish responsiveness. +2. Hidden panes do not hydrate terminal screens or fetch conversation bodies until revealed. Cheap lifecycle, ownership, and status subscriptions remain available. +3. Earlier retained history is available in bounded pages without disturbing the live terminal; distinguish unloaded history from history that has expired. +4. A dropped connection keeps what was already shown. With compatible terminal state and retained missing output, reconnect resumes from applied progress. If reconstruction is impossible, preserve the view and show an explicit recovery state instead of silently looping or restarting a healthy process. +5. No pane is covered by a full-screen "refreshing" curtain while it has content to show. + +## Plan + +### Shared restore contract — implement before enabling truncation + +Terminal output is a stateful instruction stream. An escape-sequence boundary is not necessarily a point from which a blank terminal can reconstruct cells, cursor position, text attributes, scroll regions, or alternate-screen state. Existing `terminal.modes.sync` restores mode flags only. The server currently has no canonical screen snapshot that proves a retained suffix is independently renderable. + +Use these distinctions throughout server responses, client state, history controls, and tests: + +| Condition | Meaning | Required client behavior | +| --- | --- | --- | +| Intentionally unloaded older history (`replay_budget_exceeded`) | Older bytes remain fetchable, and the current screen has an independently validated baseline | Show a history boundary; do not invalidate that baseline merely because older history is unloaded | +| Retention loss (`replay_window_exceeded`) | Requested bytes have expired | Mark the unavailable span honestly; resume only from a replacement valid baseline, otherwise retain the visible surface and show recovery state | +| Delivery loss (`queue_overflow`) | This connection missed sequenced output | Repair from retained output or a valid screen snapshot; never silently advance the applied cursor across the gap | + +The existing `onOutputGap` records every span as a lost range and `markParserAppliedSeq` cannot cross it (`src/lib/terminal-attach-seq-state.ts`). Thus emitting a budget gap for `1..N` currently leaves a fresh surface's applied cursor at zero even after its retained tail is rendered. Add explicit baseline coverage and history availability to the state model; do not fix this by treating missing parser input as applied. A raw tail without a validated baseline remains incomplete, even if it looks plausible. + +Scope all restore state to terminal ID, stream ID, server identity/boot, surface generation, attach generation, and compatible geometry. Responses must identify requested/effective sequence bounds, current head, oldest retained sequence, and any lost interval. Define additive capability negotiation for new reconstruction, continuation, and gap semantics before enabling them. Older clients must not receive newly introduced retention gaps that enter their automatic OpenCode replacement path; retain compatible behavior until negotiation selects the new contract. + +Evidence for these constraints: `TerminalView.tsx` (`completeParserAppliedFrame`, `attachTerminal`, `beginOpenCodeReplacementAfterExit`); `src/lib/terminal-surface-checkpoint.ts`; `test/unit/client/lib/terminal-attach-seq-state.test.ts` (lost-prefix checkpoint remains zero); `crates/freshell-terminal/src/registry.rs` (`apply_attach_geometry`, `attach_to_shared`, reader ingestion); `crates/freshell-terminal/src/mode_tracker.rs`; and the append-only limitation recorded in `docs/plans/2026-03-30-paginated-terminal-replay.md`. + +### Workstream 1 — Bound restore delivery and establish a correct screen (server + client) + +Deliver in two increments. Paced replay fixes the unbounded burst; a screen snapshot is needed to make fresh TUI restoration independent of history size. The first increment must not be presented as meeting the final one-second screen-first goal. + +1. **Paced forward restoration.** Negotiate a continuation/credit capability (proposed name: `pacedTerminalReplayV1`) that sends bounded, ascending sequence batches from a compatible applied cursor or a proven fresh-surface baseline. Continuation acknowledgements carry terminal, stream, attach generation, and the last fully consumed sequence; stale acknowledgements or values beyond the sent window cannot grant credit. Consumption credit and the reconstructible parser-applied checkpoint are distinct and must not be conflated: the client's ordered write queue consumes every frame — each frame is either applied by xterm or deliberately filtered by a byte-mutating parser (startup probes, OSC52, completion signals) that by design never reaches xterm, and a filtered frame must not stall continuation. Credit is granted only after the prior batch is consumed in order; the parser-applied checkpoint keeps its strict semantics and never advances across a filtered or lost range. A pinned checkpoint must not strand an interrupted restore: define a reconstruction-safe surface-coverage cursor, separate from the strict parser-applied sequence, that advances past a filtered range only when the applied filter class has a proven null screen effect (startup-probe suppression, OSC52 handling, completion-signal consumption); any unknown mutation leaves the coverage cursor pinned and keeps the quarantine path. Delta resumes and partial-hydrate continuation request their since position from the coverage cursor, so an interrupted restore after filtered or mixed pages neither duplicates already-rendered output nor forces a full baseline rebuild, and the checkpoint carries the coverage cursor through the same identity/geometry/authority validation as the applied sequence. Tests must cover a page containing only filtered frames and a mixed page — both credit continuation, the checkpoint stays pinned at its pre-filter position — plus disconnect-and-resume across both page shapes: the resumed surface converges to the uninterrupted reference with no duplicate writes and no spurious baseline recovery. Keep a fixed initial catch-up target so ongoing live output cannot move the completion condition indefinitely; subsequently deliver live output in sequence order. Do not let live frames for the same terminal overtake unconsumed replay. Input and other panes' traffic remain schedulable. +2. **Validated current-screen restoration.** Prototype a server-maintained terminal emulator/checkpoint, updated in PTY output order, that can reconstruct both normal and alternate buffers plus cursor, attributes, modes, and geometry. A snapshot carries its stream/geometry identity and covered sequence; snapshot capture and subscription installation must establish a gap-free snapshot→delta handoff. Compare its output against the real client parser before selecting the implementation. Do not add an emulator dependency solely on the assumption that it is xterm-compatible. + +The paced path can reconstruct exactly only if its starting state and required bytes are available. The retained ring may already have lost essential state, so replaying its entire remaining tail is not a proof of correctness either. In that case use a validated snapshot when available, or show an explicit incomplete-screen state with retry; do not restart the program automatically. A newly introduced server emulator cannot retroactively reconstruct missing history for an already-running stream: mark its initial baseline unknown until a verified reset/repaint boundary, or use the explicit unavailable state. + +Implementation requirements: + +- Thread `max_replay_bytes` through both geometry-authorized and geometry-skipped attach paths in `handle_attach` and the registry. Preserve the legacy field's serialized application-JSON tail-budget meaning (confirmed in `broker.ts:454-481` at `a7d36d5f8^`); introduce negotiated forward-page limits rather than silently redefining it as a continuation protocol. Apply newest-tail selection only where the snapshot/baseline contract proves it correct. +- Specify terminal-data UTF-8 bytes separately from serialized message bytes. The existing replay-frame byte count is not the full JSON wire size. Enforce a serialized delivery budget including escaping, batch metadata, and envelope; account for ready/gap/mode controls separately under bounded control limits. Cover multibyte text, JSON escaping, empty replay, zero/undersized budgets, and a frame larger than the budget. Return a bounded explicit result rather than sending an oversized frame or an endless zero-progress page; any frame fragmentation needs sequence plus offset semantics and reassembly tests. +- Bound both per-pane unacknowledged replay and total per-connection admitted bytes, including snapshot/replay copies and the frame being sent. Select pages before cloning payloads; do not first allocate one full replay copy per pane. Cancellation, superseded attaches, and disconnect release credits and temporary state. +- Do not pin an unbounded replay backlog for a slow client. If retention overtakes a continuation cursor, report the exact lost interval and enter bounded baseline recovery. A finite retention window cannot guarantee convergence against indefinitely faster output production. +- Remove the proposed resize nudge. Attach resize and replay capture currently share the terminal lock needed by the output reader; an immediate capture cannot include the asynchronous repaint, and waiting under that lock prevents ingestion. Same-size resize is a no-op, and another subscriber may own geometry. Any future repaint protocol must await a verifiable result outside that lock, respect geometry ownership, and fall back on timeout; it is not a prerequisite for the paced path. +- Hidden panes do not request screen hydration until revealed. Preserving terminal lifetime is a server-side claim, not an inference: `TerminalShared::released_by_client` starts `true` and only a successful attach clears it, and the idle reaper (`enforce_idle_kills`) applies the configured idle threshold only to `released_by_client` rows — never-attached hidden panes survive today only because hidden hydration attaches them. Removing hidden hydration therefore requires a negotiated, non-hydrating per-connection lifetime claim (additive optional field on the already-negotiated `terminal.interest` presentation snapshot, e.g. a `claimedTerminalIds` set) that marks claimed terminals wanted with release semantics mirroring attach exactly: claim membership is per-connection and sweeps away with its socket, but transport loss is not release — a dropped connection must keep the durable wanted state, exactly as `remove_connection` never restores fast reapability today and only an explicit `terminal.detach` of the last reference does. The only path back to configured-threshold reapability is an explicit release action: claim withdrawal, or the existing detach reconciler acting when the terminal leaves every pane layout. The claim never grants replay or output delivery and does not touch geometry or stream identity. A client whose server did not negotiate the claim field keeps today's hidden attach (`keepalive_delta`) as its lifetime claim — compatibility outranks the hidden-pane optimization. Preserve cheap lifecycle/status subscriptions separately from output delivery and terminal lifetime. Do not use ordinary `terminal.detach` as a visibility toggle without checking its release/reap semantics. Remove hidden hydration-queue grants rather than sending a nominal zero budget that the current truthy serializer omits. On reveal, use a valid retained surface checkpoint or request a fresh baseline. + +Tests: + +- Registry/WS: byte budgets, continuation bounds, fixed catch-up target, snapshot→live ordering under concurrent output, expired continuation, cancellation, and bounded aggregate memory with many panes. Hidden-pane lifetime: a created-hidden and a reconnect-hidden terminal held only by the negotiated lifetime claim survives the configured idle-kill threshold (the claim clears `released_by_client` without attaching or delivering replay); a claim-connection drop (transport loss) keeps the terminal wanted like any attached-then-disconnected terminal (24-hour hard cap only); an explicit claim withdrawal or detach of the last reference re-exposes the terminal to the configured threshold. +- Client/WS: withhold parser callbacks to prove replay credit does not advance on receipt; release callbacks and prove forward progress without duplicate output or same-terminal reordering. +- Reconstruction: compare visible cells, cursor, attributes, modes, and subsequent input behavior against an uninterrupted xterm reference using shell, OpenCode, Codex, and Claude recordings, plus real TUI smoke tests. Include quiet/no-repaint output, split control sequences, alternate-buffer transitions, different geometry, second viewers, and screen representations larger than the nominal budget. +- Compatibility: new/old client-server pairs select supported behavior; lack of capability never causes an automatic kill or claims a raw suffix is a complete screen. + +### Workstream 2 — Resume instead of restarting (client + server) + +Implementation notes: + +- Reuse the existing parser-applied checkpoint machinery. `completeParserAppliedFrame` saves progress after xterm's write callback; receipt/highest-observed sequence is not sufficient. The defect is that `surfaceFreshRef` can override an otherwise usable partial checkpoint and force zero replay, not that progress recording is entirely absent. +- Represent incomplete hydration separately from a newly created blank surface. A compatible partial hydrate can resume on the same mounted xterm without clearing it, replaying a mode preamble, or claiming `surfaceReset` again. After filtered or mixed pages the resume position comes from Workstream 1's surface-coverage cursor, never from a checkpoint the filter pinned below already-rendered content. Applying a complete validated snapshot establishes its covered-sequence baseline only after the parser applies it; incomplete snapshot installation must not be mistaken for resumable terminal deltas. +- Retain `canUseCheckpointForDeltaReplay` checks for terminal/stream/server identity, surface generation, geometry and authority, scrollback, parser readiness, and xterm version. Recreated surfaces, changed streams, or incompatible geometry need a new valid baseline, even if the old sequence is nonzero. Scope checkpoints to the actual surface so sibling panes showing the same terminal cannot borrow each other's rendered progress. +- When writes are in flight, freeze generation transitions and wait boundedly for the owned queue to drain before deciding whether its applied checkpoint remains usable. Do not clear a surface while an earlier write can still mutate it. If current quarantine rules have already invalidated that generation, preserve quarantine and rebuild from a valid baseline rather than accepting stale callbacks. +- Implement reason-aware gaps from the shared contract. In particular, remove the `replay_window_exceeded` → `beginOpenCodeReplacementAfterExit` automatic kill path before enabling new server retention-gap notifications. Restore gaps must not emit `terminal.kill`, change terminal identity, or spawn a replacement for a healthy process. Any explicit restart action remains separate user intent. +- Bound automatic recovery by attempts and lack of applied progress. Reset recovery accounting on genuine parser progress or explicit retry, not merely on receiving `attach.ready` or another reconnect. Preserve visible content and show an accessible failure/retry state when the limit is reached. + +Tests: + +- Interrupt a hydrate after some xterm callbacks: same surface requests only the remainder, without clear/reset/preamble, and converges to the uninterrupted reference. Include disconnect-and-resume across a filtered-only page and a mixed page (Workstream 1's coverage-cursor case): no duplicate writes, no spurious baseline recovery, convergence to the reference. +- Interrupt before a callback, delay old callbacks across recovery, and independently change each checkpoint identity/geometry field: no false progress, duplicate writes, or stale checkpoint acceptance. Page reload and sibling surfaces cannot reuse a cursor without their corresponding state. +- Start with a validated bounded screen and unloaded history, apply live output, reconnect: the applied cursor advances while the history boundary remains. A true delivery gap still blocks unsafe advancement. +- A healthy idle OpenCode terminal with expired retained output reports incomplete history/screen state as appropriate; terminal ID and process remain unchanged and subsequent live output still arrives. Repeat for a queue gap; separately verify the compatible path for an older unnegotiated client. +- Recovery with no progress reaches a visible retry state and does not enter an automatic kill/recreate or reconnect loop. + +### Workstream 3 — Spill old output instead of hanging up (server) + +Implementation notes: + +- Set consistent byte limits so normal output pressure reaches bounded admission/spill before any pressure-related disconnect. Include the leased in-flight frame; `DeliveryQueue::outstanding_bytes` already charges reserved bytes. Keep independent hard bounds on metadata/control mailboxes. Validate environment overrides so a configuration cannot silently restore the 32 MB spill / 16 MB disconnect mismatch. +- Replace the sustained-byte-count-only disconnect decision for healthy draining clients. Preserve the existing independent socket-send timeout and keepalive failure handling; slow consumption alone is not a dead socket. If tracking drain progress, count successful sends, not falling queue size: eviction and superseded attachments also reduce queue bytes. Continuation credit from workstream 1 governs restore production independently of socket liveness. +- Keep ordered, generation-scoped overflow gaps and non-evictable sequenced controls. A spill bounds memory but does not repair a terminal screen: route the loss through workstream 2's repair contract. Do not exempt replay from memory accounting or silently drop control frames. +- Reuse existing focused/visible/background byte-fair scheduling. Keep input, completion/status, and other panes responsive while a terminal restores; avoid a second competing priority scheduler. + +Tests: + +- A slow consumer that continues making sends stays connected under sustained replay pressure; ordinary replay is paced and overflow produces an exact gap when necessary. +- A genuinely blocked send and a missing keepalive response still close the connection; eviction alone cannot masquerade as send progress. +- Queue + in-flight + restore-state memory and control/metadata counts stay bounded under a burst larger than all limits. Include an oversized indivisible frame and conflicting environment overrides. +- A gap followed by repair yields the correct screen; focused-pane input and sequenced exit/status delivery survive other panes' backlog. + +### Workstream 4 — Load earlier history on demand (client + server) + +Implementation direction: + +- Use a separate read-only history surface within the pane, with a clear return-to-live control. xterm has no supported prepend operation; `handleLoadMoreHistory` currently clears and reattaches from zero. Do not feed backwards pages into the live terminal or its forward-only sequence/cursor state. +- Before implementation, select a historical content representation: independently renderable normal-buffer rows from the validated server emulator/checkpoint machinery, or pages reconstructed from a retained parser checkpoint plus subsequent output. Raw ANSI fragments alone are not independently renderable history. Bound checkpoint/row retention alongside replay retention; no unlimited archival store is included in this change. Do not manufacture older content that was never retained. +- Add a read-only, byte-bounded earlier-history endpoint with a terminal/stream-scoped cursor, exclusive before-position, oldest available position, and explicit exhausted/expired outcomes. Prefer the existing authenticated HTTP infrastructure for this separate read path; it must not attach, resize, claim geometry, change subscription generation, or alter the live applied cursor. Keep concurrency low and visible demand explicit. +- Preserve the history view's stable row/turn anchor when inserting older pages. Live terminal output continues on its own surface; returning to live reveals its current screen. Alternate-screen TUI redraw streams are not a chronological transcript: expose only meaningful retained normal-buffer history or clearly state that earlier history is unavailable. +- The affordance says "Load earlier" only while a valid continuation exists. If the ring advances while browsing, show an expired boundary and stop requesting that cursor instead of triggering full reattachment. + +Tests: + +- E2E: enter history, load several bounded pages while the terminal produces live output, preserve the history scroll anchor, and return to an unchanged live stream identity and correct current screen. No `terminal.attach`, resize, or cursor reset is emitted by a history fetch. +- Integration: cursor expiry, stream replacement, Unicode/control-sequence boundaries, alternate-screen history unavailability, and oversized content all produce bounded, honest results with no repeated zero-progress fetches. + +### Workstream 5 — Stop covering panes that have content (client + server) + +Split this work into a refresh-correctness change and a subsequent pagination change. The former does not need to wait for a new transcript format. + +Refresh correctness: + +- Replace the full-pane overlay in `FreshAgentView.tsx` with the last known conversation plus a small accessible refresh/error status and retry control. Preserve composer focus, selection, and scroll; an empty first load can still show a loading state. +- Fix the dirty-state condition, not just its presentation. Hidden reconnect calls `markSnapshotDirty`, capturing the current revision, but reveal currently accepts only a strictly greater revision (`FreshAgentView.tsx`, `markSnapshotDirty`, reconnect effect, and `revealRevisionIsFresh`). An unchanged idle conversation can therefore retry every 250 ms until error despite successful GETs. +- Track why a snapshot is dirty and the invalidation generation. For transport-only uncertainty, accept an authoritative same-revision response from a request started after that invalidation, with matching thread, owner, and request identity and no newer invalidation. A lower revision or response started before the invalidation cannot clear it. For a known mutation, require the advertised minimum revision or mutation-specific acknowledgement; do not globally replace `>` with `>=` and accidentally accept pre-mutation content. +- Retain the shared snapshot scheduler's single-flight, trailing refresh, and 429 backoff behavior. Hidden polling and event-triggered fetch suppression already exist; close the initial identity-fetch gap and add execution-time visibility demand to queued work. A hidden pane must not abort a shared request needed by a visible sibling. If no consumer remains visible, skip unstarted body fetches; a request already in flight may finish without scheduling follow-up work. +- Make freshness refer to actual scheduler execution, not just a caller's render/effect time. Preserve the existing rule that a request scheduled during an in-flight GET receives a trailing execution. Define visible-demand registration per snapshot key so the latest hidden caller cannot suppress work required by a visible consumer. Provider revisions are not one global sequence: audit each provider's revision and successful-response ownership metadata before defining mutation freshness, adding missing contract fields where necessary. +- Preserve lightweight fresh-agent attachment and status delivery for hidden panes: these support ownership, approvals, and completion notifications and are not transcript hydration. + +Pagination contract: + +- Reuse or extend `FreshAgentTurnPageSchema` and `FreshAgentTurnBodySchema` in `shared/fresh-agent-contract.ts` and the turns/body query and stale-revision schemas in `shared/read-models.ts`; their existence is not evidence of implemented Rust paging routes. `crates/freshell-freshagent/src/snapshot.rs` currently serves the full snapshot route. Define typed query validation and exhausted/stale/invalid-cursor responses, then update Rust responses, strict client schemas, API helpers, and consumers together. The existing whole-turn body schema cannot represent item previews or partial bodies without an explicit extension. +- Return current status, capabilities, approval/question state, revision, and a bounded recent-turn page. Older pages have stable turn/item IDs and a cursor scoped to thread identity and transcript revision. Explicitly report exhausted or invalidated cursors. Decide capability negotiation before changing the existing full-snapshot response consumed by older clients and the settings UI. +- Enforce serialized UTF-8 response-byte limits, including metadata and `rolledBackTurns`, not just a turn count. Large text, tool results, commands, attachments, and extension bodies need explicit previews plus lazy, bounded body/item pages. Never split JSON or pretend a truncated tool result is complete. A single oversized turn must still make progress through its body cursor. Keep approval/question semantics intact; if essential control metadata itself exceeds the supported budget, return an explicit bounded error instead of silently omitting controls. +- Replace whole-transcript replacement with revision-aware page storage. `mergeSnapshotForDisplay` currently replaces all turns on idle snapshots, which would discard loaded older pages. Define refresh page replacement, deduplication, and invalidation when revision changes; retain older pages only when continuity is established. Rollback/redo/fork must invalidate affected pages and bodies so a late response cannot resurrect removed turns. Treat rolled-back markers as separately paged history while retaining current redo metadata. +- Keep optimistic sends, send acknowledgements, and pending approvals separate from page membership. Absence from a partial recent page is not proof that a submitted turn was rejected or removed. Audit `localEchoLanded` / `shouldClearStaleLocalEcho` and the outgoing-turn completion path for that assumption. +- Bound server work as well as response size: prefer provider paging where available; otherwise reuse a bounded per-thread normalized index/cache. Do not rebuild and serialize a 6 MB transcript for every small page or every sibling pane. Measure provider-specific time and memory before claiming the one-second goal. + +Tests: + +- Hide → reconnect → reveal an unchanged idle conversation: one current same-revision response clears transport dirtiness; an older in-flight response or a response racing a new invalidation does not. +- A reveal scheduled while an earlier GET is in flight receives a fresh trailing execution. Hiding during debounce, 429 backoff, or an active shared request neither starts unnecessary follow-up work nor blocks a visible sibling. +- Refresh failure and 429 leave existing content usable; retry converges without a full-pane cover or focus/scroll loss. +- Hidden initial mount, poll, event, and queued refresh do not start body fetches; a visible sibling sharing the key still receives its response. Hidden status/completion/approval updates continue. +- Load older pages, then refresh while idle and while busy: no duplicates or unintended history loss; rollback/redo/fork followed by a late page/body response does not resurrect stale turns. +- A single multi-megabyte tool result, a large rolled-back history, and an optimistic turn outside the recent page exercise the real HTTP-to-client path. Assert bounded page/body bytes, honest continuation, and correct send state. + +## Sequencing + +1. **Contract and reconstruction validation.** Specify baseline coverage, continuation credit, capability negotiation, loss outcomes, and screen comparison fixtures. Prototype the current-screen representation and record supported/unsupported terminal state. Select the history representation before implementing workstream 4. A failed prototype is a design blocker for screen-first delivery, not permission to ship arbitrary tail reconstruction. +2. **Bounded recovery increment: workstreams 1 (paced path), 2, and 3 together.** Implement and test partial-hydrate checkpoints, removal of automatic OpenCode gap kills, bounded replay production, gap repair, and liveness limits as one coherent behavior change. A red base gate blocks implementation per repository rules. This increment can improve connection stability without yet meeting fresh-screen latency targets when no valid baseline exists. +3. **Screen-first increment: workstream 1 snapshot path.** Enable only after the reconstruction gate passes for the supported terminal modes and snapshot→delta handoff. Measure actual first-screen/input latency on the incident layout. A capability or supported-state fallback must remain explicit and non-destructive. +4. **History increment: workstream 4.** Implement its chosen content model and separate history view after the baseline machinery is established. Update `docs/index.html` for the significant UI change and README usage guidance. +5. **Conversation increments: workstream 5.** Refresh correctness and visibility scheduling can proceed independently of terminal work. Paginated transcripts follow their strict page/body/merge contract and provider validation; update the UI mock and README when the new history controls land. + +Use red/green/refactor for implementation, extending existing behavior tests rather than tests that assert plan text. Changes are authored in dedicated worktrees from a freshly checked green base; coordinate broad tests and use the configured backends. No PR creation or live deployment is authorized by this plan. Prepare and verify each increment, then obtain the user's explicit PR approval under repository rules. + +## Acceptance criteria + +Correctness and resource limits (required for the bounded recovery increment): + +- Every rendered current screen is backed by a compatible applied checkpoint or validated reconstruction. Incomplete/unavailable state is labeled honestly; no healthy process is killed or replaced because replay is missing. +- Cutting the network mid-restore preserves the visible surface. A compatible partial restore requests only bytes after its parser-applied position; incompatible state takes the explicit baseline-recovery path. Retention expiry has a bounded failure outcome. +- Serialized batches, unacknowledged windows, per-connection restore state, writer queue/in-flight bytes, and metadata stay within declared limits. Large individual frames and snapshots have explicit continuation or bounded failure behavior, not silent over-budget sends. +- A slow, progressing consumer remains connected under terminal-output pressure; a dead socket still times out. Gap delivery and repair preserve per-terminal ordering and never advance checkpoints over missing parser input. +- Hidden panes start no terminal screen hydration while hidden; necessary lifecycle, ownership, and lightweight status subscriptions continue. Conversation body-fetch visibility belongs to workstream 5 and is not this increment's acceptance. A visibility change preserves terminal lifetime and other devices' geometry. +- Mixed-version tests prove new restore messages are negotiated and legacy OpenCode clients do not encounter newly introduced kill-triggering retention gaps. + +Final user experience (required before calling the entire plan complete): + +- On the incident layout (7 terminals + 7 agent panes), target p95 active-pane correct-screen and input-echo latency of approximately one second over repeated cold-client and reconnect trials on the incident hardware/network. Report trial count, terminal modes, retained bytes, and provider/snapshot sizes. Record main-thread long tasks and input latency during restore so "no multi-second freezes" is measured rather than inferred from payload size. +- Use 128 KiB as the initial terminal batch target, not a universal claim about total reconstruction size. Record total startup bytes separately from in-flight bytes; a full paced replay may legitimately exceed one batch, while a validated screen snapshot should avoid reading all history. Select aggregate window limits from measurements and publish them with the implementation. +- Earlier retained terminal history loads in bounded pages with a stable scroll anchor and an honest unloaded/expired boundary; live output continues independently. Returning to live shows the correct screen. +- Conversations with existing content remain usable during refresh, including same-revision reconnects and refresh failure. Recent pages and large bodies are byte-bounded; older loaded content, optimistic sends, and rollback state remain consistent. + +Use a deterministic fixture matching the incident's data sizes for repeatable integration/browser tests, then a manually observed Windows Electron smoke check. Validate affected browser specs on the configured backend and ensure they actually execute; destructive restart/kill test cases belong in the disposable sandbox. Do not experiment on live user terminals to manufacture loss. + +Observability: add structured events for restore start/completion/failure, baseline source, requested/applied sequence, raw versus serialized bytes, unacknowledged bytes, queue/in-flight pressure, gap reason, successful-send progress, retries, and first-screen/input timing. For conversation refreshes include trigger, execution generation, revision acceptance reason, page bytes, and visible demand. Log identifiers and measurements, not terminal or conversation contents. Keep diagnostic logging lightweight and debug perf logging off outside investigations. + +## Risks and open questions + +- **Screen reconstruction gate:** choose and validate the emulator/checkpoint format, state coverage, initialization for already-running streams, and maximum snapshot size. This is the main unresolved design decision; mode flags and resize nudges are not substitutes. If it cannot meet fidelity and latency requirements, revise the screen-first design before implementation of that increment. +- **History representation gate:** choose bounded normal-buffer rows or independently reconstructible pages. Define scroll anchors and retention cost; do not claim arbitrary TUI redraw history is a transcript. +- **Protocol details:** finalize the continuation-credit window (consumption credit distinct from the parser-applied checkpoint), oversized-frame handling, atomic baseline installation, capability names, and compatibility behavior. Preserve the existing `maxReplayBytes` meaning and distinguish raw payload from wire accounting. +- **Limits and retries:** select aggregate byte/metadata limits and recovery-attempt/no-progress deadlines from fixture measurements. Define explicit behavior when output production outruns finite retention; "always finishes" is not a defensible unconditional promise. +- **Provider pagination:** provider revision semantics and paging support differ. Validate stale-page handling, large-body access, pending-send reconciliation, ownership changes, and full-snapshot callers before enabling partial responses. +- **Deployment:** server changes require the approved self-hosted launch runbook and an explicit `APPROVED` before restarting the live server. Neither this plan edit nor test-instance permission authorizes that restart. +- **Base health:** the original investigation reported a red latest full-suite result. That is historical, not a current gate result; re-check `scripts/base-gate.sh` before creating an implementation worktree. This documentation revision does not require a production restart or a broad test run. + +## Appendix — incident log excerpts + +``` +WARN ws.terminal_stream.catastrophic_close pending_bytes=21754500 threshold=16777216 +INFO ws.connection.closed reason=catastrophic_backpressure code=4008 +INFO ws.connection.established +... repeating every 15-20 s +``` + +Client console during the loop: `xterm.js: Parsing error` entries while re-parsing replay data on each reconnect; the app remained open but could not finish restoring. diff --git a/port/contract/ws-message-inventory.json b/port/contract/ws-message-inventory.json index d209c534d..aa2b9398d 100644 --- a/port/contract/ws-message-inventory.json +++ b/port/contract/ws-message-inventory.json @@ -1,6 +1,6 @@ { "clientToServer": { - "count": 41, + "count": 42, "types": [ "amplifier.activity.list", "claude.activity.list", @@ -40,6 +40,7 @@ "terminal.input", "terminal.interest", "terminal.kill", + "terminal.replay.credit", "terminal.resize", "ui.layout.sync", "ui.screenshot.result" diff --git a/port/contract/ws-protocol.schema.json b/port/contract/ws-protocol.schema.json index bf5bae99f..cb6f52258 100644 --- a/port/contract/ws-protocol.schema.json +++ b/port/contract/ws-protocol.schema.json @@ -522,6 +522,10 @@ "capabilities": { "additionalProperties": false, "properties": { + "pacedTerminalReplayV1": { + "const": true, + "type": "boolean" + }, "paneReconcileFreshAgentV1": { "const": true, "type": "boolean" @@ -534,6 +538,10 @@ "const": true, "type": "boolean" }, + "terminalLifetimeClaimV1": { + "const": true, + "type": "boolean" + }, "terminalOutputBatchV1": { "type": "boolean" }, @@ -1017,6 +1025,12 @@ ], "type": "string" }, + "replayPageBytes": { + "description": "Optional upper bound on each paced replay page's serialized bytes (pacedTerminalReplayV1 connections only), clamped to the server's page-budget cap. Pages are bounded by max(requested, the atomic frame size): a single frame larger than the request forms its own atomic page, bounded by the server fragment cap.", + "exclusiveMinimum": 0, + "maximum": 9007199254740991, + "type": "integer" + }, "rows": { "maximum": 500, "minimum": 2, @@ -1055,6 +1069,15 @@ { "additionalProperties": false, "properties": { + "claimedTerminalIds": { + "items": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "maxItems": 1024, + "type": "array" + }, "focusedTerminalId": { "anyOf": [ { @@ -1346,6 +1369,43 @@ ], "type": "object" }, + { + "additionalProperties": false, + "properties": { + "attachRequestId": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "consumedSeq": { + "maximum": 9007199254740991, + "minimum": 0, + "type": "integer" + }, + "streamId": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "terminalId": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "type": { + "const": "terminal.replay.credit", + "type": "string" + } + }, + "required": [ + "type", + "terminalId", + "streamId", + "attachRequestId", + "consumedSeq" + ], + "type": "object" + }, { "additionalProperties": false, "properties": { @@ -4834,6 +4894,10 @@ "capabilities": { "additionalProperties": false, "properties": { + "pacedTerminalReplayV1": { + "const": true, + "type": "boolean" + }, "paneReconcileFreshAgentV1": { "const": true, "type": "boolean" @@ -4846,6 +4910,10 @@ "const": true, "type": "boolean" }, + "terminalLifetimeClaimV1": { + "const": true, + "type": "boolean" + }, "terminalOutputBatchV1": { "type": "boolean" }, @@ -7815,6 +7883,10 @@ "$schema": "https://json-schema.org/draft/2020-12/schema", "additionalProperties": false, "properties": { + "pacedTerminalReplayV1": { + "const": true, + "type": "boolean" + }, "paneReconcileFreshAgentV1": { "const": true, "type": "boolean" @@ -7826,6 +7898,10 @@ "terminalInterestV1": { "const": true, "type": "boolean" + }, + "terminalLifetimeClaimV1": { + "const": true, + "type": "boolean" } }, "type": "object" @@ -8042,6 +8118,12 @@ ], "type": "string" }, + "replayPageBytes": { + "description": "Optional upper bound on each paced replay page's serialized bytes (pacedTerminalReplayV1 connections only), clamped to the server's page-budget cap. Pages are bounded by max(requested, the atomic frame size): a single frame larger than the request forms its own atomic page, bounded by the server fragment cap.", + "exclusiveMinimum": 0, + "maximum": 9007199254740991, + "type": "integer" + }, "rows": { "maximum": 500, "minimum": 2, @@ -8452,6 +8534,15 @@ "$schema": "https://json-schema.org/draft/2020-12/schema", "additionalProperties": false, "properties": { + "claimedTerminalIds": { + "items": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "maxItems": 1024, + "type": "array" + }, "focusedTerminalId": { "anyOf": [ { @@ -8749,6 +8840,44 @@ ], "type": "object" }, + "TerminalReplayCreditSchema": { + "$schema": "https://json-schema.org/draft/2020-12/schema", + "additionalProperties": false, + "properties": { + "attachRequestId": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "consumedSeq": { + "maximum": 9007199254740991, + "minimum": 0, + "type": "integer" + }, + "streamId": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "terminalId": { + "maxLength": 512, + "minLength": 1, + "type": "string" + }, + "type": { + "const": "terminal.replay.credit", + "type": "string" + } + }, + "required": [ + "type", + "terminalId", + "streamId", + "attachRequestId", + "consumedSeq" + ], + "type": "object" + }, "TerminalResizeSchema": { "$schema": "https://json-schema.org/draft/2020-12/schema", "additionalProperties": false, diff --git a/port/contract/ws-server-messages.schema.json b/port/contract/ws-server-messages.schema.json index 4643e0c3c..e71eba596 100644 --- a/port/contract/ws-server-messages.schema.json +++ b/port/contract/ws-server-messages.schema.json @@ -2922,6 +2922,9 @@ "capabilities": { "additionalProperties": false, "properties": { + "pacedTerminalReplayV1": { + "type": "boolean" + }, "paneReconcileFreshAgentV1": { "type": "boolean" }, @@ -2930,6 +2933,9 @@ }, "terminalInterestV1": { "type": "boolean" + }, + "terminalLifetimeClaimV1": { + "type": "boolean" } }, "type": "object" @@ -3951,11 +3957,17 @@ "headSeq": { "type": "number" }, + "oldestRetainedSeq": { + "type": "number" + }, "replayFromSeq": { "type": "number" }, "replayResetReason": { - "const": "geometry_authority_unknown", + "enum": [ + "geometry_authority_unknown", + "retention_lost" + ], "type": "string" }, "replayToSeq": { @@ -5151,11 +5163,18 @@ "fromSeq": { "type": "number" }, + "headSeq": { + "type": "number" + }, + "oldestRetainedSeq": { + "type": "number" + }, "reason": { "enum": [ "queue_overflow", "replay_window_exceeded", - "replay_budget_exceeded" + "replay_budget_exceeded", + "handoff_boundary_reached" ], "type": "string" }, @@ -5306,9 +5325,9 @@ }, "reason": { "enum": [ + "retention_lost", "new_pty_session", "codex_pty_recovery", - "retention_lost", "server_restart_incompatible_retention" ], "type": "string" diff --git a/port/oracle/harness/normalize.ts b/port/oracle/harness/normalize.ts index b7aa782bf..b82f2d9eb 100644 --- a/port/oracle/harness/normalize.ts +++ b/port/oracle/harness/normalize.ts @@ -523,6 +523,75 @@ export function normalizeTranscript( return new Normalizer(opts).normalize(transcript) } +// ── field-scoped envelope shape masking (task-010) ────────────────────────── + +/** Stable tag per family, matching the transcript placeholder convention. */ +const SHAPE_FAMILY_TAG: Record = { + id: 'ID', + timestamp: 'TS', + seq: 'SEQ', + port: 'PORT', + path: 'PATH', + opaque: 'OPAQUE', +} + +/** + * The T1 PTY-envelope flake (task-010, task-008 review F1): `terminal.attach.ready`'s + * raw seq values depend on whether the spawned shell's banner bytes landed in the + * retained ring before the attach snapshotted it — a contract-valid, inherently + * timing-dependent race. The transcript normalizer's value-dedup placeholders + * make a single-message comparison sensitive to WHICH fields happen to share a + * raw value (the coincidence partition flips with the race), so two boots of + * identical code produced different "normalized" envelopes. + * + * This mask replaces every registered nondeterministic LEAF with a stable + * per-(family, field) placeholder — a PURE function of the envelope's + * structure. Presence, nesting, array lengths, and every deterministic contract + * value (type, enums, booleans) survive verbatim, so a REAL structural + * divergence still differs, while any two boots — whatever the race produced — + * canonicalize to the same string. Used ONLY for single-message cross-boot + * envelope comparison (T1); the transcript-diff lanes keep + * [`normalizeTranscript`]'s value-dedup semantics, which are load-bearing for + * cross-message reference tracking. + */ +export function maskEnvelopeShape(parsed: unknown): string { + return stableStringify(maskShapeValue(parsed)) +} + +function maskShapeValue(value: unknown): unknown { + if (Array.isArray(value)) return value.map(maskShapeValue) + if (value && typeof value === 'object') { + const obj = value as Record + const out: Record = {} + for (const key of Object.keys(obj)) { + const spec = FIELD_FAMILIES[key] + if (spec && isLeaf(obj[key])) { + out[key] = maskShapeLeaf(spec.family, key, obj[key]) + } else if (spec && Array.isArray(obj[key]) && (obj[key] as unknown[]).every(isLeaf)) { + // Arrays of registered leaves (e.g. `recoverableTerminalIds`) keep + // their element COUNT (a real structural property) with every + // element masked to the same per-field placeholder. + out[key] = (obj[key] as Array).map((el) => + maskShapeLeaf(spec.family, key, el), + ) + } else { + out[key] = maskShapeValue(obj[key]) + } + } + return out + } + return value +} + +function maskShapeLeaf(family: Family, field: string, value: string | number | boolean | null): unknown { + if (value === null) return value + if (typeof value === 'string' && isPlaceholder(value)) return value // idempotent + // Booleans under a registered name are deterministic (parity with the + // transcript normalizer's leaf rule); everything else is the family tag. + if (typeof value === 'boolean') return value + return `<${SHAPE_FAMILY_TAG[family]}:${field}>` +} + /** * The full canonical string form of a normalized transcript — direction-tagged, * key-sorted, newline-delimited — suitable for persisting as a golden baseline. diff --git a/port/oracle/harness/pty-capture.ts b/port/oracle/harness/pty-capture.ts index 73a898761..e8bb475d3 100644 --- a/port/oracle/harness/pty-capture.ts +++ b/port/oracle/harness/pty-capture.ts @@ -1,6 +1,6 @@ import { createHash, randomUUID } from 'node:crypto' import { WsCaptureClient } from './ws-capture-client.js' -import { normalizeTranscript } from './normalize.js' +import { maskEnvelopeShape } from './normalize.js' import type { ExternalServerHandle } from './external-server.js' import type { PtyScenario } from '../fixtures/pty-scenarios.js' @@ -115,9 +115,14 @@ export interface PtyCaptureResult { /** streamId from terminal.attach.ready (raw). */ streamId: string /** - * The `terminal.created` + `terminal.attach.ready` envelopes with - * nondeterministic fields (ids/seqs) normalised — stable across boots, so the - * envelope shape can be compared even though the golden is compared by bytes. + * The `terminal.created` + `terminal.attach.ready` envelopes with every + * nondeterministic leaf masked to a stable per-field placeholder — the + * field-scoped SHAPE form (see `maskEnvelopeShape`): stable across boots + * AND across the contract-valid banner/attach startup race that decides + * whether the shell banner lands in the ring before the attach snapshots + * it (which raw seq values the ready carries, and which coincide). The + * presence/nesting of fields and every deterministic contract value still + * survive, so a genuine structural divergence remains visible. */ normalizedEnvelope: { created: string; attachReady: string } /** `terminal.output.gap` occurrences (should be empty; non-empty = lost bytes). */ @@ -321,12 +326,6 @@ export function hexDiff(a: Buffer, b: Buffer, context = 16): string { ].join('\n') } -/** Normalise a single captured message envelope to its stable serialized form. */ -function normalizeEnvelope(parsed: unknown): string { - const { normalized } = normalizeTranscript([{ dir: 'in', parsed }]) - return normalized[0]?.serialized ?? '{}' -} - /** * Capture the golden byte stream for one scenario against a running server. * @@ -435,8 +434,8 @@ export async function capturePtyScenario( terminalId, streamId, normalizedEnvelope: { - created: normalizeEnvelope(created.parsed), - attachReady: normalizeEnvelope(attachReady.parsed), + created: maskEnvelopeShape(created.parsed), + attachReady: maskEnvelopeShape(attachReady.parsed), }, gaps, outputBatches: batches, diff --git a/shared/ws-protocol.ts b/shared/ws-protocol.ts index 5152aeda5..4d779143d 100644 --- a/shared/ws-protocol.ts +++ b/shared/ws-protocol.ts @@ -435,6 +435,18 @@ export const HelloSchema = z.object({ // STRIP unknown keys, so without this the capability would silently no-op. paneReconcileV1: z.literal(true).optional(), paneReconcileFreshAgentV1: z.literal(true).optional(), + // Paced terminal restore (responsive-terminal-restore Workstream 1): the + // client understands bounded, ascending paced replay batches with + // continuation credit. Additive optional — declared, not just sent (same + // strip hazard as above); absent for the frozen client shape. + pacedTerminalReplayV1: z.literal(true).optional(), + // Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1): + // the client understands non-hydrating per-connection terminal lifetime + // claims carried on `terminal.interest.claimedTerminalIds`, sent only + // after the `ready` echo advertises the capability. Additive optional — + // declared, not just sent (same strip hazard as above); absent for the + // frozen client shape. + terminalLifetimeClaimV1: z.literal(true).optional(), }).optional(), client: z.object({ mobile: z.boolean().optional(), @@ -532,6 +544,29 @@ export const TerminalAttachSchema = z.object({ expectedSessionRef: SessionLocatorSchema.optional(), sinceSeq: z.number().int().nonnegative().optional(), maxReplayBytes: z.number().int().positive().optional(), + /** Paced terminal restore (responsive-terminal-restore Workstream 1): + * the negotiated forward-page limit the client requests — an optional + * UPPER BOUND on each paced replay page's serialized bytes, honored + * only on pacedTerminalReplayV1 connections and clamped by the server + * to its own page-budget cap (min(requested, server cap)). E2R1 + * finding 2 (the honest bound): pages are bounded by + * max(requested, the atomic frame size) — a single frame larger than + * the request forms its own ATOMIC single-frame page, bounded by the + * server's fragment cap (every frame is pre-fragmented, so one + * frame's serialized size never exceeds it). Additive optional; + * absent or invalid values keep the server's default. */ + replayPageBytes: z + .number() + .int() + .positive() + .describe( + 'Optional upper bound on each paced replay page\'s serialized bytes ' + + '(pacedTerminalReplayV1 connections only), clamped to the server\'s ' + + 'page-budget cap. Pages are bounded by max(requested, the atomic ' + + 'frame size): a single frame larger than the request forms its own ' + + 'atomic page, bounded by the server fragment cap.', + ) + .optional(), attachRequestId: z.string().min(1).optional(), /** Positive marker: the attaching xterm surface was freshly constructed * (page load / renderer recreation / user reset). Servers that know this @@ -1104,6 +1139,13 @@ export const ReadyCapabilitiesSchema = z terminalInterestV1: z.literal(true).optional(), paneReconcileV1: z.literal(true).optional(), paneReconcileFreshAgentV1: z.literal(true).optional(), + // Paced terminal restore (Workstream 1): echoed only for a hello that + // opted in via capabilities.pacedTerminalReplayV1. + pacedTerminalReplayV1: z.literal(true).optional(), + // Hidden-pane lifetime claims (Workstream 1): echoed only for a hello + // that opted in via capabilities.terminalLifetimeClaimV1. Present iff the + // client may send `terminal.interest.claimedTerminalIds`. + terminalLifetimeClaimV1: z.literal(true).optional(), }) .optional() @@ -1115,9 +1157,33 @@ export const TerminalInterestSchema = z.object({ revision: z.number().int().min(1).max(Number.MAX_SAFE_INTEGER), focusedTerminalId: z.string().min(1).max(512).nullable().optional(), visibleTerminalIds: z.array(z.string().min(1).max(512)).max(1024), + /** Hidden-pane lifetime claims (negotiated `terminalLifetimeClaimV1` only): + * terminals this connection wants kept alive WITHOUT attaching. The claim + * never grants replay or output delivery and never touches geometry or + * stream identity; a later snapshot omitting an id is the explicit + * withdrawal (release). The client sends the field only after the ready + * echo; older clients never send it. */ + claimedTerminalIds: z.array(z.string().min(1).max(512)).max(1024).optional(), }) export type TerminalInterestMessage = z.infer +/** + * Paced replay continuation credit (responsive-terminal-restore Workstream 1): + * sent by a client whose hello negotiated `pacedTerminalReplayV1` after it + * fully consumed an ordered replay page. `consumedSeq` is the last sequence + * consumed in order; `attachRequestId` scopes the credit to one attach + * generation. Additive optional — older servers accept-and-strip it and + * protocol version stays 10. + */ +export const TerminalReplayCreditSchema = z.object({ + type: z.literal('terminal.replay.credit'), + terminalId: z.string().min(1).max(512), + streamId: z.string().min(1).max(512), + attachRequestId: z.string().min(1).max(512), + consumedSeq: z.number().int().min(0).max(Number.MAX_SAFE_INTEGER), +}) +export type TerminalReplayCreditMessage = z.infer + // ── Client message discriminated union ── export const ClientMessageSchema = z.discriminatedUnion('type', [ @@ -1138,6 +1204,7 @@ export const ClientMessageSchema = z.discriminatedUnion('type', [ TerminalInputSchema, TerminalResizeSchema, TerminalKillSchema, + TerminalReplayCreditSchema, CodexActivityListSchema, OpencodeActivityListSchema, ClaudeActivityListSchema, @@ -1278,7 +1345,9 @@ export type TerminalAttachReadyMessage = { geometryAuthority?: TerminalGeometryAuthority requestedSinceSeq?: number effectiveSinceSeq?: number - replayResetReason?: 'geometry_authority_unknown' + /** Restore contract (negotiated pacedTerminalReplayV1 only): earliest sequence position still available for replay (headSeq+1 when nothing older is retained). */ + oldestRetainedSeq?: number + replayResetReason?: 'geometry_authority_unknown' | 'retention_lost' headSeq: number replayFromSeq: number replayToSeq: number @@ -1467,8 +1536,16 @@ export type TerminalOutputGapMessage = { streamId: string fromSeq: number toSeq: number - reason: 'queue_overflow' | 'replay_window_exceeded' | 'replay_budget_exceeded' + reason: + | 'queue_overflow' + | 'replay_window_exceeded' + | 'replay_budget_exceeded' + | 'handoff_boundary_reached' attachRequestId?: string + /** Restore contract (negotiated pacedTerminalReplayV1 only): the terminal's current headSeq at gap-emission time. */ + headSeq?: number + /** Restore contract (negotiated pacedTerminalReplayV1 only): earliest sequence position still available for replay at gap-emission time. */ + oldestRetainedSeq?: number } export type TerminalTitleUpdatedMessage = { diff --git a/src/components/TerminalInterestReporter.tsx b/src/components/TerminalInterestReporter.tsx index 577318726..d7b5f78e9 100644 --- a/src/components/TerminalInterestReporter.tsx +++ b/src/components/TerminalInterestReporter.tsx @@ -10,7 +10,23 @@ export function TerminalInterestReporter({ workspaceVisible = true }: { workspac useEffect(() => { const ws = getWsClient() const publisher = createInterestPublisher({ - read: () => selectTerminalInterest(store.getState(), document.hidden || !workspaceVisible), + read: () => { + const snapshot = selectTerminalInterest(store.getState(), document.hidden || !workspaceVisible) + if (snapshot === null) return null + // Hidden-pane lifetime claims (responsive-terminal-restore WS1) are a + // negotiated field: strip them unless the CURRENT connection's ready + // echoed `terminalLifetimeClaimV1`. The stripped snapshot is + // byte-identical to today's — an old server never sees the field, + // and the dedupe key tracks exactly what is sendable. + const claimEcho = typeof ws.getServerCapabilities === 'function' + ? ws.getServerCapabilities().terminalLifetimeClaimV1 === true + : false + if (!claimEcho) { + const { claimedTerminalIds: _claims, ...wire } = snapshot + return wire + } + return snapshot + }, send: (snapshot) => ws.sendTerminalInterest(snapshot), scheduleTask: (task) => { const timer = window.setTimeout(task, 0) @@ -21,8 +37,11 @@ export function TerminalInterestReporter({ workspaceVisible = true }: { workspac const onState = () => { const state = store.getState() const tab = state.tabs.activeTabId - const dependencies = [tab, tab ? state.panes.layouts[tab] : undefined, - tab ? state.panes.activePane[tab] : undefined, + // Claims aggregate EVERY tab's layout (hidden panes claim their + // terminals), so the dependency set watches all layouts, not just the + // active tab's — a hidden tab's terminalId assignment or pane close + // must republish the claim set. + const dependencies = [tab, state.panes.layouts, tab ? state.panes.activePane[tab] : undefined, tab ? state.panes.zoomedPane?.[tab] : undefined] if (previous && dependencies.every((value, index) => value === previous![index])) return previous = dependencies diff --git a/src/components/TerminalView.tsx b/src/components/TerminalView.tsx index 65661e366..45a66e11b 100644 --- a/src/components/TerminalView.tsx +++ b/src/components/TerminalView.tsx @@ -107,6 +107,13 @@ import { canUseCheckpointForDeltaReplay, type TerminalGeometryAuthority, } from '@/lib/terminal-surface-checkpoint' +import { + beginRecoveryAttempt, + createTerminalRecoveryAccounting, + recordRecoveryProgress, + resetRecoveryAccounting, + type TerminalRecoveryAccounting, +} from '@/lib/terminal-recovery-accounting' import { resolveRevealAttachPlan, type DeferredAttachReason, @@ -126,6 +133,14 @@ import { type AttachSeqState, type OutputBatchAcceptedSegment, } from '@/lib/terminal-attach-seq-state' +import { + beginPacedReplayConsumption, + pacedReplayConsumeThrough, + pacedReplayMarkReceived, + pacedReplayNextCredit, + pacedReplayOnReady, + type PacedReplayConsumptionState, +} from '@/lib/paced-replay-consumption' import { useMobile } from '@/hooks/useMobile' import { usePaneFocusAdoption } from '@/hooks/usePaneFocusAdoption' import { useKeyboardInset } from '@/hooks/useKeyboardInset' @@ -224,7 +239,11 @@ const TOUCH_SCROLL_PIXELS_PER_LINE = 18 const LIGHT_THEME_MIN_CONTRAST_RATIO = 4.5 const DEFAULT_MIN_CONTRAST_RATIO = 1 const MAX_LAST_SENT_VIEWPORT_CACHE_ENTRIES = 200 -const TRUNCATED_REPLAY_BYTES = 128 * 1024 +// One replay page (responsive-terminal-restore Workstream 1): the legacy +// maxReplayBytes truncation budget and the paced replayPageBytes request are +// deliberately the same 128 KiB — the paced path pages what the legacy path +// truncated. +const REPLAY_PAGE_BYTES = 128 * 1024 const INPUT_BLOCKED_NOTICE_THROTTLE_MS = 2000 const TERMINAL_OUTPUT_BATCH_BARRIER_REASONS = new Set([ 'control', @@ -281,10 +300,18 @@ const xtermLogger: ILogger = { }, } -function viewportHydrateReplayOptions(content?: TerminalPaneContent | null): { maxReplayBytes: number } | undefined { +function viewportHydrateReplayOptions( + content?: TerminalPaneContent | null, + pacedReplay?: boolean, +): { maxReplayBytes: number } | { replayPageBytes: number } | undefined { + if (pacedReplay) { + // Negotiated: page-sized replay delivery on every hydrate, no byte-budget + // truncation (the paced path pages the whole retained window). + return { replayPageBytes: REPLAY_PAGE_BYTES } + } return content?.mode === 'opencode' ? undefined - : { maxReplayBytes: TRUNCATED_REPLAY_BYTES } + : { maxReplayBytes: REPLAY_PAGE_BYTES } } function buildSessionAssociationContentUpdates( @@ -371,6 +398,14 @@ type StartupProbeReplayDiscardState = { type TerminalOutputSubmission = { submittedWrite: boolean submittedBytesEqualInput: boolean + /** + * M2 (task-4 review): the null-screen-effect pre-parsers fully consumed the + * frame — the exact `cleaned === ''` condition computed below. Distinguishes + * full pre-parser consumption from an enqueue failure (no write queue, + * disposed surface, thrown write): both yield `submittedWrite === false`, + * but only the former is genuine consumption. + */ + preParserConsumedAllBytes: boolean } function resolveMinimumContrastRatio(theme?: { isDark?: boolean } | null): number { @@ -483,19 +518,30 @@ type LaunchAttemptState = { attachReady: boolean } -type PendingDurableReplacement = { - terminalId: string - requestId: string - reason: 'opencode_replay_window_exceeded' -} - type AttachTerminalOptions = { clearViewportFirst?: boolean suppressNextMatchingResize?: boolean skipPreAttachFit?: boolean maxReplayBytes?: number + /** Negotiated paced attaches only (viewportHydrateReplayOptions): the + * page-size request carried instead of maxReplayBytes. */ + replayPageBytes?: number priority?: TerminalAttachPriority sinceSeq?: number + /** Delivery-loss repair fallback (responsive-terminal-restore WS3): + * attach as a full hydrate WITHOUT clearing the viewport first — the + * surface is replaced only when the hydrate's content actually arrives + * (the deferred content reset). If the repair dies before content, the + * pre-gap surface stays visible. Never on the wire. */ + deferViewportClearUntilContent?: boolean + /** + * Internal recovery-accounting hint (never on the wire): the reconcile + * episode this attach belongs to (M-1). Every attach of ONE episode + * collapses into a single counted recovery attempt — the episode's + * deliberate re-attaches (pending-mark re-fire, verdict-fold re-fire) + * are pane lifecycle, not recovery cycling. + */ + recoveryAttemptKey?: string } type SentViewport = { @@ -627,6 +673,22 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te ) const reconcilePendingSinceRef = useRef(reconcilePendingSince) reconcilePendingSinceRef.current = reconcilePendingSince + // Reconcile-episode anchor for the recovery accounting (M-1): the last + // reconcile-pending/reconcile-epoch pair this attach effect consumed. A + // change in either re-fires the effect for a RECONCILE-DRIVEN attach + // (the pending mark or the verdict fold) — deliberate pane lifecycle, + // not recovery cycling. The derived per-episode key collapses every + // attach of ONE reconcile episode into a single counted attempt. + // task-009b: the anchor also records the identity it ran with + // (createRequestId + terminalId) so a later re-run can tell a + // reconcile-bookkeeping-only re-drive (nothing about the pane changed) + // from a fold that must re-attach. + const reconcileDriveStateRef = useRef<{ + pendingSince: number | undefined + epoch: number + createRequestId: string | undefined + terminalId: string | undefined + } | null>(null) // Branch-5 / reconcile-verdict interaction (design invariant 7): a pane // listed in the dead-session adjudication panel is owned by the user's // explicit panel decision -- the INVALID_TERMINAL_ID auto-recovery must @@ -666,6 +728,37 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te && window.__FRESHELL_TEST_HARNESS__?.isTerminalNetworkEffectsSuppressed?.(paneId) === true const [isAttaching, setIsAttaching] = useState(false) const [truncatedHistoryGap, setTruncatedHistoryGap] = useState<{ fromSeq: number; toSeq: number } | null>(null) + // Honest incomplete-history state (responsive-terminal-restore): a + // NEGOTIATED retention gap's visible, accessible notice. `headSeq` / + // `oldestRetainedSeq` are null when the gap omitted them — absence means + // UNKNOWN BOUNDS, never non-negotiation. + const [retentionLossNotice, setRetentionLossNotice] = useState<{ + fromSeq: number + toSeq: number + headSeq: number | null + oldestRetainedSeq: number | null + } | null>(null) + // Honest delivery-gap state (responsive-terminal-restore, round-5 + // finding 2): a NEGOTIATED queue_overflow / handoff_boundary_reached + // gap's visible, accessible notice — in UI CHROME, never in the xterm + // surface. During a checkpoint-based delta repair NOTHING may mutate + // the surface before the replayed bytes apply (the pre-fix local + // notice shifted the cursor/parser state the checkpoint captured and + // corrupted the repaired screen), so the notice rides React state — + // which the repair attach's generation change (dropQueuedStaleWrites, + // a WRITE-queue concern) cannot discard. Set when the delivery-loss + // repair initiates; cleared by the next attach (like the retention + // notice) and by a retention-gap resolution (the authoritative + // honest-loss state then takes over). + const [deliveryGapNotice, setDeliveryGapNotice] = useState<{ + fromSeq: number + toSeq: number + reason: string + } | null>(null) + // Visible, accessible retry state (WS2): automatic re-attach cycling was + // stopped by the recovery bound; the surface content below is PRESERVED and + // an explicit retry re-arms it. + const [recoveryExhausted, setRecoveryExhausted] = useState(false) const [backgroundHydrationTriggered, setBackgroundHydrationTriggered] = useState(false) const wasCreatedFreshRef = useRef(paneContent.kind === 'terminal' && paneContent.status === 'creating') const [pendingLinkUri, setPendingLinkUri] = useState(null) @@ -942,6 +1035,34 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const terminalIdRef = useRef(terminalContent?.terminalId) const seqStateRef = useRef(createAttachSeqState()) const parserAppliedSeqRef = useRef(0) + // Surface-coverage cursor (responsive-terminal-restore WS1/WS2): the + // reconstruction-safe contiguous coverage of the stream on THIS surface — + // advanced by applied frames AND fully-consumed null-screen-effect + // filtered frames (the same classes the paced consumption frontier uses); + // pinned below unknown mutations, lost ranges, and locally-unapplied + // ranges (contiguity-gated — never jumps a hole). Monotonic within a + // surface generation; resets with the surface (same epoch discipline). + // DISTINCT from parserAppliedSeqRef, which never advances across a + // filtered or lost range and keeps its strict checkpoint semantics. + const surfaceCoverageSeqRef = useRef(0) + // Bounded automatic recovery accounting (WS2): consecutive progressless + // attach attempts vs the coverage cursor; gates automatic re-attach + // cycling and drives the visible retry state. Never kills or replaces. + // Round-4 reversal (plan:166): the streak resets ONLY on genuine parser + // progress (an ADVANCE of the applied surface) or an explicit user + // retry — never on attach.ready/reconnect completions. + const recoveryAccountingRef = useRef(createTerminalRecoveryAccounting()) + // Delivery-loss repair fallback (responsive-terminal-restore WS3): the + // attach generation of a full-hydrate repair that must NOT clear the + // viewport at attach time — the surface is replaced only when that + // generation's first content frame arrives (the deferred content reset: + // "the viewport must not be cleared before the new baseline is actually + // established by attach content"). Null when no deferred reset is + // pending; cleared once fired and re-armed per attach. A + // `replay_window_exceeded` gap in THIS generation DISARMS it (the + // server just declared the prefix unreconstructible — the pre-gap + // screen is the best available surface and must be preserved). + const repairContentResetPendingRef = useRef(null) const surfaceEpochRef = useRef(0) const geometryEpochRef = useRef(1) const geometryAuthorityRef = useRef('single_client') @@ -953,6 +1074,15 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te startedAt: number timedOut: boolean timer: ReturnType | null + /** + * Write items COMPLETED since the quarantined attach armed this repair + * (any generation — the frozen window's old in-flight writes included: + * their bytes were already submitted when they went in flight). Zero at + * drain time proves the surface still matches its last checkpoint; any + * completion means the checkpoint under-describes the surface and the + * repair must rebuild from a full hydrate. + */ + completedWrites: number } | null>(null) const abandonedAttachRequestIdsRef = useRef(new Set()) const attachCounterRef = useRef(0) @@ -1005,7 +1135,6 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te requestId: string terminalId: string } | null>(null) - const pendingDurableReplacementRef = useRef(null) // Wedge-backstop (Task 4 review, Minor-1): in-flight lock for the stuck // card's shared kill-await path. The card's buttons stay enabled during the // bounded KILL_ACK_TIMEOUT_MS wait; without the lock a second click @@ -1066,6 +1195,13 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te streamId: getTerminalCheckpointStreamId(), serverInstanceId, surfaceEpoch: surfaceEpochRef.current, + // Surface-instance discriminator (WS2 reload contract): the id of the + // exact xterm surface instance this component currently renders on — + // mount-stable AND renderer-recreation-stable. Checkpoint loads + // validate it and saves carry it, so a remounted pane (same store + // key, colliding epoch) can never resume the previous mount's cursor + // past the new surface's rendered position. + surfaceInstanceId: terminalInstanceIdRef.current, cols: normalizeDimension(dimensions?.cols, term?.cols ?? 80), rows: normalizeDimension(dimensions?.rows, term?.rows ?? 24), geometryEpoch: geometryEpochRef.current, @@ -1084,15 +1220,26 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te if (!checkpointInput) { return { ok: false as const, reason: 'missing_checkpoint' as const } } + // Surface-scoped (WS2): sibling panes rendering the same terminal keep + // isolated checkpoint stores — this pane can only resume ITS OWN surface's + // rendered progress. const checkpoint = loadTerminalSurfaceCheckpoint(terminalId, { streamId: checkpointInput.streamId, serverInstanceId: checkpointInput.serverInstanceId, - }) + surfaceInstanceId: checkpointInput.surfaceInstanceId, + }, { paneId: paneIdRef.current }) return canUseCheckpointForDeltaReplay(checkpoint, checkpointInput) }, [buildCheckpointReplayInput]) - const resetParserAppliedSurface = useCallback((seq = 0, opts?: { incrementEpoch?: boolean }) => { + const resetParserAppliedSurface = useCallback((seq = 0, opts?: { incrementEpoch?: boolean; surfaceCoverageSeq?: number }) => { parserAppliedSeqRef.current = Math.max(0, Math.floor(Number.isFinite(seq) ? seq : 0)) + // Coverage follows the surface's epoch discipline: a rebuild resets it to + // the new baseline position; a local-notice invalidation (which keeps the + // applied position and only bumps the epoch) may preserve it — the notice + // appends non-stream bytes but never un-accounts stream coverage. + surfaceCoverageSeqRef.current = Math.max(0, Math.floor( + Number.isFinite(opts?.surfaceCoverageSeq) ? opts!.surfaceCoverageSeq! : seq, + )) if (opts?.incrementEpoch !== false) { surfaceEpochRef.current += 1 } @@ -1125,6 +1272,136 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te quarantineRepairRef.current = null }, []) + // Deferred content reset (responsive-terminal-restore WS3, delivery-loss + // repair fallback): when a hydrate attach deferred its viewport clear + // (`deferViewportClearUntilContent`), the pre-gap surface stays visible + // until that generation's first RENDERING frame — replacement content + // the sequence validator ACCEPTED and whose write path will write bytes + // to xterm — actually establishes the new baseline (round-3 fix: the + // clear consumes at the render moment, never on mere envelope arrival — + // a rejected duplicate/overlap or a fully filtered frame such as the + // OSC52-only case renders nothing and must leave the clear armed). + // The consumption runs inside `handleTerminalOutput` immediately before + // the triggering frame's write is enqueued, so the surface is always + // cleared-then-written; if the enqueue fails (no surface), the caller + // re-arms the pending clear because nothing rendered. Two paths retire + // the pending clear WITHOUT wiping: the repair dying without content + // (nothing is cleared and the pre-gap surface survives), and — the + // round-2 disarm — a `replay_window_exceeded` gap in this generation + // (the server declared the prefix unreconstructible, so the pre-gap + // screen is the best available surface and the existing honest-loss UX + // proceeds; see the gap arm). + // Round-4 F2: the consumption returns the clear as a THUNK instead of + // running it synchronously — the caller hands it to the write queue as + // `clearBeforeWrite`, so the clear applies INSIDE the same + // generation-guarded queue item as the replacement bytes (atomic + // clear-then-write at flush time: a dropped generation drops clear+write + // together, a stale generation is refused at apply time, and an + // in-flight earlier write completes before the clear runs). The two + // no-wipe retirement paths (death without content; the round-2 + // declared-unreconstructible disarm) are unchanged. + const consumeRepairContentReset = useCallback((terminalId: string, attachRequestId?: unknown): (() => void) | null => { + if (repairContentResetPendingRef.current !== attachRequestId) return null + repairContentResetPendingRef.current = null + return () => { + try { + termRef.current?.clear() + } catch { + // disposed + } + clearTerminalCursor(terminalId) + } + }, []) + + // ── Paced terminal replay consumption (responsive-terminal-restore + // Workstream 1, client side) ── + // All paced behavior is gated on the CURRENT connection's capability echo; + // absent (or a ws client without the accessor) → today's exact wire + // behavior everywhere. + const isPacedReplayNegotiated = useCallback((): boolean => { + const capabilities = typeof ws.getServerCapabilities === 'function' + ? ws.getServerCapabilities() + : undefined + return capabilities?.pacedTerminalReplayV1 === true + }, [ws]) + + // Hidden-pane lifetime claims (responsive-terminal-restore Workstream 1): + // the CURRENT connection's ready echoed `terminalLifetimeClaimV1` — hidden + // panes claim their terminals in the interest snapshot instead of attaching. + // Both flags must be present: claims ride `terminal.interest` snapshots, so + // an interest-capable echo without the claim echo (or vice versa) cannot + // deliver claims and must fall back to today's hidden attach. Absent (or a + // ws client without the accessor) → today's exact wire behavior everywhere. + const isLifetimeClaimNegotiated = useCallback((): boolean => { + const capabilities = typeof ws.getServerCapabilities === 'function' + ? ws.getServerCapabilities() + : undefined + return capabilities?.terminalLifetimeClaimV1 === true + && capabilities?.terminalInterestV1 === true + }, [ws]) + + // The consumption frontier for the CURRENT attach generation only — replaced + // by the next attach, absent for non-negotiated attaches. + const pacedReplayRef = useRef(null) + + const advancePacedReplayConsumption = useCallback((attachRequestId: string | undefined, seqEnd: number) => { + const paced = pacedReplayRef.current + if (!paced || !attachRequestId || paced.attachRequestId !== attachRequestId) return + const next = pacedReplayConsumeThrough(paced, seqEnd) + if (next !== paced) { + pacedReplayRef.current = next + } + }, []) + + const markPacedReplayReceived = useCallback((attachRequestId: string | undefined, seqEnd: number) => { + const paced = pacedReplayRef.current + if (!paced || !attachRequestId || paced.attachRequestId !== attachRequestId) return + const next = pacedReplayMarkReceived(paced, seqEnd) + if (next !== paced) { + pacedReplayRef.current = next + } + }, []) + + const flushPacedReplayCredit = useCallback(() => { + const paced = pacedReplayRef.current + if (!paced) return + const activeAttach = currentAttachRef.current + if (!activeAttach || activeAttach.requestId !== paced.attachRequestId) return + const streamId = activeAttach.streamId + if (typeof streamId !== 'string' || streamId.length === 0) return + const decision = pacedReplayNextCredit(paced) + if (decision.state !== paced) { + pacedReplayRef.current = decision.state + } + if (!decision.credit) return + ws.send({ + type: 'terminal.replay.credit', + terminalId: decision.credit.terminalId, + streamId, + attachRequestId: decision.credit.attachRequestId, + consumedSeq: decision.credit.consumedSeq, + }) + }, [ws]) + + // A consumption advance with no write of its own (a fully pre-filtered + // frame) rides the write queue as a task so the SAME drain-tick coalescing + // applies: the credit flushes once per drain, never per frame. Armed for + // the current paced attach generation only — non-negotiated panes keep + // today's queue traffic byte-identical. + const schedulePacedReplayCreditFlush = useCallback(( + outputSource: TerminalOutputSource, + attachRequestId: string | undefined, + ) => { + const paced = pacedReplayRef.current + if (!paced || !attachRequestId || paced.attachRequestId !== attachRequestId) return + const queue = writeQueueRef.current + if (!queue) { + flushPacedReplayCredit() + return + } + queue.enqueueTask(() => {}, { mode: outputSource, generation: attachRequestId }) + }, [flushPacedReplayCredit]) + const recordTerminalPerfAuditEvent = useCallback((event: string, data: Record = {}) => { const payload = Object.fromEntries(Object.entries({ event, @@ -1158,10 +1435,29 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te return } if (!pending.queue.hasInFlightWrites()) { + // The frozen window closed with the queue drained: NOW repair + // (WS2 — never decide while an earlier write can still mutate the + // surface). The quarantined attach deferred the applied-surface + // reset, so the repair ALWAYS rebuilds from a full hydrate: a + // quarantine arms only while a write is in flight, that write's + // completion always lands in the completedWrites ledger (the write + // wrapper reports every completion, including a disposed-surface + // throw), and the drain only happens after in-flight writes + // complete — so the window provably mutated the surface and the + // pre-quarantine checkpoint under-describes it. The former + // completedWrites === 0 resume branch was unreachable in + // production shapes and was removed (M-2); the ledger stays in the + // audit event as the honest evidence. clearQuarantineRepair(attachRequestId) + recordTerminalPerfAuditEvent('terminal.catchup.surface_quarantine_repair', { + terminalId, + attachRequestId, + resumable: false, + completedWritesDuringQuarantine: pending.completedWrites, + }) attachTerminalRef.current?.(terminalId, 'viewport_hydrate', { clearViewportFirst: true, - ...viewportHydrateReplayOptions(contentRef.current), + ...viewportHydrateReplayOptions(contentRef.current, isPacedReplayNegotiated()), }) return } @@ -1191,9 +1487,88 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te queue, startedAt, timedOut: false, + completedWrites: 0, timer: setTimeout(poll, QUARANTINE_REPAIR_POLL_MS), } - }, [clearQuarantineRepair, recordTerminalPerfAuditEvent]) + }, [clearQuarantineRepair, isPacedReplayNegotiated, recordTerminalPerfAuditEvent]) + + // Persist the current surface checkpoint (applied + coverage) for the given + // attach context. Shared by the applied-frame save path and the + // coverage-advance save paths (a filtered advance must persist its cursor + // too, or a disconnect right after a filtered page resumes from below it). + const persistSurfaceCheckpointForAttach = useCallback(( + terminalId: string | undefined, + attach: { + requestId: string + terminalId: string + cols: number + rows: number + streamId?: string | null + surfaceQuarantined?: boolean + } | null, + parserAppliedSeq: number, + ) => { + if (!terminalId || !Number.isFinite(parserAppliedSeq)) return + if (attach?.surfaceQuarantined === true) return + if (!attach || attach.terminalId !== terminalId) return + if (typeof attach.streamId !== 'string' || attach.streamId.length === 0) return + const checkpointInput = buildCheckpointReplayInput(terminalId, { + cols: attach.cols, + rows: attach.rows, + }) + if (!checkpointInput) return + + saveTerminalSurfaceCheckpoint({ + terminalId: checkpointInput.terminalId, + streamId: checkpointInput.streamId, + serverInstanceId: checkpointInput.serverInstanceId, + surfaceEpoch: checkpointInput.surfaceEpoch, + surfaceInstanceId: checkpointInput.surfaceInstanceId, + attachRequestId: attach.requestId, + parserAppliedSeq, + surfaceCoverageSeq: surfaceCoverageSeqRef.current, + cols: checkpointInput.cols, + rows: checkpointInput.rows, + geometryEpoch: checkpointInput.geometryEpoch, + geometryAuthority: checkpointInput.geometryAuthority, + scrollback: checkpointInput.scrollback, + xtermVersion: checkpointInput.xtermVersion, + // Task 3 cannot yet prove normal vs alternate buffer. Keep checkpoints conservative + // until the geometry/buffer authority work supplies this context. + bufferType: 'unknown', + parserIdle: true, + }, { paneId: paneIdRef.current }) + }, [buildCheckpointReplayInput]) + + // Advance the surface-coverage cursor across [seqStart, seqEnd]: + // contiguity-gated — only a range that starts exactly at the cursor's next + // position extends it, so unknown mutations, lost ranges, and unapplied + // ranges can never be jumped past (the cursor pins below them until a + // surface reset). Genuine progress here resets the recovery accounting. + const advanceSurfaceCoverageForRange = useCallback((seqStart: number, seqEnd: number): boolean => { + const coverage = surfaceCoverageSeqRef.current + if (seqStart !== coverage + 1 || seqEnd <= coverage) return false + surfaceCoverageSeqRef.current = seqEnd + const wasExhausted = recoveryAccountingRef.current.exhausted + recoveryAccountingRef.current = recordRecoveryProgress(recoveryAccountingRef.current, seqEnd, Date.now()) + if (wasExhausted && !recoveryAccountingRef.current.exhausted) { + setRecoveryExhausted(false) + } + return true + }, []) + + // Coverage advance + checkpoint persist in one step (both advance sites): + // the persisted checkpoint must carry the new coverage position, else an + // interrupt right after a filtered page would resume from below it. + const advanceSurfaceCoverageAndPersist = useCallback(( + terminalId: string | undefined, + attach: Parameters[1], + seqStart: number, + seqEnd: number, + ) => { + if (!advanceSurfaceCoverageForRange(seqStart, seqEnd)) return + persistSurfaceCheckpointForAttach(terminalId, attach, parserAppliedSeqRef.current) + }, [advanceSurfaceCoverageForRange, persistSurfaceCheckpointForAttach]) const markParserAppliedFrame = useCallback((terminalId: string | undefined, seq: number, attachContext?: { requestId: string @@ -1219,38 +1594,15 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te streamId: attach?.streamId ?? getTerminalCheckpointStreamId(), parserAppliedSeq, previousParserAppliedSeq, + surfaceCoverageSeq: surfaceCoverageSeqRef.current, surfaceEpoch: surfaceEpochRef.current, surfaceQuarantined, }) if (surfaceQuarantined) return if (!attach || attach.terminalId !== terminalId) return - if (typeof attach.streamId !== 'string' || attach.streamId.length === 0) return - const checkpointInput = buildCheckpointReplayInput(terminalId, { - cols: attach.cols, - rows: attach.rows, - }) - if (!checkpointInput) return - - saveTerminalSurfaceCheckpoint({ - terminalId: checkpointInput.terminalId, - streamId: checkpointInput.streamId, - serverInstanceId: checkpointInput.serverInstanceId, - surfaceEpoch: checkpointInput.surfaceEpoch, - attachRequestId: attach.requestId, - parserAppliedSeq, - cols: checkpointInput.cols, - rows: checkpointInput.rows, - geometryEpoch: checkpointInput.geometryEpoch, - geometryAuthority: checkpointInput.geometryAuthority, - scrollback: checkpointInput.scrollback, - xtermVersion: checkpointInput.xtermVersion, - // Task 3 cannot yet prove normal vs alternate buffer. Keep checkpoints conservative - // until the geometry/buffer authority work supplies this context. - bufferType: 'unknown', - parserIdle: true, - }) - }, [buildCheckpointReplayInput, getTerminalCheckpointStreamId, recordTerminalPerfAuditEvent]) + persistSurfaceCheckpointForAttach(terminalId, attach, parserAppliedSeq) + }, [getTerminalCheckpointStreamId, persistSurfaceCheckpointForAttach, recordTerminalPerfAuditEvent]) const writeLocalXtermNotice = useCallback((term: Terminal, data: string) => { const terminalInstanceId = terminalInstanceIdRef.current @@ -1263,7 +1615,13 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te return } const invalidateAppliedSurface = () => { - resetParserAppliedSurface(parserAppliedSeqRef.current) + // Keep the applied position (the notice does not unapply stream bytes), + // bump the epoch (stream-unpure surface state), and preserve the + // coverage cursor — the notice appends non-stream bytes but never + // un-accounts stream coverage. + resetParserAppliedSurface(parserAppliedSeqRef.current, { + surfaceCoverageSeq: surfaceCoverageSeqRef.current, + }) } const generation = currentAttachRef.current?.requestId const queue = writeQueueRef.current @@ -1916,11 +2274,16 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te // above satisfies exhaustive-deps without new warnings. }, [mayFocusNow, suppressNetworkEffects, syncGeometryEpochForViewport, ws, tabId]) - const enqueueTerminalWrite = useCallback((data: string, onWritten?: () => void, options?: TerminalWriteQueueOptions): boolean => { + const enqueueTerminalWrite = useCallback(( + data: string, + onWritten?: () => void, + options?: TerminalWriteQueueOptions, + clearBeforeWrite?: () => void, + ): boolean => { if (!data) return false const queue = writeQueueRef.current if (queue) { - queue.enqueue(data, onWritten, options) + queue.enqueue(data, onWritten, clearBeforeWrite ? { ...options, clearBeforeWrite } : options) return true } const term = termRef.current @@ -1935,6 +2298,10 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te suppressExternalSideEffects: mode === 'replay', }) try { + // No write queue (the direct fallback): the clear runs + // synchronously immediately before the direct write — the same + // clear-then-write ordering, in one synchronous step. + clearBeforeWrite?.() term.write(data, () => { try { onWritten?.() @@ -2067,19 +2434,52 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te } const submittedBytesEqualInput = cleaned === raw + // THE RENDER MOMENT (responsive-terminal-restore WS3, round-3 fix; + // round-4 atomicity): a deferred repair clear consumes ONLY here — + // the sequence validator already accepted this frame (the caller + // runs the acceptance before submitting), and `cleaned` non-empty + // proves the frame's write path WILL write bytes to xterm. The + // consumed clear is returned as a THUNK and handed to the write + // queue as `clearBeforeWrite` on the SAME generation-guarded item as + // the replacement write: the flush applies clear-then-write as ONE + // atomic unit — a dropped generation drops clear+write together + // (the old surface survives a disconnect/supersede mid-window), a + // stale-generation write can never apply after the clear (the + // generation is checked at apply time), and an in-flight earlier + // write completes BEFORE the clear runs (the queue is serial). A + // rejected or fully filtered frame (`cleaned === ''`) never reaches + // this point and leaves the clear armed for the next writing frame. + // Rejected/filtered frames of a LATER generation than the armed one + // no-op the ref check inside, exactly like before. + let consumedDeferredClear: (() => void) | null = null + if (cleaned && tid) { + consumedDeferredClear = consumeRepairContentReset(tid, writeOptions?.generation) + } const submittedWrite = cleaned ? enqueueTerminalWrite( cleaned, submittedBytesEqualInput ? onParserApplied : undefined, writeOptions, + consumedDeferredClear ?? undefined, ) : false + if (consumedDeferredClear && !submittedWrite) { + // Nothing will render (no surface / disposed write): the clear was + // consumed for content that never renders — re-arm it. Nothing else + // can touch the ref inside this synchronous block, so the re-arm is + // exact. + repairContentResetPendingRef.current = writeOptions?.generation ?? null + } for (const event of osc.events) { handleOsc52Event(event, outputSource, mode) } - return { submittedWrite, submittedBytesEqualInput: submittedWrite && submittedBytesEqualInput } - }, [dispatch, enqueueTerminalWrite, handleOsc52Event, sendInput, tabId]) + return { + submittedWrite, + submittedBytesEqualInput: submittedWrite && submittedBytesEqualInput, + preParserConsumedAllBytes: cleaned === '', + } + }, [consumeRepairContentReset, dispatch, enqueueTerminalWrite, handleOsc52Event, sendInput, tabId]) const findNext = useCallback((value: string = searchQuery) => { const terminalId = terminalIdRef.current @@ -2298,8 +2698,16 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te surfaceFreshRef.current = true surfaceFreshMarkerRef.current = null surfaceWritesSinceFreshRef.current = 0 + // A brand-new surface has no covered stream position. + surfaceCoverageSeqRef.current = 0 const writeQueue = createTerminalWriteQueue({ terminalInstanceId, + // One paced-replay credit per drain tick (Workstream 1): the flush is + // idempotent (lastSentCreditSeq guard), so non-paced panes and drains + // with no frontier movement are no-ops. + onDrain: () => { + flushPacedReplayCredit() + }, onItemApplied: (item) => { surfaceWritesSinceFreshRef.current += 1 // Coupled clear (plan round-3): the marker-bearing attach's own @@ -2312,6 +2720,18 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te surfaceFreshMarkerRef.current = null } }, + // Surface-mutation ledger for the quarantine repair (WS2): EVERY + // completed write during a quarantined attach's frozen window — + // including stale-generation completions (their bytes were already + // submitted when they went in flight) — proves the surface moved + // beyond its last checkpoint, so the repair must rebuild instead of + // resuming a checkpoint that under-describes the surface. + onWriteCompleted: () => { + const pendingQuarantine = quarantineRepairRef.current + if (pendingQuarantine) { + pendingQuarantine.completedWrites += 1 + } + }, write: (data, onWritten) => { const recordForTest = (phase: 'submitted' | 'written') => { try { @@ -2490,6 +2910,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te surfaceFreshRef.current = true surfaceFreshMarkerRef.current = null surfaceWritesSinceFreshRef.current = 0 + surfaceCoverageSeqRef.current = 0 }, scrollToBottom: () => { if (!allowCurrentTerminalAction()) return @@ -2755,13 +3176,24 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te pendingSinceSeq: 0, pendingReason: 'initial_hydrate', } + // Round-4 reversal to plan:166's literal rule: the progressless + // streak resets ONLY on genuine parser progress (an ADVANCE of the + // applied surface — recorded at the frame-application site through + // recordRecoveryProgress) or an explicit user retry + // (resetRecoveryAccounting). Receiving attach.ready or another + // reconnect — however clean, cursor-confirmed, or gap-free — is NOT + // progress and must never reset the accounting: a converged idle + // pane's flap cycle exhausts to the visible retry strip exactly + // like any other progressless cycle (the empty-window ready path + // calling this is the plan's own example). The 81566b88f/40afe39b7 + // ready-keyed reset contradicted the plan's sentence and is removed. const queue = getHydrationQueue() if (hiddenRef.current) { queue.onHydrationComplete(paneId) } else { queue.onActiveTabReady(tabId, tabOrderRef.current) } - }, [paneId, tabId]) + }, [paneId, tabId, recordTerminalPerfAuditEvent]) const markTerminalOutputRangeLost = useCallback((input: { terminalId: string @@ -2835,6 +3267,18 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te }, options) }, [tabId]) + // Hidden-pane lifetime claims (responsive-terminal-restore WS1): a + // claim-negotiated hidden pane never holds a background-hydration + // registration — a stale OLD-SERVER registration (armed before the + // connection re-negotiated with claims) must not attach on a late grant + // under the negotiated regime. + const unregisterBackgroundHydration = useCallback(() => { + if (hydrationRegisteredRef.current) { + getHydrationQueue().unregister(paneIdRef.current) + hydrationRegisteredRef.current = false + } + }, []) + const isCurrentAttachMessage = useCallback((msg: { type: string terminalId: string @@ -2926,7 +3370,9 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const gapDecision = onOutputGap(previousSeqState, { fromSeq, toSeq }) const nextSeqState = gapDecision.state applySeqState(nextSeqState) - resetParserAppliedSurface(parserAppliedSeqRef.current) + resetParserAppliedSurface(parserAppliedSeqRef.current, { + surfaceCoverageSeq: surfaceCoverageSeqRef.current, + }) recordTerminalPerfAuditEvent('terminal.catchup.surface_quarantined', { terminalId: msg.terminalId, messageType: msg.type, @@ -2948,7 +3394,9 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te markAttachComplete() } } else { - resetParserAppliedSurface(parserAppliedSeqRef.current) + resetParserAppliedSurface(parserAppliedSeqRef.current, { + surfaceCoverageSeq: surfaceCoverageSeqRef.current, + }) recordTerminalPerfAuditEvent('terminal.catchup.surface_quarantined', { terminalId: msg.terminalId, messageType: msg.type, @@ -3014,6 +3462,35 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te }) return } + // Bounded automatic recovery (WS2): consecutive attach/hydrate attempts + // without a coverage-cursor advance count against a bounded limit; past + // it, automatic cycling STOPS and the visible retry state shows (the + // surface content is preserved). Never kills, replaces, or changes + // identity. Explicit user paths (pane refresh, the retry control) reset + // the accounting before they get here. + const recoveryDecision = beginRecoveryAttempt(recoveryAccountingRef.current, { + coverageSeq: surfaceCoverageSeqRef.current, + now: Date.now(), + attemptKey: opts?.recoveryAttemptKey, + }) + recoveryAccountingRef.current = recoveryDecision.state + if (!recoveryDecision.allowed) { + setRecoveryExhausted(true) + log.debug('Terminal restore recovery bound reached; automatic attach declined', { + terminalId: tid, + paneId: paneIdRef.current, + intent, + attempts: recoveryDecision.state.attempts, + coverageSeq: surfaceCoverageSeqRef.current, + }) + recordTerminalPerfAuditEvent('terminal.restore.recovery_exhausted', { + terminalId: tid, + intent, + attempts: recoveryDecision.state.attempts, + coverageSeq: surfaceCoverageSeqRef.current, + }) + return + } const term = termRef.current if (!term) return const runtime = runtimeRef.current @@ -3041,27 +3518,57 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te let effectiveIntent = intent let clearViewportFirst = opts?.clearViewportFirst === true let fullHydrateFallbackReason: string | null = null + // Partial-hydrate exemption (WS2): a fresh-flagged surface that already + // CONSUMED content for its marker generation is NOT blank — an interrupted + // hydrate. A valid checkpoint lets the same mounted xterm resume the + // remainder via transport_reconnect (sinceSeq from the coverage cursor): + // no wipe, no surfaceReset re-claim, no mode-preamble re-send. A hydrate + // interrupted before ANY consumption (nothing to resume from) keeps the + // full-hydrate path below. + let partialHydrateSinceSeq: number | null = null if (surfaceFreshRef.current) { - // A fresh surface has NO usable delta window: any checkpointed sinceSeq - // would continue content onto a blank xterm (data hole). Force the full - // hydrate. Wipe ONLY when a marker exists — a marker means an EARLIER - // claim was abandoned mid-hydration (trap-door), so this surface may - // hold partial replay content; a genuinely fresh surface (marker null, - // flag just set at construction/user-reset) is blank by construction - // and must not pay a spurious term.clear(). - if (effectiveIntent !== 'viewport_hydrate') { - effectiveIntent = 'viewport_hydrate' - fullHydrateFallbackReason = 'surface_fresh' - } - if (surfaceFreshMarkerRef.current !== null || surfaceWritesSinceFreshRef.current > 0) { - // Wipe before the forced full replay whenever the surface may hold - // content: an abandoned in-flight claim (marker set) or ANY applied - // write since the fresh marking (e.g. live output after a user - // reset). A genuinely blank fresh surface pays neither. - clearViewportFirst = true + const partiallyConsumed = surfaceWritesSinceFreshRef.current > 0 + || surfaceCoverageSeqRef.current > 0 + partialHydrateSinceSeq = surfaceFreshMarkerRef.current !== null + && partiallyConsumed + && checkpointDecision.ok + && effectiveIntent !== 'viewport_hydrate' + ? checkpointDecision.sinceSeq + : null + if (partialHydrateSinceSeq !== null) { + recordTerminalPerfAuditEvent('terminal.restore.partial_hydrate_resume', { + terminalId: tid, + attachRequestId, + requestedIntent: intent, + sinceSeq: partialHydrateSinceSeq, + coverageSeq: surfaceCoverageSeqRef.current, + surfaceWritesSinceFresh: surfaceWritesSinceFreshRef.current, + }) + } else { + // A fresh surface has NO usable delta window: any checkpointed sinceSeq + // would continue content onto a blank xterm (data hole). Force the full + // hydrate. Wipe ONLY when a marker exists — a marker means an EARLIER + // claim was abandoned mid-hydration (trap-door), so this surface may + // hold partial replay content; a genuinely fresh surface (marker null, + // flag just set at construction/user-reset) is blank by construction + // and must not pay a spurious term.clear(). + if (effectiveIntent !== 'viewport_hydrate') { + effectiveIntent = 'viewport_hydrate' + fullHydrateFallbackReason = 'surface_fresh' + } + if (surfaceFreshMarkerRef.current !== null || surfaceWritesSinceFreshRef.current > 0) { + // Wipe before the forced full replay whenever the surface may hold + // content: an abandoned in-flight claim (marker set) or ANY applied + // write since the fresh marking (e.g. live output after a user + // reset). A genuinely blank fresh surface pays neither. + clearViewportFirst = true + } } } if (hasInFlightWrites && effectiveIntent !== 'viewport_hydrate') { + // In-flight-writes freeze (WS2): never DECIDE (or clear) a surface while + // an earlier write can still mutate it — the quarantined attach defers + // the checkpoint decision to the bounded drain wait and its repair. effectiveIntent = 'viewport_hydrate' clearViewportFirst = true fullHydrateFallbackReason = 'in_flight_writes' @@ -3101,21 +3608,34 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te setIsAttaching(true) setTruncatedHistoryGap(null) + setRetentionLossNotice(null) + setDeliveryGapNotice(null) // Startup probes must not leak across attach generations. resetStartupProbeParser() if (effectiveIntent === 'viewport_hydrate') { - resetParserAppliedSurface() - if (clearViewportFirst && !surfaceQuarantined) { - try { - termRef.current?.clear() - } catch { - // disposed + if (!surfaceQuarantined) { + // A live hydrate rebuilds the surface from zero — new surface + // generation (epoch), zero applied, zero coverage. A QUARANTINED + // hydrate defers this reset: the surface is frozen pending the + // bounded drain wait, and the repair decides whether to resume the + // survived checkpoint or rebuild. + resetParserAppliedSurface() + if (clearViewportFirst) { + try { + termRef.current?.clear() + } catch { + // disposed + } } } applySeqState(beginAttach(createAttachSeqState({ lastSeq: 0 }))) } else { + // Delta resume: the surface keeps its rendered content and its + // coverage position (≥ the resume baseline by construction — the + // baseline came from this surface's own checkpoint); only the seq + // state re-bases to the resume position. applySeqState(beginAttach(createAttachSeqState({ lastSeq: deltaSeq, parserAppliedSeq: deltaSeq, @@ -3129,6 +3649,21 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te pendingReason: opts?.priority === 'background' ? 'background_catchup' : 'initial_hydrate', } + // Paced replay consumption (Workstream 1): arm the frontier for THIS + // generation only on a negotiated connection — the next attach replaces + // it; non-negotiated attaches never arm one. + const pacedReplayNegotiated = isPacedReplayNegotiated() + pacedReplayRef.current = pacedReplayNegotiated + ? beginPacedReplayConsumption({ terminalId: tid, attachRequestId, sinceSeq }) + : null + + // Per-generation deferred content reset (delivery-loss repair + // fallback): armed only for the hydrate that asked to defer its + // viewport clear until its content establishes the new baseline. + repairContentResetPendingRef.current = opts?.deferViewportClearUntilContent === true + ? attachRequestId + : null + currentAttachRef.current = { requestId: attachRequestId, intent: effectiveIntent, @@ -3172,12 +3707,18 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te ? { terminalId: tid, cols, rows } : null - // surfaceReset is claimed on the wire whenever the surface is fresh — - // independently of the wire intent swap (hidden panes attach as - // keepalive_delta yet still need the preamble: background hydration is - // exactly where a recreated hidden pane receives its mode bytes). - const claimSurfaceReset = surfaceFreshRef.current - if (claimSurfaceReset) { + // surfaceReset is claimed on the wire whenever the surface is fresh AND + // this attach actually rebuilds it — independently of the wire intent + // swap (hidden panes attach as keepalive_delta yet still need the + // preamble: background hydration is exactly where a recreated hidden + // pane receives its mode bytes). A partial-hydrate EXEMPTION resume + // re-claims nothing (its preamble was already delivered with the + // interrupted attach) but re-targets the coupled marker to THIS + // generation so the claim's consumption sites (its replay content + // applying, or its attach completing) keep working across the + // continuation. + const claimSurfaceReset = surfaceFreshRef.current && effectiveIntent === 'viewport_hydrate' + if (surfaceFreshRef.current) { surfaceFreshMarkerRef.current = { attachRequestId } } ws.send(buildTerminalAttachMessage({ @@ -3194,7 +3735,13 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te // fence — terminal.attach participates in the coordinator (a queued // cross-device attach is generation-fenced server-side). ownerFence: selectPaneOwnerFence(appStore.getState(), contentRef.current ?? {}) ?? undefined, - ...(opts?.maxReplayBytes ? { maxReplayBytes: opts.maxReplayBytes } : {}), + // Paced replay negotiation (Workstream 1): negotiated attaches — + // fresh AND delta, with or without explicit options — request + // page-sized delivery and never carry the legacy truncation budget. + // Non-negotiated stays byte-identical to today. + ...(pacedReplayNegotiated + ? { replayPageBytes: opts?.replayPageBytes ?? REPLAY_PAGE_BYTES } + : opts?.maxReplayBytes ? { maxReplayBytes: opts.maxReplayBytes } : {}), ...(claimSurfaceReset ? { surfaceReset: true } : {}), })) rememberSentViewport(tid, cols, rows) @@ -3218,6 +3765,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te clearQuarantineRepair, getCheckpointDeltaReplayDecision, getTerminalCheckpointStreamId, + isPacedReplayNegotiated, recordTerminalPerfAuditEvent, resetParserAppliedSurface, scheduleQuarantineRepair, @@ -3237,6 +3785,16 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te if (!paneRefreshTargetMatchesContent(request.target, currentContent)) return false handledRefreshRequestIdRef.current = request.requestId + // An explicit pane refresh is user intent, not automatic recovery cycling: + // reset the recovery accounting so the refresh attach is never blocked by + // the no-progress bound (and the retry strip, if shown, is re-armed from + // the current coverage position). + recoveryAccountingRef.current = resetRecoveryAccounting( + recoveryAccountingRef.current, + surfaceCoverageSeqRef.current, + Date.now(), + ) + setRecoveryExhausted(false) ws.send({ type: 'terminal.detach', terminalId: tid }) if (hiddenRef.current) { @@ -3248,23 +3806,50 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te pendingReason: 'explicit_refresh', } setIsAttaching(false) - // F8: the detach was already sent above -- re-arm the attach via the - // background hydration queue so a hidden refresh cannot strand the - // pane detached until reveal. Same three-step sequence as the - // terminal.created site (see its comment for why each line exists). + // Lifetime (responsive-terminal-restore WS1): a claim-negotiated hidden + // pane does not re-arm the hydration queue — the terminal is claimed in + // the interest snapshot, and the deferred intent above re-hydrates at + // reveal. + if (isLifetimeClaimNegotiated()) return + // F8 (old-server fallback): the detach was already sent above -- re-arm + // the attach via the background hydration queue so a hidden refresh + // cannot strand the pane detached until reveal. Same three-step + // sequence as the terminal.created site (see its comment for why each + // line exists). getHydrationQueue().onHydrationComplete(paneIdRef.current) hydrationRegisteredRef.current = false registerForBackgroundHydration({ queueIfStarted: true }) } else { attachTerminal(tid, 'viewport_hydrate', { clearViewportFirst: true, - ...viewportHydrateReplayOptions(currentContent), + ...viewportHydrateReplayOptions(currentContent, isPacedReplayNegotiated()), }) } dispatch(consumePaneRefreshRequest({ tabId, paneId, requestId: request.requestId })) return true - }, [attachTerminal, dispatch, paneId, registerForBackgroundHydration, suppressNetworkEffects, tabId, ws]) + }, [attachTerminal, dispatch, isLifetimeClaimNegotiated, isPacedReplayNegotiated, paneId, registerForBackgroundHydration, suppressNetworkEffects, tabId, ws]) + + // Explicit retry of bounded recovery (WS2): the visible retry state's + // control. Resets the accounting and resumes recovery — never a kill, a + // replacement, or an identity change; the preserved surface content stays + // and the attach resumes from the checkpoint when one is valid. + const retryTerminalRestore = useCallback(() => { + const tid = terminalIdRef.current + recoveryAccountingRef.current = resetRecoveryAccounting( + recoveryAccountingRef.current, + surfaceCoverageSeqRef.current, + Date.now(), + ) + setRecoveryExhausted(false) + recordTerminalPerfAuditEvent('terminal.restore.recovery_retry', { + terminalId: tid, + coverageSeq: surfaceCoverageSeqRef.current, + }) + if (tid) { + attachTerminalRef.current?.(tid, 'transport_reconnect') + } + }, [recordTerminalPerfAuditEvent]) // Apply settings changes useEffect(() => { @@ -3299,10 +3884,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const deferred = deferredAttachStateRef.current if (tid && deferred.mode === 'waiting_for_geometry' && deferred.pendingIntent) { // Unregister from background queue — this tab is now being directly hydrated - if (hydrationRegisteredRef.current) { - getHydrationQueue().unregister(paneId) - hydrationRegisteredRef.current = false - } + unregisterBackgroundHydration() getHydrationQueue().onActiveTabChanged(tabId, tabOrderRef.current) const checkpointDecision = getCheckpointDeltaReplayDecision(tid) const revealPlan = resolveRevealAttachPlan({ @@ -3317,21 +3899,28 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te suppressNextMatchingResize: true, skipPreAttachFit: true, ...(revealPlan.intent === 'viewport_hydrate' - ? viewportHydrateReplayOptions(contentRef.current) + ? viewportHydrateReplayOptions(contentRef.current, isPacedReplayNegotiated()) : undefined), }) return } requestTerminalLayout({ fit: true, resize: true }) } - }, [hidden, isTerminal, paneId, requestTerminalLayout, tabId, attachTerminal, getCheckpointDeltaReplayDecision]) + }, [hidden, isTerminal, requestTerminalLayout, tabId, attachTerminal, getCheckpointDeltaReplayDecision, isPacedReplayNegotiated, unregisterBackgroundHydration]) - // Background hydration: triggered by the hydration queue for hidden tabs + // Background hydration: triggered by the hydration queue for hidden tabs. + // OLD-SERVER FALLBACK ONLY (responsive-terminal-restore WS1): a + // claim-negotiated hidden pane never registers, so a grant here means a + // stale pre-renegotiation registration — drop it and attach nothing. useEffect(() => { if (!backgroundHydrationTriggered) return setBackgroundHydrationTriggered(false) const tid = terminalIdRef.current if (!tid || !hiddenRef.current) return + if (isLifetimeClaimNegotiated()) { + unregisterBackgroundHydration() + return + } const checkpointDecision = getCheckpointDeltaReplayDecision(tid) if (checkpointDecision.ok) { attachTerminal(tid, 'keepalive_delta', { @@ -3344,9 +3933,9 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te attachTerminal(tid, 'viewport_hydrate', { clearViewportFirst: true, priority: 'background', - ...viewportHydrateReplayOptions(contentRef.current), + ...viewportHydrateReplayOptions(contentRef.current, isPacedReplayNegotiated()), }) - }, [backgroundHydrationTriggered, attachTerminal, getCheckpointDeltaReplayDecision]) + }, [backgroundHydrationTriggered, attachTerminal, getCheckpointDeltaReplayDecision, isPacedReplayNegotiated, isLifetimeClaimNegotiated, unregisterBackgroundHydration]) // Create or attach to backend terminal useEffect(() => { @@ -3581,100 +4170,48 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te return true } - const completeDurableReplacement = (pending: PendingDurableReplacement) => { - if (pendingDurableReplacementRef.current?.requestId !== pending.requestId) { - return - } - pendingDurableReplacementRef.current = null - addTerminalRestoreRequestId(pending.requestId) - requestIdRef.current = pending.requestId - terminalIdRef.current = undefined - launchAttemptRef.current = null - reviveAttemptedRef.current = null - clearQuarantineRepair() - currentAttachRef.current = null - deferredAttachStateRef.current = { - mode: 'none', - pendingIntent: null, - pendingSinceSeq: 0, - pendingReason: 'initial_hydrate', - } - setIsAttaching(false) - setTruncatedHistoryGap(null) - dispatch(clearPaneRuntimeActivity({ paneId: paneIdRef.current })) - applySeqState(createAttachSeqState()) - updateContent({ - terminalId: undefined, - serverInstanceId: undefined, - streamId: undefined, - createRequestId: pending.requestId, - status: 'creating', - restoreError: undefined, - }) - const currentTab = tabRef.current - if (currentTab) { - dispatch(updateTab({ id: currentTab.id, updates: { status: 'creating' } })) - } - } - - const beginOpenCodeReplacementAfterExit = (terminalId: string) => { - const current = contentRef.current - const sessionRef = current?.sessionRef - if ( - current?.mode !== 'opencode' - || sessionRef?.provider !== 'opencode' - || !sessionRef.sessionId - ) { - return false - } - - const existing = pendingDurableReplacementRef.current - if (existing?.terminalId === terminalId) { - return true - } - - const requestId = nanoid() - pendingDurableReplacementRef.current = { - terminalId, - requestId, - reason: 'opencode_replay_window_exceeded', - } - clearRateLimitRetry() - clearQuarantineRepair() - currentAttachRef.current = null - launchAttemptRef.current = null - deferredAttachStateRef.current = { - mode: 'none', - pendingIntent: null, - pendingSinceSeq: 0, - pendingReason: 'initial_hydrate', - } - setIsAttaching(true) - setTruncatedHistoryGap(null) - dispatch(clearPaneRuntimeActivity({ paneId: paneIdRef.current })) - clearTerminalCursor(terminalId) - resetParserAppliedSurface() - forgetSentViewport(terminalId) - lastSentViewportRef.current = null - applySeqState(createAttachSeqState()) - writeLocalXtermNotice(term, '\r\n[Restarting OpenCode session because the saved terminal replay is no longer available]\r\n') - // b8ke ext r20 F2: the replacement kill carries the session's - // observed (epoch, generation) pair — a reconnect-queued stale - // kill is typed-refused instead of killing a newer owner. - sendTerminalKill( - terminalId, - resolveTerminalKillFence(appStore, { - provider: sessionRef?.provider, - sessionRef: sessionRef ?? undefined, - }), - ) - return true - } - async function ensure() { clearRateLimitRetry() // Connection is owned by App.tsx; messages will queue until ready + // Reconcile-episode key derivation (M-1), BEFORE any early return so + // the anchor updates on EVERY effect run: an attach fired because the + // pending window OPENED, because the verdict FOLDED (pending cleared — + // epoch bumped for corrective folds, unchanged for the task-009b + // no-change folds whose re-drive still fires on the non-negotiated + // lane), or because a fold landed without a pending window, carries + // the episode's key into the recovery gate. Keyless runs (mount, + // transport reconnects, other dep changes) count normally. + const reconcileEpoch = terminalContent?.reconcileEpoch ?? 0 + const previousReconcileDrive = reconcileDriveStateRef.current + let reconcileAttemptKey: string | undefined + if (previousReconcileDrive) { + if (reconcilePendingSince !== undefined && reconcilePendingSince !== previousReconcileDrive.pendingSince) { + // A new reconcile pending window opened (setReconcilePendingPanes) + // — the episode key is the window's startedAt. + reconcileAttemptKey = `pending:${reconcilePendingSince}` + } else if ( + reconcilePendingSince === undefined + && previousReconcileDrive.pendingSince !== undefined + ) { + // The window this run closes (the verdict fold — or its bounded + // timeout release) is the SAME episode's closing attach, whether + // or not the fold bumped the epoch. + reconcileAttemptKey = `pending:${previousReconcileDrive.pendingSince}` + } else if (reconcileEpoch !== previousReconcileDrive.epoch) { + // A fold without a pending window (the exhaustion reconcile, the + // D7 revival / restore-offer fold): each fold is its own + // one-count episode. + reconcileAttemptKey = `epoch:${reconcileEpoch}` + } + } + reconcileDriveStateRef.current = { + pendingSince: reconcilePendingSince, + epoch: reconcileEpoch, + createRequestId, + terminalId: terminalIdRef.current, + } + const failLaunch = (message: string, restore: boolean, terminalId?: string) => { clearRateLimitRetry() clearQuarantineRepair() @@ -3925,6 +4462,8 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te mode: TerminalPaneContent['mode'] terminalInstanceId: string parserAppliedSeq: number + seqStart: number + seqEnd: number completedAttach: boolean }) => { const activeAttach = currentAttachRef.current @@ -3941,6 +4480,25 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const nextSeqState = markParserAppliedSeq(seqStateRef.current, input.parserAppliedSeq) applySeqState(nextSeqState) markParserAppliedFrame(tid, nextSeqState.parserAppliedSeq, activeAttach) + // Surface-coverage cursor (WS2): an applied frame advances BOTH + // cursors — the coverage advance is contiguity-gated against the + // coverage cursor itself, so it extends past null-screen-effect + // filtered ranges the strict applied position refuses to cross, + // but never past a lost/unapplied hole. The advance persists the + // checkpoint so the coverage position survives an interrupt even + // when the strict applied position did not move (mixed page). + advanceSurfaceCoverageAndPersist( + tid, + activeAttach, + input.seqStart, + input.seqEnd, + ) + // Paced replay consumption (Workstream 1): the write-queue applied + // this frame — the consumption frontier advances independent of the + // parser-applied checkpoint's quarantine clamping (the checkpoint + // still refuses filtered/lost ranges; the frontier is about + // consumption, not surface state). + advancePacedReplayConsumption(input.attachRequestId, input.parserAppliedSeq) if (input.completedAttach) { completeAttachGeneration({ attachRequestId: input.attachRequestId, @@ -4052,6 +4610,8 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te mode: input.mode, terminalInstanceId: outputTerminalInstanceId, parserAppliedSeq: input.parserAppliedSeq, + seqStart: input.seqStart, + seqEnd: input.seqEnd, completedAttach: input.completedAttach, }) : undefined, @@ -4065,6 +4625,36 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te || !inputBytesEqualSubmission || !submission.submittedBytesEqualInput ) { + // Paced replay consumption (Workstream 1): a frame the + // null-screen-effect pre-parsers fully consumed (nothing reached + // the queue, no replay-discard mutation) IS consumed — the + // frontier advances without any xterm write. M2 (task-4 review): + // `preParserConsumedAllBytes` is the exact `cleaned === ''` + // condition, so an enqueue failure (no write queue, disposed + // surface, thrown write) can NEVER masquerade as consumption — + // unconsumed bytes must never be credited. Partial or unknown + // mutations keep today's quarantine and forfeit the range's + // credit (the server's retention/expiry handling covers it). + if ( + !submission.submittedWrite + && inputBytesEqualSubmission + && submission.preParserConsumedAllBytes + ) { + advancePacedReplayConsumption(input.attachRequestId, input.seqEnd) + schedulePacedReplayCreditFlush(input.outputSource, input.attachRequestId) + // Surface-coverage cursor (WS2): a fully pre-filtered frame + // advances ONLY the coverage cursor (the strict applied + // position stays pinned below it and keeps the unapplied-range + // quarantine above) — and persists it, so a disconnect right + // after a filtered page resumes from past it instead of + // re-delivering (and re-firing) the filtered bytes. + advanceSurfaceCoverageAndPersist( + tid, + currentAttachRef.current, + input.seqStart, + input.seqEnd, + ) + } applySeqState(markOutputRangeUnapplied(seqStateRef.current, { fromSeq: input.seqStart, toSeq: input.seqEnd, @@ -4248,6 +4838,11 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te } if (!outputSource) return + // Deferred-clear repair hydrate (round-3 fix): the clear now + // consumes at the RENDER moment inside `submitAcceptedOutput`'s + // write path — never here on envelope arrival. A rejected + // duplicate/overlap or a fully filtered frame must not consume + // it; the next writing frame clears-then-writes. const previousSeqState = seqStateRef.current const batchDecision = onOutputBatchSegments(previousSeqState, batchSegments) if (!batchDecision.accept) { @@ -4262,6 +4857,33 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te return } + if (tid && batchDecision.implicitGaps.length > 0) { + // Implicit gap (responsive-terminal-restore): the batch jumped + // forward across sequences no gap frame declared. The seq state + // already folded the hole (known lost range + quarantine, the + // applied cursor pinned below it); surface it honestly — the + // quarantine is observable, and a local notice names the + // exact lost range. Never a silent applied-cursor advance. + for (const implicitGap of batchDecision.implicitGaps) { + recordTerminalPerfAuditEvent('terminal.catchup.surface_quarantined', { + terminalId: tid, + attachRequestId: msg.attachRequestId, + activeAttachRequestId: currentAttachRef.current?.requestId, + streamId: msg.streamId, + fromSeq: implicitGap.fromSeq, + toSeq: implicitGap.toSeq, + parserAppliedSeq: parserAppliedSeqRef.current, + highestObservedSeq: batchDecision.state.highestObservedSeq, + reason: 'implicit_sequence_jump', + }) + writeLocalXtermNotice( + term, + `\r\n[Output gap ${implicitGap.fromSeq}-${implicitGap.toSeq}: unexplained sequence jump]\r\n`, + ) + } + resetParserAppliedSurface(parserAppliedSeqRef.current) + } + if (tid && batchDecision.freshReset) { clearTerminalCursor(tid) resetParserAppliedSurface() @@ -4271,6 +4893,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const completedAttachOnBatch = !batchDecision.state.pendingReplay && (Boolean(previousSeqState.pendingReplay) || previousSeqState.awaitingFreshSequence) applySeqState(batchDecision.state) + markPacedReplayReceived(msg.attachRequestId, batchSeqEnd) const containsBarrier = batchSegments.some((segment) => segment.barrier) if (!containsBarrier) { @@ -4331,6 +4954,11 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te } return } + // Deferred-clear repair hydrate (round-3 fix): the clear now + // consumes at the RENDER moment inside `submitAcceptedOutput`'s + // write path — never here on envelope arrival. A rejected + // duplicate/overlap or a fully filtered frame must not consume + // it; the next writing frame clears-then-writes. const previousSeqState = seqStateRef.current const frameDecision = onOutputFrame(previousSeqState, { seqStart: msg.seqStart, @@ -4349,6 +4977,31 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te return } + if (tid && frameDecision.implicitGap) { + // Implicit gap (responsive-terminal-restore): the frame jumped + // forward across sequences no gap frame declared. The seq state + // already folded the hole (known lost range + quarantine, the + // applied cursor pinned below it); surface it honestly — the + // quarantine is observable, and a local notice names the exact + // lost range. Never a silent applied-cursor advance. + recordTerminalPerfAuditEvent('terminal.catchup.surface_quarantined', { + terminalId: tid, + attachRequestId: msg.attachRequestId, + activeAttachRequestId: currentAttachRef.current?.requestId, + streamId: msg.streamId, + fromSeq: frameDecision.implicitGap.fromSeq, + toSeq: frameDecision.implicitGap.toSeq, + parserAppliedSeq: parserAppliedSeqRef.current, + highestObservedSeq: frameDecision.state.highestObservedSeq, + reason: 'implicit_sequence_jump', + }) + writeLocalXtermNotice( + term, + `\r\n[Output gap ${frameDecision.implicitGap.fromSeq}-${frameDecision.implicitGap.toSeq}: unexplained sequence jump]\r\n`, + ) + resetParserAppliedSurface(parserAppliedSeqRef.current) + } + if (tid && frameDecision.freshReset) { clearTerminalCursor(tid) resetParserAppliedSurface() @@ -4362,6 +5015,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const completedAttachOnFrame = !frameDecision.state.pendingReplay && (Boolean(previousSeqState.pendingReplay) || previousSeqState.awaitingFreshSequence) applySeqState(frameDecision.state) + markPacedReplayReceived(msg.attachRequestId, msg.seqEnd) submitAcceptedOutput({ raw: msg.data || '', seqStart: msg.seqStart, @@ -4393,29 +5047,101 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te // Only show "load more" when the server confirms the gap is from // byte-budget truncation (recoverable), not ring overflow (data gone). + const pacedReplayNegotiated = isPacedReplayNegotiated() const isTruncatedReplay = msg.reason === 'replay_budget_exceeded' && seqStateRef.current.pendingReplay - const isUnrecoverableOpenCodeViewportHydrate = msg.reason === 'replay_window_exceeded' - && currentAttachRef.current?.intent === 'viewport_hydrate' - && currentAttachRef.current.sinceSeq === 0 - && contentRef.current?.mode === 'opencode' - && contentRef.current.sessionRef?.provider === 'opencode' - if (isUnrecoverableOpenCodeViewportHydrate && beginOpenCodeReplacementAfterExit(tid)) { - return - } + // The delivery-loss repair decision (pure — computed from the + // pre-gap seq state; the state itself is applied further below). + // Hoisted ABOVE the notice branches so the queue_overflow / + // handoff_boundary_reached notice can be SUPPRESSED here and + // re-emitted UNDER THE REPAIR'S NEW generation below: the repair + // attach mints a new generation with dropQueuedStaleWrites, so a + // notice enqueued under the OLD generation is discarded before + // the animation-frame flush can render it (round-4 F4 — the + // honest notice must survive the repair). + const gapDecisionForNotice = onOutputGap(seqStateRef.current, { + fromSeq: msg.fromSeq, + toSeq: msg.toSeq, + }) + const deliveryRepairPending = pacedReplayNegotiated + && (msg.reason === 'queue_overflow' || msg.reason === 'handoff_boundary_reached') + && gapDecisionForNotice.requiresSurfaceQuarantine + // Retention gaps NEVER trigger an automatic OpenCode replacement + // (responsive-terminal-restore WS2): the auto-kill path is removed + // entirely — a retention gap is honest state, not a license to + // kill or replace a healthy process. Any explicit restart remains + // separate user intent. Negotiated gaps show the accessible + // incomplete-history notice; the legacy path keeps its local gap + // notice (old servers never emit this gap shape anyway — the paced + // core is its only emitter and it requires negotiation). if (isTruncatedReplay) { setTruncatedHistoryGap({ fromSeq: msg.fromSeq, toSeq: msg.toSeq }) + } else if (pacedReplayNegotiated && msg.reason === 'replay_window_exceeded') { + // Honest incomplete-history state on the paced path: some earlier + // output is gone; live output continues. Absent bounds fields + // mean UNKNOWN BOUNDS (the terminal may have vanished at gap + // emission) — never non-negotiation. + const gapHeadSeq = typeof msg.headSeq === 'number' ? msg.headSeq : null + const gapOldestRetainedSeq = typeof msg.oldestRetainedSeq === 'number' ? msg.oldestRetainedSeq : null + setRetentionLossNotice({ + fromSeq: msg.fromSeq, + toSeq: msg.toSeq, + headSeq: gapHeadSeq, + oldestRetainedSeq: gapOldestRetainedSeq, + }) + // The retention-loss notice is the AUTHORITATIVE honest state + // here — a delivery-gap repair whose fetch resolved to + // unrecoverable retention must not leave its (now-stale) + // "refetching" chrome claim on screen alongside it. + setDeliveryGapNotice(null) + recordTerminalPerfAuditEvent('terminal.restore.retention_gap', { + terminalId: tid, + attachRequestId: msg.attachRequestId, + fromSeq: msg.fromSeq, + toSeq: msg.toSeq, + headSeq: gapHeadSeq, + oldestRetainedSeq: gapOldestRetainedSeq, + }) + // Round-2 disarm (the deferred content reset must not survive + // a server-declared unreconstructible prefix): when THIS + // generation's repair hydrate deferred its viewport clear + // (the no-checkpoint fallback), the retention gap declares + // the missing prefix CANNOT be rebuilt — the pre-gap screen + // is the best available surface, so the pending clear is + // DISARMED here, before any suffix frame can consume it and + // wipe the screen mid-restore. The honest-loss UX proceeds + // and the retained suffix appends onto the preserved + // surface. + if ( + typeof msg.attachRequestId === 'string' + && repairContentResetPendingRef.current === msg.attachRequestId + ) { + repairContentResetPendingRef.current = null + } + } else if (deliveryRepairPending) { + // Suppressed here — the negotiated delivery-loss repair owns + // the notice (round-5 finding 2): the honest notice rides UI + // CHROME below (set after the repair attach fires), never a + // local xterm write. The pre-fix immediate write mutated the + // surface BETWEEN the checkpoint state and the repair's + // replayed bytes — corrupting the repaired screen whenever the + // gap straddled an in-flight escape sequence or cursor + // position; the legacy non-negotiated lane (the else branch) + // keeps its existing immediate-notice behavior. } else { const reason = msg.reason === 'replay_window_exceeded' ? 'reconnect window exceeded' - : 'slow link backlog' + : msg.reason === 'handoff_boundary_reached' + ? 'restore boundary reached' + : 'slow link backlog' writeLocalXtermNotice(term, `\r\n[Output gap ${msg.fromSeq}-${msg.toSeq}: ${reason}]\r\n`) } const previousSeqState = seqStateRef.current const gapDecision = onOutputGap(previousSeqState, { fromSeq: msg.fromSeq, toSeq: msg.toSeq }) const nextSeqState = gapDecision.state applySeqState(nextSeqState) + markPacedReplayReceived(msg.attachRequestId, msg.toSeq) resetParserAppliedSurface(parserAppliedSeqRef.current) if (gapDecision.requiresSurfaceQuarantine) { recordTerminalPerfAuditEvent('terminal.catchup.surface_quarantined', { @@ -4437,6 +5163,85 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te setIsAttaching(false) markAttachComplete() } + // Delivery-loss repair (responsive-terminal-restore WS3→WS2, + // negotiated lane only): a queue_overflow gap on the STILL-OPEN + // connection means this connection missed sequenced output — the + // shared restore contract requires repair from retained output, + // never silent advancement and never a stranded screen. The + // round-4 fixed-boundary exit (plan:146) is the SAME delivery + // loss: the paced session completed at its FIXED boundary with + // output staged past it, and the server declared the exact + // retained interval as the `handoff_boundary_reached` gap — the + // ring still holds it, so the identical checkpoint-cursor + // repair fetches it. In both shapes the ring retained the + // declared range, so the repair RESUMES from the surface + // checkpoint cursor (the checkpoint-aware delta resume): a delta + // attach refills the visible surface WITHOUT clearing it. One + // bounded repair attach per gap, through the recovery accounting + // — repeated gaps exhaust to the visible retry strip. If the + // server answers that retention has expired past the cursor (the + // bounds-carrying gap), the existing honest-loss UX applies + // below — never a destructive rebuild. Only when no valid + // checkpoint exists may the repair fall back to a full hydrate, + // and even then the viewport is not cleared before the new + // baseline is actually established by attach content (the + // deferred content reset). Old servers never emit these + // negotiated gap shapes; their local-notice behavior (pinned + // above) is unchanged. + if ( + pacedReplayNegotiated + && (msg.reason === 'queue_overflow' || msg.reason === 'handoff_boundary_reached') + && gapDecision.requiresSurfaceQuarantine + ) { + recordTerminalPerfAuditEvent('terminal.restore.queue_overflow_repair', { + terminalId: tid, + attachRequestId: msg.attachRequestId, + activeAttachRequestId: currentAttachRef.current?.requestId, + fromSeq: msg.fromSeq, + toSeq: msg.toSeq, + reason: msg.reason, + }) + // The gap's local-notice invalidation bumped the surface + // epoch; re-save the quarantined surface's pinned cursor under + // the new epoch so the delta repair's checkpoint decision + // validates and resumes from the coverage cursor. + persistSurfaceCheckpointForAttach(tid, currentAttachRef.current, parserAppliedSeqRef.current) + const checkpointDecision = getCheckpointDeltaReplayDecision(tid) + if (checkpointDecision.ok) { + attachTerminal(tid, 'transport_reconnect', { + clearViewportFirst: false, + sinceSeq: checkpointDecision.sinceSeq, + }) + } else { + attachTerminal(tid, 'viewport_hydrate', { + clearViewportFirst: false, + deferViewportClearUntilContent: true, + ...viewportHydrateReplayOptions(contentRef.current, pacedReplayNegotiated), + }) + } + // THE HONEST NOTICE RIDES CHROME (round-5 finding 2, the + // round-4 F4 fix's successor): during a checkpoint-based + // delta repair NOTHING may write to the xterm surface before + // the replayed bytes apply — the pre-fix local notice was + // enqueued right here, between the surface state the + // checkpoint captured and the repair's replayed terminal + // instructions, shifting the cursor/parser state the + // replayed bytes were addressed to (an in-flight multi-frame + // escape sequence made the corruption direct: the notice + // bytes were consumed as escape continuation). The notice is + // React state — set AFTER the repair attach (which clears + // stale notices) — so it is honestly shown through the + // generation change, and the repaired surface stays exactly + // the uninterrupted reference. + const noticeReason = msg.reason === 'handoff_boundary_reached' + ? 'restore boundary reached' + : 'slow link backlog' + setDeliveryGapNotice({ + fromSeq: msg.fromSeq, + toSeq: msg.toSeq, + reason: noticeReason, + }) + } } if (msg.type === 'terminal.stream.changed' && msg.terminalId === tid) { @@ -4541,7 +5346,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te updateContent({ streamId: undefined }) attachTerminal(tid, 'viewport_hydrate', { clearViewportFirst: true, - ...viewportHydrateReplayOptions(contentRef.current), + ...viewportHydrateReplayOptions(contentRef.current, isPacedReplayNegotiated()), }) return } @@ -4576,7 +5381,7 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te resetParserAppliedSurface(parserAppliedSeqRef.current) attachTerminal(tid, 'viewport_hydrate', { clearViewportFirst: true, - ...viewportHydrateReplayOptions(contentRef.current), + ...viewportHydrateReplayOptions(contentRef.current, isPacedReplayNegotiated()), }) return } @@ -4628,6 +5433,29 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te replayToSeq: msg.replayToSeq, }) applySeqState(nextSeqState) + // Paced replay consumption (Workstream 1): record the session + // window end (replayToSeq on a paced ready describes the SESSION + // window — the fixed catch-up target — never a single page's + // bounds) and the restore-contract fields. Legacy-path ready frames + // on a negotiated connection (an already-Exited terminal) carry the + // contract fields too; their credits are inert server-side. + const pacedReadyState = pacedReplayRef.current + if (pacedReadyState && msg.attachRequestId === pacedReadyState.attachRequestId) { + pacedReplayRef.current = pacedReplayOnReady(pacedReadyState, { + replayToSeq: msg.replayToSeq, + }) + recordTerminalPerfAuditEvent('terminal.restore.paced_ready', { + terminalId: tid, + attachRequestId: msg.attachRequestId, + requestedSinceSeq: msg.requestedSinceSeq, + effectiveSinceSeq: msg.effectiveSinceSeq, + oldestRetainedSeq: msg.oldestRetainedSeq, + replayResetReason: msg.replayResetReason, + headSeq: msg.headSeq, + replayFromSeq: msg.replayFromSeq, + replayToSeq: msg.replayToSeq, + }) + } setIsAttaching(Boolean(nextSeqState.pendingReplay)) if (!nextSeqState.pendingReplay) { // Completion-clear edge for empty-tracker / no-replay attaches @@ -4804,11 +5632,21 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te pendingReason: 'terminal_created', } setIsAttaching(false) - // F8: a hidden pane still owes the server an attach. Drive it - // through the background hydration queue (one-at-a-time stagger) - // instead of waiting for reveal -- otherwise the terminal sits - // detached server-side and is idle-reaped after 15 minutes. - // Order matters (verified queue semantics): + // Lifetime (responsive-terminal-restore WS1): a claim-negotiated + // hidden pane NEVER attaches while hidden — its terminalId is + // claimed in the interest snapshot (TerminalInterestReporter) and + // the deferred intent above arms the reveal hydrate. The + // server-side claim keeps the terminal alive without replay. + if (isLifetimeClaimNegotiated()) { + unregisterBackgroundHydration() + return + } + // F8 (old-server fallback — no claim echo): a hidden pane still + // owes the server an attach. Drive it through the background + // hydration queue (one-at-a-time stagger) instead of waiting for + // reveal -- otherwise the terminal sits detached server-side and + // is idle-reaped after 15 minutes. Order matters (verified queue + // semantics): // 1) clear this pane's stale active slot -- a background // hydration that died with the old connection otherwise // wedges the whole queue (no-op when not the active pane); @@ -4959,11 +5797,6 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te exitCode: msg.exitCode, at: Date.now(), })) - const pendingReplacement = pendingDurableReplacementRef.current - if (pendingReplacement?.terminalId === tid) { - completeDurableReplacement(pendingReplacement) - return - } const launchAttempt = launchAttemptRef.current const exitedDuringLaunch = launchAttempt?.terminalId === tid && !launchAttempt.attachReady @@ -5366,7 +6199,6 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te const currentTerminalId = terminalIdRef.current const current = contentRef.current const launchAttempt = launchAttemptRef.current - const pendingReplacement = pendingDurableReplacementRef.current if (debugRef.current) log.debug('[TRACE resumeSessionId] INVALID_TERMINAL_ID received', { paneId: paneIdRef.current, msgTerminalId: msg.terminalId, @@ -5376,13 +6208,6 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te currentResumeSessionId: current?.resumeSessionId, currentStatus: current?.status, }) - if ( - pendingReplacement - && (!msg.terminalId || msg.terminalId === pendingReplacement.terminalId) - ) { - completeDurableReplacement(pendingReplacement) - return - } if (msg.terminalId && msg.terminalId !== currentTerminalId) { // Show feedback if the terminal already exited (the ID was cleared by // the exit handler, so msg.terminalId no longer matches the ref) @@ -5654,9 +6479,19 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te pendingSinceSeq: 0, pendingReason: 'hidden_reveal', } - // Same three-step re-register as the terminal.created hidden path - // (~:4506): a stale active slot or a consumed registration guard - // otherwise wedges this pane out of the post-reconnect pump entirely. + // Lifetime (responsive-terminal-restore WS1): a claim-negotiated + // hidden pane reconnects WITHOUT attaching — the fresh connection's + // interest snapshot re-claims its terminalId (the ready-frame + // re-flush carries the claim set), and the deferred intent above + // arms the reveal attach. + if (isLifetimeClaimNegotiated()) { + unregisterBackgroundHydration() + return + } + // Old-server fallback: same three-step re-register as the + // terminal.created hidden path (~:4506): a stale active slot or a + // consumed registration guard otherwise wedges this pane out of the + // post-reconnect pump entirely. getHydrationQueue().onHydrationComplete(paneIdRef.current) hydrationRegisteredRef.current = false // Always queueIfStarted: a hidden pane's reattach must not wait for @@ -5710,15 +6545,76 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te } setIsAttaching(false) - // Register with hydration queue for progressive background hydration + // Lifetime (responsive-terminal-restore WS1): a claim-negotiated + // hidden pane does NOT register for background hydration — its + // terminalId is claimed in the interest snapshot, and the deferred + // intent above arms the reveal attach. + if (isLifetimeClaimNegotiated()) { + unregisterBackgroundHydration() + return + } + // Old-server fallback: register with hydration queue for + // progressive background hydration. registerForBackgroundHydration() } else { - const intent: AttachIntent = deferredAttachStateRef.current.mode === 'live' + // Goal-4 re-drive gate (task-009b): a reconcile-driven re-run whose + // pane identity is unchanged — the ready round's pending-window + // opening, or a no-change verdict fold closing it + // (applyReconcileAttach skips the epoch bump for no-change folds) + // — must NOT re-attach a pane whose attach lifecycle is already in + // motion. The old mode heuristic re-fired a viewport_hydrate here + // on every reconnect flap and superseded the in-flight checkpoint + // delta resume with a full refetch. Scoped to the negotiated + // (paced-replay) lane AND to a current attach generation that was + // itself built post-negotiation (pacedReplayRef is absent for + // non-negotiated attaches): a pre-ready MOUNT attach carries the + // legacy 128 KiB truncation budget, and the re-drive chain was + // load-bearing as the repair that replaced it with the full paced + // restore — such a pane must still re-drive. Old servers + // (no paced replay) keep today's exact re-drive chain. Corrective + // folds (epoch changed), identity changes, and an un-anchored pane + // (mode 'none') always re-drive below. + const deferredMode = deferredAttachStateRef.current.mode + const reconcileNoopRedrive = isPacedReplayNegotiated() + && pacedReplayRef.current !== null + && previousReconcileDrive !== null + && reconcileEpoch === previousReconcileDrive.epoch + && reconcilePendingSince !== previousReconcileDrive.pendingSince + && createRequestId === previousReconcileDrive.createRequestId + && currentTerminalId === previousReconcileDrive.terminalId + && (deferredMode === 'live' || deferredMode === 'attaching') + if (reconcileNoopRedrive) { + if (debugRef.current) { + log.debug('[TRACE resumeSessionId] skipping reconcile re-drive: identity unchanged, attach lifecycle in motion', { + paneId: paneIdRef.current, + terminalId: currentTerminalId, + reconcileEpoch, + reconcilePendingSince, + deferredMode, + }) + } + return + } + // Checkpoint-aware re-drive intent (task-009b): a corrective fold + // that lands mid-attach used to pick viewport_hydrate off the + // 'attaching' mode alone and refetch the whole retained buffer; + // on the negotiated lane, with a healthy checkpoint for THIS + // surface, the re-drive resumes instead (attachTerminal's internal + // geometry-aware decision still falls back to viewport_hydrate + // when the checkpoint is unusable — the same contract as the + // hidden branch above). Non-negotiated connections keep the mode + // heuristic exactly. + const checkpointDecision = getCheckpointDeltaReplayDecision(currentTerminalId) + const intent: AttachIntent = deferredMode === 'live' + || (isPacedReplayNegotiated() && checkpointDecision.ok) ? 'keepalive_delta' : 'viewport_hydrate' - attachTerminal(currentTerminalId, intent, intent === 'viewport_hydrate' - ? viewportHydrateReplayOptions(contentRef.current) - : undefined) + attachTerminal(currentTerminalId, intent, { + ...(intent === 'viewport_hydrate' + ? viewportHydrateReplayOptions(contentRef.current, isPacedReplayNegotiated()) + : undefined), + ...(reconcileAttemptKey !== undefined ? { recoveryAttemptKey: reconcileAttemptKey } : {}), + }) // One-shot reconcile notice (attach/corrected/duplicate verdicts): // render it on the attach that the verdict fold re-fired, then // clear it from the store. @@ -5865,9 +6761,11 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te getTerminalCheckpointStreamId, isCurrentAttachMessage, isCurrentAttachStreamMessage, + isLifetimeClaimNegotiated, markAttachComplete, markParserAppliedFrame, markTerminalOutputRangeLost, + advanceSurfaceCoverageAndPersist, recordTerminalPerfAuditEvent, registerForBackgroundHydration, resetParserAppliedSurface, @@ -5932,9 +6830,9 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te ): Promise => { const tid = terminalIdRef.current if (!tid) return null - // Advisory guard (matrix arm E): the opencode replay-window replacement - // flow already owns this pane's recovery — a second kill would race it. - if (pendingDurableReplacementRef.current) return null + // (The opencode replay-window replacement flow's advisory guard is gone + // with the flow itself: this branch's checkpoint-resume contract replaces + // the durable-PTY-replacement mechanism — the ref is never set anymore.) // In-flight re-entrancy lock (matrix arm G, review Minor-1): a second // click while this pane's kill-await is outstanding is a no-op. Both // stuck-card actions share this entry, so the lock covers restart AND @@ -6295,8 +7193,61 @@ function TerminalView({ tabId, paneId, paneContent, hidden, focusEpoch = 0 }: Te )} + {retentionLossNotice && ( + // Honest incomplete-history state (responsive-terminal-restore): a + // negotiated retention gap's accessible notice — live output keeps + // flowing below it. +
+ Some earlier terminal output is no longer available on the server. Live output continues. +
+ )} + {deliveryGapNotice && ( + // Honest delivery-loss state (responsive-terminal-restore, + // round-5 finding 2): a negotiated queue_overflow / + // handoff_boundary_reached gap's accessible CHROME notice. Never a + // surface write — during the checkpoint delta repair NOTHING may + // mutate the xterm surface before the replayed bytes apply, so the + // notice rides React state and the screen below it is repaired to + // exactly the uninterrupted reference. +
+ {`Terminal output gap ${deliveryGapNotice.fromSeq}-${deliveryGapNotice.toSeq} (${deliveryGapNotice.reason}). The missing output is being refetched — the screen below is completed as it arrives.`} +
+ )} + {recoveryExhausted && ( + // Bounded automatic recovery (responsive-terminal-restore WS2): the + // visible, accessible retry state. Automatic re-attach cycling was + // stopped after repeated attempts with no restore progress; the + // terminal content below is PRESERVED and an explicit retry resumes. +
+ Terminal restore is not making progress. What is on screen is unchanged. + +
+ )} {truncatedHistoryGap && ( -
+