diff --git a/scripts/config-templates/azure-config-template.yaml b/scripts/config-templates/azure-config-template.yaml index 2a86f0124..419fd1e85 100644 --- a/scripts/config-templates/azure-config-template.yaml +++ b/scripts/config-templates/azure-config-template.yaml @@ -40,7 +40,7 @@ tcp_socket_options: nodelay: true inetstack_config: mtu: 1500 - mss: 1500 + # mss: 1460 # optional: defaults to the MSS derived from the mtu above enable_jumbo_frames: false udp_checksum_offload: false tcp_checksum_offload: false diff --git a/scripts/config-templates/baremetal-config-template.yaml b/scripts/config-templates/baremetal-config-template.yaml index a82a1721e..bb2f63c96 100644 --- a/scripts/config-templates/baremetal-config-template.yaml +++ b/scripts/config-templates/baremetal-config-template.yaml @@ -43,7 +43,7 @@ tcp_socket_options: nodelay: true inetstack_config: mtu: 1500 - mss: 1500 + # mss: 1460 # optional: defaults to the MSS derived from the mtu above enable_jumbo_frames: false udp_checksum_offload: false tcp_checksum_offload: false diff --git a/src/catpowder/win/runtime.rs b/src/catpowder/win/runtime.rs index f1a47fc32..f11139980 100644 --- a/src/catpowder/win/runtime.rs +++ b/src/catpowder/win/runtime.rs @@ -88,7 +88,10 @@ impl SharedCatpowderRuntime { let stats: CatpowderStats = CatpowderStats::new(&interface, vf_interface.as_ref())?; let always_send_on_vf: bool = config.xdp_always_send_on_vf()? && vf_interface.is_some(); - let max_body_size: usize = config.mss()? as usize - MAX_HEADER_SIZE; + // The MSS is a *payload* size, so it must not be shrunk by the header sizes again. Derive the maximum body + // size from the MTU, just like the other runtimes do. Saturating subtraction keeps a bogus (too small) MTU + // from underflowing. + let max_body_size: usize = (config.mtu()? as usize).saturating_sub(MAX_HEADER_SIZE); Ok(Self(SharedObject::new(CatpowderRuntime { api, diff --git a/src/inetstack/config/tcp.rs b/src/inetstack/config/tcp.rs index b24fa3291..0871842f1 100644 --- a/src/inetstack/config/tcp.rs +++ b/src/inetstack/config/tcp.rs @@ -7,10 +7,13 @@ use crate::{ demikernel::config::Config, - inetstack::consts::{DEFAULT_MSS, MAX_MSS, MIN_MSS, TCP_ACK_DELAY_TIMEOUT, TCP_HANDSHAKE_TIMEOUT}, + inetstack::consts::{ + default_mss_from_mtu, DEFAULT_MSS, DEFAULT_MTU, FALLBACK_MSS, MAX_MSS, MIN_MSS, TCP_ACK_DELAY_TIMEOUT, + TCP_HANDSHAKE_TIMEOUT, + }, runtime::fail::Fail, }; -use ::std::time::Duration; +use ::std::{cmp, time::Duration}; //====================================================================================================================== // Structures @@ -38,11 +41,36 @@ impl TcpConfig { pub fn new(config: &Config) -> Result { let mut options = Self::default(); - if let Ok(value) = config.mss() { - assert!(value >= MIN_MSS); - assert!(value <= MAX_MSS); - options.advertised_mss = value; + // The MSS that we advertise in our SYN and SYN+ACK segments. An explicitly configured MSS always wins; + // otherwise we derive it from the MTU, as prescribed by RFC 793 and RFC 6691. + let mtu: usize = config.mtu().map_or(DEFAULT_MTU, |value: u16| value as usize); + let derived_mss: usize = default_mss_from_mtu(mtu); + + match config.mss() { + Ok(value) => { + if !(MIN_MSS..=MAX_MSS).contains(&value) { + let cause: String = format!( + "parameter \"mss\" is out of range: {} (must be in {}..={})", + value, MIN_MSS, MAX_MSS + ); + error!("TcpConfig::new(): {}", cause); + return Err(Fail::new(libc::ERANGE, &cause)); + } + if value > derived_mss { + warn!( + "configured MSS ({}) exceeds what MTU ({}) can carry ({})", + value, mtu, derived_mss + ); + } + options.advertised_mss = value; + }, + // No MSS configured: derive it from the MTU. + Err(_) => { + options.advertised_mss = derived_mss; + info!("no MSS configured, advertising MSS derived from MTU {}: {}", mtu, derived_mss); + }, } + if let Ok(value) = config.tcp_checksum_offload() { options.rx_checksum_offload = value; options.tx_checksum_offload = value; @@ -51,10 +79,20 @@ impl TcpConfig { Ok(options) } + /// Returns the MSS that we advertise to our peers. pub fn get_advertised_mss(&self) -> usize { self.advertised_mss } + /// Returns the MSS to use for a connection whose peer advertised `received_mss`. + /// + /// The MSS option only tells us how large the segments that we *send* may be, so a peer is free to advertise an + /// MSS that is larger than what our own interface can carry. Per RFC 1122 (Section 4.2.2.6), we must then use the + /// smaller of the two values, otherwise the sender would build segments that do not fit in our transmit buffers. + pub fn get_effective_mss(&self, received_mss: Option) -> usize { + received_mss.map_or(FALLBACK_MSS, |mss: usize| cmp::min(mss, self.advertised_mss)) + } + pub fn get_handshake_retries(&self) -> usize { self.handshake_retries } @@ -109,9 +147,23 @@ impl Default for TcpConfig { #[cfg(test)] mod tests { - use crate::inetstack::{config::TcpConfig, consts::DEFAULT_MSS}; - use ::anyhow::Result; + use crate::{ + demikernel::config::Config, + inetstack::{ + config::TcpConfig, + consts::{default_mss_from_mtu, DEFAULT_MSS, DEFAULT_MTU, FALLBACK_MSS}, + }, + }; + use ::anyhow::{ensure, Result}; use ::std::time::Duration; + use ::yaml_rust::{Yaml, YamlLoader}; + + /// Builds a [Config] out of an inline `inetstack_config` section. + fn new_config(inetstack_config: &str) -> Config { + let yaml: String = format!("inetstack_config:\n{}", inetstack_config); + let docs: Vec = YamlLoader::load_from_str(&yaml).expect("should be valid yaml"); + Config(docs[0].clone()) + } #[test] fn test_tcp_config_default() -> Result<()> { @@ -126,4 +178,64 @@ mod tests { Ok(()) } + + /// The MSS should be derived from the MTU whenever it is not explicitly configured. + #[test] + fn test_tcp_config_mss_derived_from_mtu() -> Result<()> { + // Standard Ethernet MTU: 1500 - 20 (IPv4) - 20 (TCP) = 1460. + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 1500\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), 1460); + + // Jumbo frames. + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 9000\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), 8960); + + // Azure's accelerated networking MTU. + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 4000\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), 3960); + + // An MTU that cannot carry a full 536-byte segment must not underflow: we clamp to the RFC 793 fallback. + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 576\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), 536); + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 20\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), FALLBACK_MSS); + + Ok(()) + } + + /// An explicitly configured MSS must take precedence over the value derived from the MTU. + #[test] + fn test_tcp_config_mss_overrides_mtu() -> Result<()> { + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 1500\n mss: 1450\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), 1450); + + // Neither MTU nor MSS configured: we fall back to the MSS derived from DEFAULT_MTU. + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" arp_cache_ttl: 600\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), DEFAULT_MSS); + crate::ensure_eq!(tcp_config.get_advertised_mss(), default_mss_from_mtu(DEFAULT_MTU)); + + // An out-of-range MSS is rejected instead of panicking. + ensure!(TcpConfig::new(&new_config(" mtu: 1500\n mss: 535\n")).is_err()); + ensure!(TcpConfig::new(&new_config(" mtu: 1500\n mss: 65536\n")).is_err()); + + Ok(()) + } + + /// The MSS used for a connection must never exceed what we can actually send. + #[test] + fn test_tcp_config_effective_mss() -> Result<()> { + let tcp_config: TcpConfig = TcpConfig::new(&new_config(" mtu: 1500\n"))?; + crate::ensure_eq!(tcp_config.get_advertised_mss(), 1460); + + // Peer did not advertise an MSS: fall back to the RFC 793 value. + crate::ensure_eq!(tcp_config.get_effective_mss(None), FALLBACK_MSS); + + // Peer advertised something smaller: honor it. + crate::ensure_eq!(tcp_config.get_effective_mss(Some(536)), 536); + + // Peer advertised something larger (e.g., it sits on a jumbo-frame link): clamp it to our own MSS. + crate::ensure_eq!(tcp_config.get_effective_mss(Some(8960)), 1460); + + Ok(()) + } } diff --git a/src/inetstack/consts.rs b/src/inetstack/consts.rs index e702bccdb..4962b84a4 100644 --- a/src/inetstack/consts.rs +++ b/src/inetstack/consts.rs @@ -12,7 +12,23 @@ use ::std::time::Duration; // Constants //====================================================================================================================== -/// Fallback MSS Parameter for TCP +/// Size of the IPv4 and TCP headers that are accounted for when deriving an MSS from an MTU. +/// +/// An MTU is the largest *IP datagram* that may be sent over a link, hence the Ethernet header is deliberately *not* +/// part of this sum. We use the minimum (option-free) header sizes because that is what the inetstack actually emits +/// for data segments. Note that if the stack ever starts sending TCP options on data segments (e.g., the 12-byte +/// timestamp option from RFC 7323), then this sum must grow accordingly. See RFC 6691. +pub const MSS_HEADER_OVERHEAD: usize = layer3::ipv4::IPV4_HEADER_MIN_SIZE as usize + layer4::tcp::MIN_TCP_HEADER_SIZE; + +/// Default MTU Parameter for the inetstack. +/// +/// Only used when the underlying configuration does not specify an MTU. +pub const DEFAULT_MTU: usize = 1500; + +/// Fallback MSS Parameter for TCP. +/// +/// This is the value that a TCP implementation must assume about the other end when the peer did not advertise an MSS +/// option during the handshake. See RFC 793 and RFC 1122. pub const FALLBACK_MSS: usize = 536; /// Minimum MSS Parameter for TCP @@ -32,10 +48,32 @@ pub const TCP_ACK_DELAY_TIMEOUT: Duration = Duration::from_millis(500); /// Handshake timeout for tcp. pub const TCP_HANDSHAKE_TIMEOUT: Duration = Duration::from_secs(3); -/// Default MSS Parameter for TCP +/// Derives the default MSS to advertise from a given MTU, as prescribed by RFC 793 and RFC 6691. +/// +/// The result is clamped to [`MIN_MSS`]..=[`MAX_MSS`], so that degenerate MTUs can neither underflow nor produce a +/// value that does not fit in the 16-bit MSS option. +pub const fn default_mss_from_mtu(mtu: usize) -> usize { + let mss: usize = match mtu.checked_sub(MSS_HEADER_OVERHEAD) { + Some(mss) => mss, + None => 0, + }; + + if mss < MIN_MSS { + MIN_MSS + } else if mss > MAX_MSS { + MAX_MSS + } else { + mss + } +} + +/// Default MSS Parameter for TCP. +/// +/// This is the MSS that [`TcpConfig::default`] falls back to when there is no configuration at all to read an MTU +/// from. It is the MSS derived from [`DEFAULT_MTU`], so that both defaults agree with each other. /// -/// TODO: Auto-Discovery MTU Size -pub const DEFAULT_MSS: usize = 1450; +/// [`TcpConfig::default`]: crate::inetstack::config::TcpConfig::default +pub const DEFAULT_MSS: usize = default_mss_from_mtu(DEFAULT_MTU); /// Max batch size of packets for both transmit and receive up and down the stack. This is based on the /// DEMI_SGARRAY_MAXLEN and should always be bigger than that to receiving an entire sga worth of packets at once. diff --git a/src/inetstack/protocols/layer4/tcp/active_open.rs b/src/inetstack/protocols/layer4/tcp/active_open.rs index 317eb7ee0..e0c1f5f21 100644 --- a/src/inetstack/protocols/layer4/tcp/active_open.rs +++ b/src/inetstack/protocols/layer4/tcp/active_open.rs @@ -10,7 +10,7 @@ use crate::{ expect_some, inetstack::{ config::TcpConfig, - consts::{FALLBACK_MSS, MAX_HEADER_SIZE, MAX_WINDOW_SCALE}, + consts::{MAX_HEADER_SIZE, MAX_WINDOW_SCALE}, protocols::{ layer3::SharedLayer3Endpoint, layer4::tcp::{ @@ -140,7 +140,7 @@ impl SharedActiveOpenSocket { .transmit_tcp_packet_nonblocking(dst_ipv4_addr, pkt)?; let mut remote_window_scale_bits = None; - let mut mss = FALLBACK_MSS; + let mut received_mss: Option = None; for option in header.iter_options() { match option { TcpOptions2::WindowScale(w) => { @@ -149,11 +149,15 @@ impl SharedActiveOpenSocket { }, TcpOptions2::MaximumSegmentSize(m) => { info!("Received advertised MSS: {}", m); - mss = *m as usize; + received_mss = Some(*m as usize); }, _ => continue, } } + // The MSS advertised by the peer bounds the segments that we send, but it may be larger than what our own + // interface can carry, so clamp it to our advertised MSS (RFC 1122, Section 4.2.2.6). + let mss: usize = self.tcp_config.get_effective_mss(received_mss); + info!("Using MSS: {}", mss); let (local_window_scale_bits, remote_window_scale_bits): (u8, u8) = match remote_window_scale_bits { Some(remote_window_scale_bits) => { diff --git a/src/inetstack/protocols/layer4/tcp/passive_open.rs b/src/inetstack/protocols/layer4/tcp/passive_open.rs index 271390f02..ec01f36b1 100644 --- a/src/inetstack/protocols/layer4/tcp/passive_open.rs +++ b/src/inetstack/protocols/layer4/tcp/passive_open.rs @@ -13,7 +13,7 @@ use crate::{ expect_some, inetstack::{ config::TcpConfig, - consts::{FALLBACK_MSS, MAX_HEADER_SIZE, MAX_WINDOW_SCALE}, + consts::{MAX_HEADER_SIZE, MAX_WINDOW_SCALE}, protocols::{ layer3::SharedLayer3Endpoint, layer4::tcp::{ @@ -259,7 +259,7 @@ impl SharedPassiveSocket { ) { // Set up new inflight accept connection. let mut remote_window_scale = None; - let mut mss = FALLBACK_MSS; + let mut received_mss: Option = None; for option in tcp_hdr.iter_options() { match option { TcpOptions2::WindowScale(w) => { @@ -268,11 +268,15 @@ impl SharedPassiveSocket { }, TcpOptions2::MaximumSegmentSize(m) => { info!("Received advertised MSS: {}", m); - mss = *m as usize; + received_mss = Some(*m as usize); }, _ => continue, } } + // The MSS advertised by the peer bounds the segments that we send, but it may be larger than what our own + // interface can carry, so clamp it to our advertised MSS (RFC 1122, Section 4.2.2.6). + let mss: usize = self.tcp_config.get_effective_mss(received_mss); + info!("Using MSS: {}", mss); let mut handshake_retries: usize = self.tcp_config.get_handshake_retries(); let handshake_timeout: Duration = self.tcp_config.get_handshake_timeout();