From 38bc204013ac59664cd7ac0e452154c8ed41acfb Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Tue, 18 Aug 2026 20:20:46 -0600 Subject: [PATCH 01/21] fix(net): Validate the RFC 4884 original datagram field as the RFC says `check_full_payload` decides whether an ICMP error carries the whole of the packet that provoked it. Its two length checks were expressed in bits where the values are octets, and between them they enforced neither of the two requirements RFC 4884 section 3 states: the "original datagram" field MUST contain at least 128 octets, and it MUST be zero padded to the nearest 32-bit boundary (ICMPv4) or 64-bit boundary (ICMPv6). What the code did: if icmp_length > buf.len() || !icmp_length.is_multiple_of(32) { ... } if padding_length < 32 && ... `icmp_length` is `length_attribute * 4` for ICMPv4 and `* 8` for ICMPv6, because the attribute counts 32- and 64-bit words. So it is a multiple of 4 (or 8) octets by construction, and a correct alignment check cannot fail. `is_multiple_of(32)` instead demanded a multiple of 32 *octets* -- requiring the length attribute itself to be divisible by 8, which nothing asks for. Seven of every eight lengths a conforming sender can express were refused: field 120 octets (30 words) -> refused field 124 octets (31 words) -> refused field 128 octets (32 words) -> accepted field 132 octets (33 words) -> refused Meanwhile the 128-octet minimum was not checked at all, so a 32-octet field was accepted. The check that existed rejected valid messages and admitted invalid ones. `padding_length < 32` was wrong in the same way and in the other direction: when the original datagram is shorter than 128 octets the sender pads it up to 128, so legitimate padding is routinely more than 31 octets. So: check what the RFC requires -- at least 128 octets, no more than the buffer holds, and the region past the original datagram all zeroes. The alignment test goes, being vacuous, and the padding bound goes, being invented. Each surviving check carries the sentence it implements, in the citation format duvet reads, so the next reader can check the code against the specification without leaving the file. Severity is low today and latent. `is_full_payload()` and `payload_length()` have no callers anywhere in the workspace -- the flag is computed and never read -- so nothing forwards differently. It would have failed quietly, by declining to treat valid ICMP errors as complete, whenever the feature was switched on. The part worth remembering is how it stayed hidden. `cargo-mutants` left 22 comparison mutants alive in this file, which is what prompted a closer look -- but mutation testing did not find this. It found that nobody was looking. The bug is a deviation from a specification, and no amount of "the tests notice when this line changes" can see that. Worse, the existing test actively concealed it. It padded a 120-octet packet to 128 with the comment "we need to pad on a 32-bit word boundary" -- and 120 is already on a 32-bit boundary. The padding was there to satisfy the code, and the comment rationalised it. Closing those 22 mutants with a property asserting the behaviour as found would have made the deviation load-bearing, defended by a test, and cited by the next reader as deliberate. Chasing survivors can entrench a defect as easily as expose one, and only the specification can tell the difference. The comment on the old test is corrected here for the same reason. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- net/src/headers/embedded.rs | 106 ++++++++++++++++++++++++++++++------ 1 file changed, 88 insertions(+), 18 deletions(-) diff --git a/net/src/headers/embedded.rs b/net/src/headers/embedded.rs index e8397c56fc..5e3b1f2c99 100644 --- a/net/src/headers/embedded.rs +++ b/net/src/headers/embedded.rs @@ -48,6 +48,8 @@ pub struct EmbeddedHeaders { full_payload_length: Option, } +const MIN_ORIGINAL_DATAGRAM_OCTETS: usize = 128; + impl EmbeddedHeaders { #[cfg(any(test, feature = "bolero"))] #[must_use] @@ -241,16 +243,22 @@ impl EmbeddedHeaders { // The embedded message is shorter than the original packet return; } - if icmp_length > buf.len() || !icmp_length.is_multiple_of(32) { - // Embedded payload is larger than our buffer? Or the size is not a multiple - // of 32? Something's wrong + if icmp_length > buf.len() { + return; + } + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //# When the ICMP Extension Structure is appended to an ICMP message + //# and that ICMP message contains an "original datagram" field, the + //# "original datagram" field MUST contain at least 128 octets. + if icmp_length < MIN_ORIGINAL_DATAGRAM_OCTETS { return; } - let padding_length = icmp_length - full_packet_length; - // ICMPv4: Padding is on 32-bit boundaries - if padding_length < 32 - && buf[full_packet_length..icmp_length].iter().all(|b| *b == 0) - { + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //# When the ICMP Extension Structure is appended to an ICMPv4 message + //# and that ICMPv4 message contains an "original datagram" field, the + //# "original datagram" field MUST be zero padded to the nearest + //# 32-bit boundary. + if buf[full_packet_length..icmp_length].iter().all(|b| *b == 0) { self.full_payload_length = Some(transport_payload_length as u16); } return; @@ -260,16 +268,22 @@ impl EmbeddedHeaders { // The embedded message is shorter than the original packet return; } - if icmp_length > buf.len() || !icmp_length.is_multiple_of(64) { - // Embedded payload is larger than our buffer? Or the size is not a multiple - // of 64? Something's wrong + if icmp_length > buf.len() { return; } - let padding_length = icmp_length - full_packet_length; - // ICMPv6: Padding is on 64-bit boundaries - if padding_length < 64 - && buf[full_packet_length..icmp_length].iter().all(|b| *b == 0) - { + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //# When the ICMP Extension Structure is appended to an ICMP message + //# and that ICMP message contains an "original datagram" field, the + //# "original datagram" field MUST contain at least 128 octets. + if icmp_length < MIN_ORIGINAL_DATAGRAM_OCTETS { + return; + } + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //# When the ICMP Extension Structure is appended to an ICMPv6 message + //# and that ICMPv6 message contains an "original datagram" field, the + //# "original datagram" field MUST be zero padded to the nearest + //# 64-bit boundary. + if buf[full_packet_length..icmp_length].iter().all(|b| *b == 0) { self.full_payload_length = Some(transport_payload_length as u16); } return; @@ -1298,12 +1312,68 @@ mod tests { assert!(!headers.is_full_payload()); } + fn v4_with_field_of(field_len: usize, padding_byte: u8) -> (EmbeddedHeaders, usize, Vec) { + let mut buf = create_full_ipv4_tcp_packet_with_payload(); + assert_eq!(buf.len(), 120, "the embedded packet is 120 octets"); + buf.extend(std::iter::repeat_n(padding_byte, field_len - buf.len())); + assert_eq!(buf.len(), field_len); + buf.extend_from_slice(&[0x55u8; 32]); + let (headers, consumed) = + EmbeddedHeaders::parse_with(EmbeddedIpVersion::Ipv4, &buf).unwrap(); + (headers, consumed.get() as usize, buf) + } + + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //= type=test + //# When the ICMP Extension Structure is appended to an ICMPv4 message + //# and that ICMPv4 message contains an "original datagram" field, the + //# "original datagram" field MUST be zero padded to the nearest + //# 32-bit boundary. + #[test] + fn a_field_is_accepted_at_any_32_bit_aligned_length() { + for field_len in [128usize, 132, 136, 140, 144, 148, 152, 156] { + let (mut headers, consumed, buf) = v4_with_field_of(field_len, 0); + headers.check_full_payload(&buf, buf.len(), consumed, field_len); + assert!( + headers.is_full_payload(), + "a {field_len}-octet field is 32-bit aligned and at least 128 octets, so it must \ + be accepted" + ); + assert_eq!(headers.payload_length(), Some(80)); + } + } + + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //= type=test + //# When the ICMP Extension Structure is appended to an ICMP message + //# and that ICMP message contains an "original datagram" field, the + //# "original datagram" field MUST contain at least 128 octets. + #[test] + fn a_field_shorter_than_128_octets_is_refused() { + for field_len in [120usize, 124] { + let (mut headers, consumed, buf) = v4_with_field_of(field_len, 0); + headers.check_full_payload(&buf, buf.len(), consumed, field_len); + assert!( + !headers.is_full_payload(), + "a {field_len}-octet field is below the 128-octet minimum" + ); + } + } + + #[test] + fn a_field_padded_with_anything_but_zeroes_is_refused() { + let (mut headers, consumed, buf) = v4_with_field_of(136, 0xab); + headers.check_full_payload(&buf, buf.len(), consumed, 136); + assert!( + !headers.is_full_payload(), + "non-zero padding must not be accepted as padding" + ); + } + #[test] fn test_check_full_payload_with_icmp_extensions() { let mut buf = create_full_ipv4_tcp_packet_with_payload(); - // We need to pad on a 32-bit word boundary. We have 120 bytes (20 for the IP header, 20 for - // the TCP header, 80 for the payload), add 8 to reach 128 bytes. buf.extend_from_slice(&[0u8; 8]); let icmp_payload_length = buf.len(); From b9ceef8d424ea4f333d5a6c5151d64e4faa8ee7d Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Tue, 18 Aug 2026 21:04:34 -0600 Subject: [PATCH 02/21] build(duvet): Track RFC 4884 compliance, and vendor the specification `duvet report` matches citations in the source against the requirements it extracts from a specification, so a requirement with nothing implementing it, or an implementation with nothing testing it, becomes visible. RFC 4884 is the first specification tracked because the code already cites it: the "original datagram" length checks contradicted it, and the citations added with that fix are the ones this now reads. Everything but the reports is committed, on purpose. `duvet init` generates a `.gitignore` containing only `reports/`, and that turns out to be load-bearing rather than a style choice. Measured under `unshare -rn`: with `.duvet/specifications/` present, `duvet report` runs offline; with it removed, it fails with "Network is unreachable". There is no cache anywhere else -- nothing in `~/.cache`. So vendoring the specification text is what makes the report runnable in a build sandbox at all, rather than a nicety. RFC 4884 costs 42KiB. It also means an errata, a reformat, or a fetch that quietly returns something different arrives as a reviewable diff in a pull request rather than as a change in results nobody can account for. Refreshing a specification becomes a deliberate commit. The first run says: TEXT[!MUST,implementation,test]: ..."original datagram" field MUST contain at least 128 octets. TEXT[!MUST,implementation,test]: ...MUST be zero padded to the nearest 32-bit boundary. [v4] TEXT[!MUST,implementation]: ...MUST be zero padded to the nearest 64-bit boundary. [v6] The ICMPv6 requirement is implemented and cited but has no test: the three tests added with the fix are all ICMPv4. Two adjustments. The generated `[[source]]` pattern is `src/**/*.rs`, which matches nothing in a workspace; it is `*/src/**/*.rs` here. And `routing/src/cli/display.rs` had thirteen banner comments of the form `//======== Fib ========//`; `//=` is duvet's citation marker, so each was read as a malformed citation -- 24 errors before the report could run. The banners now start `// =`. The collision is worth knowing about before writing new ones. `.duvet/snapshot.txt` is line-oriented and diffs cleanly, and its unit is a sentence somebody else wrote. Unlike a mutant's `file:line:col`, that key does not move when a function is reformatted -- which is what makes it usable as a gate where the mutation report is only usable as a report. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/.gitignore | 1 + .duvet/config.toml | 13 + .../rfc/rfc4884/section-3.toml | 51 + .../rfc/rfc4884/section-4.6.toml | 17 + .../rfc/rfc4884/section-4.toml | 41 + .../rfc/rfc4884/section-5.4.toml | 10 + .../rfc/rfc4884/section-5.5.toml | 18 + .../rfc/rfc4884/section-7.toml | 18 + .duvet/snapshot.txt | 59 + .../www.rfc-editor.org/rfc/rfc4884.txt | 1067 +++++++++++++++++ routing/src/cli/display.rs | 13 - 11 files changed, 1295 insertions(+), 13 deletions(-) create mode 100644 .duvet/.gitignore create mode 100644 .duvet/config.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-3.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.6.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.4.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.5.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-7.toml create mode 100644 .duvet/snapshot.txt create mode 100644 .duvet/specifications/www.rfc-editor.org/rfc/rfc4884.txt diff --git a/.duvet/.gitignore b/.duvet/.gitignore new file mode 100644 index 0000000000..a9a1bd38ab --- /dev/null +++ b/.duvet/.gitignore @@ -0,0 +1 @@ +reports/ diff --git a/.duvet/config.toml b/.duvet/config.toml new file mode 100644 index 0000000000..464593eb4e --- /dev/null +++ b/.duvet/config.toml @@ -0,0 +1,13 @@ +'$schema' = "https://awslabs.github.io/duvet/config/v0.4.0.json" + +[[source]] +pattern = "*/src/**/*.rs" + +[[specification]] +source = "https://www.rfc-editor.org/rfc/rfc4884" + +[report.html] +enabled = true + +[report.snapshot] +enabled = true diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-3.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-3.toml new file mode 100644 index 0000000000..869bf3345e --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-3.toml @@ -0,0 +1,51 @@ +target = "https://www.rfc-editor.org/rfc/rfc4884#section-3" + + +[[spec]] +level = "MAY" +quote = ''' +An ICMP Extension Structure MAY be appended to ICMPv4 Destination +Unreachable, Time Exceeded, and Parameter Problem messages. +''' + +[[spec]] +level = "MAY" +quote = ''' +An ICMP Extension Structure MAY be appended to ICMPv6 Destination +Unreachable, and Time Exceeded messages. +''' + +[[spec]] +level = "MUST" +quote = ''' +When the ICMP Extension Structure is appended to an ICMP message +and that ICMP message contains an "original datagram" field, the +"original datagram" field MUST contain at least 128 octets. +''' + +[[spec]] +level = "MUST" +quote = ''' +When the ICMP Extension Structure is appended to an ICMPv4 message +and that ICMPv4 message contains an "original datagram" field, the +"original datagram" field MUST be zero padded to the nearest +32-bit boundary. +''' + +[[spec]] +level = "MUST" +quote = ''' +When the ICMP Extension Structure is appended to an ICMPv6 message +and that ICMPv6 message contains an "original datagram" field, the +"original datagram" field MUST be zero padded to the nearest +64-bit boundary. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +ICMP messages defined in the future SHOULD indicate whether or not +they support the extension mechanism defined in this +specification. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.6.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.6.toml new file mode 100644 index 0000000000..9e28215efe --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.6.toml @@ -0,0 +1,17 @@ +target = "https://www.rfc-editor.org/rfc/rfc4884#section-4.6" + + +[[spec]] +level = "MAY" +quote = ''' +The ICMP Extension Structure MAY be appended to messages of the +following types: +''' + +[[spec]] +level = "MUST" +quote = ''' +The ICMP Extension Structure MUST NOT be appended to any of the other +ICMP messages mentioned in Section 4. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.toml new file mode 100644 index 0000000000..cab9c95080 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-4.toml @@ -0,0 +1,41 @@ +target = "https://www.rfc-editor.org/rfc/rfc4884#section-4" + + +[[spec]] +level = "MUST" +quote = ''' +The length attribute MUST be specified when the ICMP Extension +Structure is appended to the above mentioned ICMP messages. +''' + +[[spec]] +level = "MUST" +quote = ''' +When the length attribute is specified, the "original datagram" field +MUST be zero padded to the nearest 32-bit boundary. +''' + +[[spec]] +level = "MUST" +quote = ''' +When the length attribute is specified, the "original datagram" field +MUST be zero padded to the nearest 64-bit boundary. +''' + +[[spec]] +level = "MUST" +quote = ''' +In order to achieve backwards compatibility, when the ICMP Extension +Structure is appended to an ICMP message and that ICMP message +contains an "original datagram" field, the "original datagram" field +MUST contain at least 128 octets. +''' + +[[spec]] +level = "MUST" +quote = ''' +If the original datagram did not +contain 128 octets, the "original datagram" field MUST be zero padded +to 128 octets. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.4.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.4.toml new file mode 100644 index 0000000000..a57c651ebd --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.4.toml @@ -0,0 +1,10 @@ +target = "https://www.rfc-editor.org/rfc/rfc4884#section-5.4" + + +[[spec]] +level = "MUST" +quote = ''' +If the length attribute is zero, the compliant application +MUST determine that the message contains no extensions. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.5.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.5.toml new file mode 100644 index 0000000000..ba24083420 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-5.5.toml @@ -0,0 +1,18 @@ +target = "https://www.rfc-editor.org/rfc/rfc4884#section-5.5" + + +[[spec]] +level = "MUST" +quote = ''' +If the length attribute is zero, the compliant application +MUST determine that the message contains no extensions. +''' + +[[spec]] +level = "MUST" +quote = ''' +So, to ease transition yet encourage compliant implementation, +compliant TRACEROUTE implementations MUST include a non-default +operation mode to also interpret non-compliant responses. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-7.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-7.toml new file mode 100644 index 0000000000..95f69fa423 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4884/section-7.toml @@ -0,0 +1,18 @@ +target = "https://www.rfc-editor.org/rfc/rfc4884#section-7" + + +[[spec]] +level = "MAY" +quote = ''' +Having received an ICMP message +with extensions, application software MAY process selected objects +while ignoring others. +''' + +[[spec]] +level = "MUST" +quote = ''' +As stated above, the total length of the ICMP message, including +extensions, MUST NOT exceed the minimum reassembly buffer size. +''' + diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt new file mode 100644 index 0000000000..3fe31d4b2c --- /dev/null +++ b/.duvet/snapshot.txt @@ -0,0 +1,59 @@ +SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4884 + SECTION: [Summary of Changes to ICMP](#section-3) + TEXT[!MAY]: An ICMP Extension Structure MAY be appended to ICMPv4 Destination + TEXT[!MAY]: Unreachable, Time Exceeded, and Parameter Problem messages. + TEXT[!MAY]: An ICMP Extension Structure MAY be appended to ICMPv6 Destination + TEXT[!MAY]: Unreachable, and Time Exceeded messages. + TEXT[!MUST,implementation,test]: When the ICMP Extension Structure is appended to an ICMP message + TEXT[!MUST,implementation,test]: and that ICMP message contains an "original datagram" field, the + TEXT[!MUST,implementation,test]: "original datagram" field MUST contain at least 128 octets. + TEXT[!MUST,implementation,test]: When the ICMP Extension Structure is appended to an ICMPv4 message + TEXT[!MUST,implementation,test]: and that ICMPv4 message contains an "original datagram" field, the + TEXT[!MUST,implementation,test]: "original datagram" field MUST be zero padded to the nearest + TEXT[!MUST,implementation,test]: 32-bit boundary. + TEXT[!MUST,implementation]: When the ICMP Extension Structure is appended to an ICMPv6 message + TEXT[!MUST,implementation]: and that ICMPv6 message contains an "original datagram" field, the + TEXT[!MUST,implementation]: "original datagram" field MUST be zero padded to the nearest + TEXT[!MUST,implementation]: 64-bit boundary. + TEXT[!SHOULD]: ICMP messages defined in the future SHOULD indicate whether or not + TEXT[!SHOULD]: they support the extension mechanism defined in this + TEXT[!SHOULD]: specification. + + SECTION: [ICMP Extensibility](#section-4) + TEXT[!MUST]: The length attribute MUST be specified when the ICMP Extension + TEXT[!MUST]: Structure is appended to the above mentioned ICMP messages. + TEXT[!MUST]: When the length attribute is specified, the "original datagram" field + TEXT[!MUST]: MUST be zero padded to the nearest 32-bit boundary. + TEXT[!MUST]: When the length attribute is specified, the "original datagram" field + TEXT[!MUST]: MUST be zero padded to the nearest 64-bit boundary. + TEXT[!MUST]: In order to achieve backwards compatibility, when the ICMP Extension + TEXT[!MUST]: Structure is appended to an ICMP message and that ICMP message + TEXT[!MUST]: contains an "original datagram" field, the "original datagram" field + TEXT[!MUST]: MUST contain at least 128 octets. + TEXT[!MUST]: If the original datagram did not + TEXT[!MUST]: contain 128 octets, the "original datagram" field MUST be zero padded + TEXT[!MUST]: to 128 octets. + + SECTION: [ICMP Messages That Can Be Extended](#section-4.6) + TEXT[!MAY]: The ICMP Extension Structure MAY be appended to messages of the + TEXT[!MAY]: following types: + TEXT[!MUST]: The ICMP Extension Structure MUST NOT be appended to any of the other + TEXT[!MUST]: ICMP messages mentioned in Section 4. + + SECTION: [Compliant Application Receives ICMP Message with No Extensions](#section-5.4) + TEXT[!MUST]: If the length attribute is zero, the compliant application + TEXT[!MUST]: MUST determine that the message contains no extensions. + + SECTION: [Compliant Application Receives ICMP Message with Non-Compliant](#section-5.5) + TEXT[!MUST]: If the length attribute is zero, the compliant application + TEXT[!MUST]: MUST determine that the message contains no extensions. + TEXT[!MUST]: So, to ease transition yet encourage compliant implementation, + TEXT[!MUST]: compliant TRACEROUTE implementations MUST include a non-default + TEXT[!MUST]: operation mode to also interpret non-compliant responses. + + SECTION: [The ICMP Extension Structure](#section-7) + TEXT[!MAY]: Having received an ICMP message + TEXT[!MAY]: with extensions, application software MAY process selected objects + TEXT[!MAY]: while ignoring others. + TEXT[!MUST]: As stated above, the total length of the ICMP message, including + TEXT[!MUST]: extensions, MUST NOT exceed the minimum reassembly buffer size. diff --git a/.duvet/specifications/www.rfc-editor.org/rfc/rfc4884.txt b/.duvet/specifications/www.rfc-editor.org/rfc/rfc4884.txt new file mode 100644 index 0000000000..85def5fce9 --- /dev/null +++ b/.duvet/specifications/www.rfc-editor.org/rfc/rfc4884.txt @@ -0,0 +1,1067 @@ + + + + + + +Network Working Group R. Bonica +Request for Comments: 4884 Juniper Networks +Updates: 792, 4443 D. Gan +Category: Standards Track Consultant + D. Tappan + Consultant + C. Pignataro + Cisco Systems, Inc. + April 2007 + + + Extended ICMP to Support Multi-Part Messages + +Status of This Memo + + This document specifies an Internet standards track protocol for the + Internet community, and requests discussion and suggestions for + improvements. Please refer to the current edition of the "Internet + Official Protocol Standards" (STD 1) for the standardization state + and status of this protocol. Distribution of this memo is unlimited. + +Copyright Notice + + Copyright (C) The IETF Trust (2007). + +Abstract + + This document redefines selected ICMP messages to support multi-part + operation. A multi-part ICMP message carries all of the information + that ICMP messages carried previously, as well as additional + information that applications may require. + + Multi-part messages are supported by an ICMP extension structure. + The extension structure is situated at the end of the ICMP message. + It includes an extension header followed by one or more extension + objects. Each extension object contains an object header and object + payload. All object headers share a common format. + + This document further redefines the above mentioned ICMP messages by + specifying a length attribute. All of the currently defined ICMP + messages to which an extension structure can be appended include an + "original datagram" field. The "original datagram" field contains + the initial octets of the datagram that elicited the ICMP error + message. Although the original datagram field is of variable length, + the ICMP message does not include a field that specifies its length. + Therefore, in order to facilitate message parsing, this document + allocates eight previously reserved bits to reflect the length of the + "original datagram" field. + + + +Bonica, et al. Standards Track [Page 1] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + The proposed modifications change the requirements for ICMP + compliance. The impact of these changes on compliant implementations + is discussed, and new requirements for future implementations are + presented. + + This memo updates RFC 792 and RFC 4443. + +Table of Contents + + 1. Introduction ....................................................3 + 2. Conventions Used in This Document ...............................4 + 3. Summary of Changes to ICMP ......................................4 + 4. ICMP Extensibility ..............................................4 + 4.1. ICMPv4 Destination Unreachable .............................7 + 4.2. ICMPv4 Time Exceeded .......................................8 + 4.3. ICMPv4 Parameter Problem ...................................8 + 4.4. ICMPv6 Destination Unreachable .............................9 + 4.5. ICMPv6 Time Exceeded .......................................9 + 4.6. ICMP Messages That Can Be Extended ........................10 + 5. Backwards Compatibility ........................................10 + 5.1. Classic Application Receives ICMP Message with + Extensions ................................................12 + 5.2. Non-Compliant Application Receives ICMP Message + with No Extensions ........................................12 + 5.3. Non-Compliant Application Receives ICMP Message + with Compliant Extensions .................................13 + 5.4. Compliant Application Receives ICMP Message with + No Extensions .............................................14 + 5.5. Compliant Application Receives ICMP Message with + Non-Compliant Extensions ..................................14 + 6. Interaction with Network Address Translation ...................14 + 7. The ICMP Extension Structure ...................................15 + 8. ICMP Extension Objects .........................................16 + 9. Security Considerations ........................................16 + 10. IANA Considerations ...........................................17 + 11. Acknowledgments ...............................................17 + 12. References ....................................................17 + 12.1. Normative References .....................................17 + 12.2. Informative References ...................................17 + + + + + + + + + + + + +Bonica, et al. Standards Track [Page 2] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +1. Introduction + + This document redefines selected ICMPv4 [RFC0792] and ICMPv6 + [RFC4443] messages to include an extension structure and a length + attribute. The extension structure supports multi-part ICMP + operation. Protocol designers can make an ICMP message carry + additional information by encoding that information in the extension + structure. + + This document also addresses a fundamental problem in ICMP + extensibility. All of the ICMP messages addressed by this memo + include an "original datagram" field. The "original datagram" field + contains the initial octets of the datagram that elicited the ICMP + error message. Although the "original datagram" field is of variable + length, the ICMP message does not include a field that specifies its + length. + + Application software infers the length of the "original datagram" + field from the total length of the ICMP message. If an extension + structure were appended to the message without adding a length + attribute for the "original datagram" field, the message would become + unparsable. Specifically, application software would not be able to + determine where the "original datagram" field ends and where the + extension structure begins. Therefore, this document proposes a + length attribute as well as an extension structure that is appended + to the ICMP message. + + The current memo also addresses backwards compatibility with existing + ICMP implementations that either do not implement the extensions + defined herein or implement them without adding the required length + attributes. In particular, this document addresses backwards + compatibility with certain, widely deployed, MPLS-aware ICMPv4 + implementations that send the extensions defined herein without + adding the required length attribute. + + The current memo does not define any ICMP extension objects. It + defines only the extension header and a common header that all + extension objects share. [UNNUMBERED], [ROUTING-INST], and + [MPLS-ICMP] provide sample applications of the ICMP Extension Object. + + The above mentioned memos share a common characteristic. They all + append information to the ICMP Time Expired message for consumption + by TRACEROUTE. In this case, as in many others, appending + information to the existing ICMP Time Expired Message is preferable + to defining a new message and emitting two messages whenever a packet + is dropped due to TTL expiration. + + + + + +Bonica, et al. Standards Track [Page 3] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +2. Conventions Used in This Document + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and "OPTIONAL" in this + document are to be interpreted as described in [RFC2119]. + +3. Summary of Changes to ICMP + + The following is a summary of changes to ICMP that are introduced by + this memo: + + An ICMP Extension Structure MAY be appended to ICMPv4 Destination + Unreachable, Time Exceeded, and Parameter Problem messages. + + An ICMP Extension Structure MAY be appended to ICMPv6 Destination + Unreachable, and Time Exceeded messages. + + The above mentioned messages include an "original datagram" field, + and the message formats are updated to specify a length attribute + for the "original datagram" field. + + When the ICMP Extension Structure is appended to an ICMP message + and that ICMP message contains an "original datagram" field, the + "original datagram" field MUST contain at least 128 octets. + + When the ICMP Extension Structure is appended to an ICMPv4 message + and that ICMPv4 message contains an "original datagram" field, the + "original datagram" field MUST be zero padded to the nearest + 32-bit boundary. + + When the ICMP Extension Structure is appended to an ICMPv6 message + and that ICMPv6 message contains an "original datagram" field, the + "original datagram" field MUST be zero padded to the nearest + 64-bit boundary. + + ICMP messages defined in the future SHOULD indicate whether or not + they support the extension mechanism defined in this + specification. It is recommended that all new messages support + extensions. + +4. ICMP Extensibility + + RFC 792 defines the following ICMPv4 message types: + + - Destination Unreachable + + - Time Exceeded + + + + +Bonica, et al. Standards Track [Page 4] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + - Parameter Problem + + - Source Quench + + - Redirect + + - Echo Request/Reply + + - Timestamp/Timestamp Reply + + - Information Request/Information Reply + + [RFC1191] reserves bits for the "Next-Hop MTU" field in the + Destination Unreachable message. + + RFC 4443 defines the following ICMPv6 message types: + + - Destination Unreachable + + - Packet Too Big + + - Time Exceeded + + - Parameter Problem + + - Echo Request/Reply + + Many ICMP messages are extensible as currently defined. Protocol + designers can extend ICMP messages by simply appending fields or data + structures to them. + + However, the following ICMP messages are not extensible as currently + defined: + + - ICMPv4 Destination Unreachable (type = 3) + + - ICMPv4 Time Exceeded (type = 11) + + - ICMPv4 Parameter Problem (type = 12) + + - ICMPv6 Destination Unreachable (type = 1) + + - ICMPv6 Packet Too Big (type = 2) + + - ICMPv6 Time Exceeded (type = 3) + + - ICMPv6 Parameter Problem (type = 4) + + + + +Bonica, et al. Standards Track [Page 5] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + These messages contain an "original datagram" field which represents + the leading octets of the datagram to which the ICMP message is a + response. RFC 792 defines the "original datagram" field for ICMPv4 + messages. In RFC 792, the "original datagram" field includes the IP + header plus the next eight octets of the original datagram. + [RFC1812] extends the "original datagram" field to contain as many + octets as possible without causing the ICMP message to exceed the + minimum IPv4 reassembly buffer size (i.e., 576 octets). RFC 4443 + defines the "original datagram" field for ICMPv6 messages. In RFC + 4443, the "original datagram" field always contained as many octets + as possible without causing the ICMP message to exceed the minimum + IPv6 MTU (i.e., 1280 octets). + + Unfortunately, the "original datagram" field lacks a length + attribute. Application software infers the length of this field from + the total length of the ICMP message. If an extension structure were + appended to the message without adding a length attribute for the + "original datagram" field, the message would become unparsable. + Specifically, application software would not be able to determine + where the "original datagram" field ends and where the extension + structure begins. + + In order to solve this problem, this memo introduces an 8-bit length + attribute to the following ICMPv4 messages. + + - Destination Unreachable (type = 3) + + - Time Exceeded (type = 11) + + - Parameter Problem (type = 12) + + It also introduces an 8-bit length attribute to the following ICMPv6 + messages. + + - Destination Unreachable (type = 1) + + - Time Exceeded (type = 3) + + The length attribute MUST be specified when the ICMP Extension + Structure is appended to the above mentioned ICMP messages. + + The length attribute represents the length of the "original datagram" + field. Space for the length attribute is claimed from reserved + octets, whose value was previously required to be zero. + + For ICMPv4 messages, the length attribute represents 32-bit words. + When the length attribute is specified, the "original datagram" field + MUST be zero padded to the nearest 32-bit boundary. Because the + + + +Bonica, et al. Standards Track [Page 6] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + sixth octet of each of the impacted ICMPv4 messages was reserved for + future use, this octet was selected as the location of the length + attribute in ICMPv4. + + For ICMPv6 messages, the length attribute represents 64-bit words. + When the length attribute is specified, the "original datagram" field + MUST be zero padded to the nearest 64-bit boundary. Because the + fifth octet of each of the impacted ICMPv6 messages was reserved for + future use, this octet was selected as the location of the length + attribute in ICMPv6. + + In order to achieve backwards compatibility, when the ICMP Extension + Structure is appended to an ICMP message and that ICMP message + contains an "original datagram" field, the "original datagram" field + MUST contain at least 128 octets. If the original datagram did not + contain 128 octets, the "original datagram" field MUST be zero padded + to 128 octets. (See Section 5.1 for rationale.) + + The following sub-sections depict length attribute as it has been + introduced to selected ICMP messages. + +4.1. ICMPv4 Destination Unreachable + + Figure 1 depicts the ICMPv4 Destination Unreachable Message. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Type | Code | Checksum | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | unused | Length | Next-Hop MTU* | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Internet Header + leading octets of original datagram | + | | + | // | + | | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + + Figure 1: ICMPv4 Destination Unreachable + + The syntax and semantics of all fields are unchanged from RFC 792. + However, a length attribute is added to the second word. The length + attribute represents length of the padded "original datagram" field, + measured in 32-bit words. + + * The Next-Hop MTU field is not required in all cases. It is + depicted only to demonstrate that those bits are not available for + assignment in this memo. + + + +Bonica, et al. Standards Track [Page 7] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +4.2. ICMPv4 Time Exceeded + + Figure 2 depicts the ICMPv4 Time Exceeded Message. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Type | Code | Checksum | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | unused | Length | unused | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Internet Header + leading octets of original datagram | + | | + | // | + | | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + + Figure 2: ICMPv4 Time Exceeded + + The syntax and semantics of all fields are unchanged from RFC 792, + except for a length attribute which is added to the second word. The + length attribute represents length of the padded "original datagram" + field, measured in 32-bit words. + +4.3. ICMPv4 Parameter Problem + + Figure 3 depicts the ICMPv4 Parameter Problem Message. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Type | Code | Checksum | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Pointer | Length | unused | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Internet Header + leading octets of original datagram | + | | + | // | + | | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + + Figure 3: ICMPv4 Parameter Problem + + The syntax and semantics of all fields are unchanged from RFC 792, + except for a length attribute which is added to the second word. The + length attribute represents length of the padded "original datagram" + field, measured in 32-bit words. + + + + +Bonica, et al. Standards Track [Page 8] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +4.4. ICMPv6 Destination Unreachable + + Figure 4 depicts the ICMPv6 Destination Unreachable Message. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Type | Code | Checksum | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Length | Unused | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | As much of invoking packet | + + as possible without the ICMPv6 packet + + | exceeding the minimum IPv6 MTU [RFC4443] | + + Figure 4: ICMPv6 Destination Unreachable + + The syntax and semantics of all fields are unchanged from RFC 4443. + However, a length attribute is added to the second word. The length + attribute represents length of the padded "original datagram" field, + measured in 64-bit words. + +4.5. ICMPv6 Time Exceeded + + Figure 5 depicts the ICMPv6 Time Exceeded Message. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Type | Code | Checksum | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Length | Unused | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | As much of invoking packet | + + as possible without the ICMPv6 packet + + | exceeding the minimum IPv6 MTU [RFC4443] | + + Figure 5: ICMPv6 Time Exceeded + + The syntax and semantics of all fields are unchanged from RFC 4443, + except for a length attribute which is added to the second word. The + length attribute represents length of the padded "original datagram" + field, measured in 64-bit words. + + + + + + + + +Bonica, et al. Standards Track [Page 9] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +4.6. ICMP Messages That Can Be Extended + + The ICMP Extension Structure MAY be appended to messages of the + following types: + + - ICMPv4 Destination Unreachable + + - ICMPv4 Time Exceeded + + - ICMPv4 Parameter Problem + + - ICMPv6 Destination Unreachable + + - ICMPv6 Time Exceeded + + The ICMP Extension Structure MUST NOT be appended to any of the other + ICMP messages mentioned in Section 4. Extensions were not defined + for the ICMPv6 "Packet Too Big" and "Parameter Problem" messages + because these messages lack space for a length attribute. + +5. Backwards Compatibility + + ICMP messages can be categorized as follows: + + - Messages that do not include any ICMP extensions + + - Messages that include non-compliant ICMP extensions + + - Messages that includes compliant ICMP extensions + + Any ICMP implementation can send a message that does not include + extensions. ICMP implementations produced prior to 1999 are not + known to send ICMP extensions. + + Some ICMP implementations, produced between 1999 and the time of this + publication, may send a non-compliant version of ICMP extensions + described in this memo. Specifically, these implementations may + append the ICMP Extension Structure to the Time Exceeded and + Destination Unreachable messages. When they do this, they send + exactly 128 octets representing the original datagram, zero padding + if required. They also calculate checksums as described in this + document. However, they do not specify a length attribute to be + associated with the "original datagram" field. + + It is assumed that ICMP implementations produced in the future will + send ICMP extensions that are compliant with this specification. + + + + + +Bonica, et al. Standards Track [Page 10] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + Likewise, applications that consume ICMP messages can be categorized + as follows: + + - Classic applications + + - Non-compliant applications + + - Compliant applications + + Classic applications do not parse extensions defined in this memo. + They are insensitive to the length attribute that is associated with + the "original datagram" field. + + Non-compliant implementations parse the extensions defined in this + memo, but only in conjunction with the Time Expired and Destination + Unreachable messages. They require the "original datagram" field to + contain exactly 128 octets and are insensitive to the length + attribute that is associated with the "original datagram" field. + Non-compliant applications were produced between 1999 and the time of + publication of this memo. + + Compliant applications comply fully with the specifications of this + document. + + In order to demonstrate backwards compatibility, Table 1 describes + how members of each application category would parse each category of + ICMP message. + + +----------------+----------------+----------------+----------------+ + | | No Extensions | Non-compliant | Compliant | + | | | Extensions | Extensions | + +----------------+----------------+----------------+----------------+ + | Classic | - | Section 5.1 | Section 5.1 | + | Application | | | | + | | | | | + | Non-compliant | Section 5.2 | - | Section 5.3 | + | Application | | | | + | | | | | + | Compliant | Section 5.4 | Section 5.5 | - | + | Application | | | | + +----------------+----------------+----------------+----------------+ + + Table 1 + + In the table above, cells that contain a dash represent the nominal + case and require no explanation. In the following sections, we + assume that the ICMP message type is "Time Exceeded". + + + + +Bonica, et al. Standards Track [Page 11] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +5.1. Classic Application Receives ICMP Message with Extensions + + When a classic application receives an ICMP message that includes + extensions, it will incorrectly interpret those extensions as being + part of the "original datagram" field. Fortunately, the extensions + are guaranteed to begin at least 128 octets beyond the beginning of + the "original datagram" field. So, only those ICMP applications that + process the 129th octet of the "original datagram" field will be + adversely effected. To date, only two applications falling into this + category have been identified, and the degree to which they are + effected is minimal. + + Some TCP stacks, when they receive an ICMP message, verify the + checksum in the original datagram field [ATTACKS]. If the checksum + is incorrect, the TCP stack discards the ICMP message for security + reasons. If the trailing octets of the original datagram field are + overwritten by ICMP extensions, the TCP stack will discard an ICMP + message that it would not otherwise have discarded. The impact of + this issue is considered to be minimal because many ICMP messages are + discarded for other reasons (e.g., ICMP filtering, network + congestion, checksum was incorrect because original datagram field + was truncated.) + + Another theoretically possible, but highly improbably scenario occurs + when ICMP extensions overwrite the portion of the original datagram + field that represents the TCP header, causing the TCP stack to + operate upon the wrong TCP connection. This scenario is highly + unlikely because it occurs only when the TCP header appears at or + beyond the 128th octet of the original datagram field and then only + when the extensions approximate a valid TCP header. + +5.2. Non-Compliant Application Receives ICMP Message with No Extensions + + When a non-compliant ICMPv4 application receives a message that + contains no extensions, the application examines the total length of + the ICMPv4 message. If the total ICMPv4 message length is less than + the length of its IP header plus 144 octets, the application + correctly determines that the message does not contain any + extensions. + + The 144-octet sum is derived from 8 octets for the first two words of + the ICMPv4 Time Exceeded message, 128 octets for the "original + datagram" field, 4 octets for the ICMP Extension Header, and 4 octets + for a single ICMP Object header. All of these octets would be + required if extensions were present. + + + + + + +Bonica, et al. Standards Track [Page 12] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + If the ICMPv4 payload contains 144 octets or more, the application + must examine the 137th octet to determine whether it represents a + valid ICMPv4 Extension Header. In order to represent a valid + Extension Header, it must contain a valid version number and + checksum. If it does not contain a valid version number and + checksum, the application correctly determines that the message does + not contain any extensions. + + Non-compliant applications assume that the ICMPv4 Extension Structure + begins on the 137th octet of the Time Exceeded message, after a + 128-octet field representing the padded "original datagram" message. + + It is possible that a non-compliant application will parse an ICMPv4 + message incorrectly under the following conditions: + + - the message does not contain extensions + + - the original datagram field contains 144 octets or more + + - selected octets of the original datagram field represent the + correct values for an extension header version number and + checksum + + Although this is possible, it is very unlikely. + + A similar analysis can be performed for ICMPv6. However, the numeric + constants would change as appropriate. + +5.3. Non-Compliant Application Receives ICMP Message with Compliant + Extensions + + When a non-compliant application receives a message that contains + compliant ICMP extensions, it will parse those extensions correctly + only if the "original datagram" field contains exactly 128 octets. + This is because non-compliant applications are insensitive to the + length attribute that is associated with the "original datagram" + field. (They assume its value to be 128.) + + Provided that the entire ICMP message does not exceed the minimum + reassembly buffer size (576 octets for ICMPv4 or 1280 octets for + ICMPv6), there is no upper limit upon the length of the "original + datagram" field. However, each implementation will decide how many + octets to include. Those wishing to be backward compatible with non- + compliant TRACEROUTE implementations will include exactly 128 octets. + Those not requiring compatibility with non-compliant TRACEROUTE + applications may include more octets. + + + + + +Bonica, et al. Standards Track [Page 13] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +5.4. Compliant Application Receives ICMP Message with No Extensions + + When a compliant application receives an ICMP message, it examines + the length attribute that is associated with the "original datagram" + field. If the length attribute is zero, the compliant application + MUST determine that the message contains no extensions. + +5.5. Compliant Application Receives ICMP Message with Non-Compliant + Extensions + + When a compliant application receives an ICMP message, it examines + the length attribute that is associated with the "original datagram" + field. If the length attribute is zero, the compliant application + MUST determine that the message contains no extensions. In this + case, that determination is technically correct, but not backwards + compatible with the non-compliant implementation that originated the + ICMP message. + + So, to ease transition yet encourage compliant implementation, + compliant TRACEROUTE implementations MUST include a non-default + operation mode to also interpret non-compliant responses. + Specifically, when a TRACEROUTE application operating in non- + compliant mode receives a sufficiently long ICMP message that does + not specify a length attribute, it will parse for a valid extension + header at a fixed location, assuming a 128-octet "original datagram" + field. If the application detects a valid version and checksum, it + will treat the octets that follow as an extension structure. + +6. Interaction with Network Address Translation + + The ICMP extensions defined in this memo do not interfere with + Network Address Translation. [RFC3022] permits traditional NAT + devices to modify selected fields within ICMP messages. These fields + include the "original datagram" field mentioned above. However, if a + NAT device modifies the "original datagram" field, it should modify + only the leading octets of that field, which represent the outermost + IP header. Because the outermost IP header is guaranteed to be + contained by the first 128 octets of the "original datagram" field, + ICMP extensions and NAT will not interfere with one another. + + It is conceivable that a NAT implementation might overstep the + restrictions of RFC 3022 and overwrite the length attribute specified + by this memo. If a NAT implementation were to overwrite the length + attribute with zeros, the resulting packet will be indistinguishable + from a packet that was generated by a non-compliant ICMP + implementation. See Section 5.5 for packet details and a discussion + of backwards compatibility. + + + + +Bonica, et al. Standards Track [Page 14] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +7. The ICMP Extension Structure + + This memo proposes an optional ICMP Extension Structure that can be + appended to the ICMP messages referenced in Section 4.6 of this + document. + + The Extension Structure contains exactly one Extension Header + followed by one or more objects. Having received an ICMP message + with extensions, application software MAY process selected objects + while ignoring others. The presence of an unrecognized object does + not imply that an ICMP message is malformed. + + As stated above, the total length of the ICMP message, including + extensions, MUST NOT exceed the minimum reassembly buffer size. + Figure 6 depicts the ICMP Extension Header. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + |Version| (Reserved) | Checksum | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + + Figure 6: ICMP Extension Header + + The fields of the ICMP Extension Header are as follows: + + Version: 4 bits + + ICMP extension version number. This is version 2. + + Reserved: 12 bits + + Must be set to 0. + + Checksum: 16 bits + + The one's complement of the one's complement sum of the data + structure, with the checksum field replaced by zero for the + purpose of computing the checksum. An all-zero value means that + no checksum was transmitted. See Section 5.2 for a description of + how this field is used. + + + + + + + + + + +Bonica, et al. Standards Track [Page 15] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +8. ICMP Extension Objects + + Each extension object contains one or more 32-bit words, representing + an object header and payload. All object headers share a common + format. Figure 7 depicts the object header and payload. + + 0 1 2 3 + 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 2 3 4 5 6 7 8 9 0 1 + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | Length | Class-Num | C-Type | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + | | + | // (Object payload) // | + | | + +-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+-+ + + Figure 7: Object Header and Payload + + An object header has the following fields: + + Length: 16 bits + + Length of the object, measured in octets, including the object + header and object payload. + + Class-Num: 8 bits + + Identifies object class. + + C-Type: 8 bits + + Identifies object sub-type. + +9. Security Considerations + + Upon receipt of an ICMP message, application software must check it + for syntactic correctness. The extension checksum must be verified. + Improperly specified length attributes and other syntax problems may + result in buffer overruns. + + This memo does not define the conditions under which a router sends + an ICMP message. Therefore, it does not expose routers to any new + denial-of-service attacks. Routers may need to limit the rate at + which ICMP messages are sent. + + + + + + + +Bonica, et al. Standards Track [Page 16] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +10. IANA Considerations + + The ICMP Extension Object header contains two 8-bit fields: The + Class-Num identifies the object class, and the C-Type identifies the + class sub-type. Sub-type values are defined relative to a specific + object class value, and are defined per class. + + IANA has established a registry of ICMP extension objects classes and + class sub-types. There are no values assigned within this document + to maintain. Object classes 0xF7 - 0xFF are reserved for private + use. Object class values are assignable on a first-come-first-serve + basis. The policy for assigning sub-type values should be defined in + the document defining new class values. + +11. Acknowledgments + + Thanks to Pekka Nikander, Mark Doll, Fernando Gont, Joe Touch, + Christian Voiqt, and Sharon Chrisholm for their comments regarding + this document. + +12. References + +12.1. Normative References + + [RFC0792] Postel, J., "Internet Control Message Protocol", STD + 5, RFC 792, September 1981. + + [RFC1191] Mogul, J. and S. Deering, "Path MTU discovery", RFC + 1191, November 1990. + + [RFC1812] Baker, F., "Requirements for IP Version 4 Routers", + RFC 1812, June 1995. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, March 1997. + + [RFC4443] Conta, A., Deering, S., and M. Gupta, Ed., "Internet + Control Message Protocol (ICMPv6) for the Internet + Protocol Version 6 (IPv6) Specification", RFC 4443, + March 2006. + +12.2. Informative References + + [UNNUMBERED] Atlas, A., Bonica, R., Rivers, JR., Shen, N., and E. + Chen, "ICMP Extensions for Unnumbered Interfaces", + Work in Progress, March 2007. + + + + + +Bonica, et al. Standards Track [Page 17] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + + [MPLS-ICMP] Bonica, R., Gan, D., Tappan, D., and C. Pignataro, + "ICMP Extensions for MultiProtocol Label Switching", + Work in Progress, January 2007. + + [ATTACKS] Gont, F., "ICMP attacks against TCP", Work in + Progress, October 2006. + + [ROUTING-INST] Shen, N. and E. Chen, "ICMP Extensions for Routing + Instances", Work in Progress, November 2006. + + [RFC3022] Srisuresh, P. and K. Egevang, "Traditional IP Network + Address Translator (Traditional NAT)", RFC 3022, + January 2001. + +Authors' Addresses + + Ronald P. Bonica + Juniper Networks + 2251 Corporate Park Drive + Herndon, VA 20171 + US + + EMail: rbonica@juniper.net + + + Der-Hwa Gan + Consultant + + EMail: derhwagan@yahoo.com + + + Daniel C. Tappan + Consultant + + EMail: Dan.Tappan@gmail.com + + + Carlos Pignataro + Cisco Systems, Inc. + 7025 Kit Creek Road + Research Triangle Park, NC 27709 + US + + EMail: cpignata@cisco.com + + + + + + + +Bonica, et al. Standards Track [Page 18] + +RFC 4884 Multi-Part ICMP Messages April 2007 + + +Full Copyright Statement + + Copyright (C) The IETF Trust (2007). + + This document is subject to the rights, licenses and restrictions + contained in BCP 78, and except as set forth therein, the authors + retain all their rights. + + This document and the information contained herein are provided on an + "AS IS" basis and THE CONTRIBUTOR, THE ORGANIZATION HE/SHE REPRESENTS + OR IS SPONSORED BY (IF ANY), THE INTERNET SOCIETY, THE IETF TRUST AND + THE INTERNET ENGINEERING TASK FORCE DISCLAIM ALL WARRANTIES, EXPRESS + OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTY THAT THE USE OF + THE INFORMATION HEREIN WILL NOT INFRINGE ANY RIGHTS OR ANY IMPLIED + WARRANTIES OF MERCHANTABILITY OR FITNESS FOR A PARTICULAR PURPOSE. + +Intellectual Property + + The IETF takes no position regarding the validity or scope of any + Intellectual Property Rights or other rights that might be claimed to + pertain to the implementation or use of the technology described in + this document or the extent to which any license under such rights + might or might not be available; nor does it represent that it has + made any independent effort to identify any such rights. Information + on the procedures with respect to rights in RFC documents can be + found in BCP 78 and BCP 79. + + Copies of IPR disclosures made to the IETF Secretariat and any + assurances of licenses to be made available, or the result of an + attempt made to obtain a general license or permission for the use of + such proprietary rights by implementers or users of this + specification can be obtained from the IETF on-line IPR repository at + http://www.ietf.org/ipr. + + The IETF invites any interested party to bring to its attention any + copyrights, patents or patent applications, or other proprietary + rights that may cover technology that may be required to implement + this standard. Please address the information to the IETF at + ietf-ipr@ietf.org. + +Acknowledgement + + Funding for the RFC Editor function is currently provided by the + Internet Society. + + + + + + + +Bonica, et al. Standards Track [Page 19] + diff --git a/routing/src/cli/display.rs b/routing/src/cli/display.rs index 1c853b827e..055acc00fb 100644 --- a/routing/src/cli/display.rs +++ b/routing/src/cli/display.rs @@ -46,7 +46,6 @@ use std::time::Duration; use tracing::{error, warn}; -//================================= Common ==========================// fn fmt_opt_value( f: &mut std::fmt::Formatter<'_>, name: &str, @@ -60,7 +59,6 @@ fn fmt_opt_value( if nl { writeln!(f) } else { Ok(()) } } -//========================= Encapsulations ==========================// impl Display for VxlanEncapsulation { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { write!( @@ -83,7 +81,6 @@ impl Display for Encapsulation { } } -//=================== VRFs, routes and next-hops ====================// impl Display for RouteOrigin { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { match self { @@ -491,8 +488,6 @@ impl Display for VrfTable { } } -//========================= Interfaces ================================// - impl Display for Attachment { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { match self { @@ -594,7 +589,6 @@ impl Display for IfTable { Ok(()) } } -//========================= Interface addresses ================================// #[repr(transparent)] pub struct IfTableAddress<'a>(pub &'a IfTable); @@ -632,7 +626,6 @@ impl Display for IfTableAddress<'_> { } } -//========================= Rmac Store ================================// macro_rules! RMAC_TBL_FMT { () => { " {:<5} {:<20} {:<18} {:<8}" @@ -679,7 +672,6 @@ impl Display for RmacStore { } } -//========================= Rmac Store ================================// impl Display for Vtep { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { Heading("Local VTEP configuration").fmt(f)?; @@ -688,7 +680,6 @@ impl Display for Vtep { } } -//========================= Adjacencies ================================// macro_rules! ADJ_TBL_FMT { () => { " {:<10} {:<20} {:<18}" @@ -727,7 +718,6 @@ impl Display for AdjacencyTable { } } -//========================= Fib ================================// impl Display for FibKey { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> Result<(), std::fmt::Error> { match self { @@ -922,7 +912,6 @@ impl Display for FibGroups<'_> { } } -//========================= Time utils =========================// use chrono::Local; pub(crate) fn fmt_time(time: &DateTime) -> String { //let fmt_iso8 = "%Y-%m-%dT%H:%M:%S%.3f%:z"; @@ -954,7 +943,6 @@ pub(crate) fn fmt_time(time: &DateTime) -> String { out } -//========================= CPI ================================// macro_rules! STATS_ROW_FMT { () => { " {:<16} {:<12} {:<12} {:<12} {:<12} {:<12}" @@ -1038,7 +1026,6 @@ impl Display for CpiStats { } } -//========================= Frrmi ================================// impl Display for FrrmiStats { fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { let last_conn_time = &self From 865b61d5ef8157d2f1e9f6e41e55e990cec5cb18 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Tue, 18 Aug 2026 21:16:05 -0600 Subject: [PATCH 03/21] build(duvet): Track RFC 5382, and record where masquerade departs from it RFC 5382 states ten numbered requirements for how a NAT must treat TCP. It constrains values this codebase already has and chose without reference to it. Four are now cited; the other six are visible in the report as uncovered, which is the point of tracking it. Conformant, with a test: * REQ-7, no port overloading. The allocator's bitmaps enforce it and the exclusivity property in `masquerade::fuzz` asserts it -- "two live flows never share a translation" was written before anyone read RFC 5382, and turns out to be exactly REQ-7. * REQ-10, an ICMP message must not terminate the mapping. Held by construction: no arm of `next_flow_status_icmp` yields `Closed` or `Reset`, and the test that says so was written for other reasons. Both are cited rather than changed. The value is that they are now *claims* somebody can check, instead of accidents. Marked `todo`, because they are decisions rather than defects: * REQ-5, idle timeouts. "Established connection idle-timeout" MUST NOT be under 2 hours 4 minutes; "transitory" MUST NOT be under 4 minutes. Ours are 5, 3 and 2 seconds for the transitory states, and the established timeout is `idle_timeout` from the masquerade config, which defaults to two minutes and has no lower bound and no validation, so a deployment can set it anywhere. The short values look deliberate -- `protocol.rs` says the statuses exist "to know how much to extend the lifetime of flows for port conservation", and a gateway holding a public port for two hours per idle connection conserves nothing. But whether we are willing to state that as a deviation from a BCP is a product decision. * REQ-1, endpoint-independent mapping. Not held as stated: the allocation depends on `dst_vpcd`, so one internal endpoint reaching two destination VPCs can be given two public tuples. RFC 5382 assumes a NAT facing a single external realm; here destination VPCs are distinct address spaces behind distinct peerings, and sharing a pool across them would be the surprising choice. So this is probably an exception rather than a defect -- and "probably" is why it is `todo`. Answering it is cheap now; discovering it mattered after a peer-to-peer application fails is not. `todo` rather than `exception` in both cases deliberately. duvet has both, and an exception asserts a decision was taken. Neither of these was. Converting them needs a rationale somebody is willing to sign, and the report is where that queue lives. REQ-2, handling the TCP simultaneous-open, was traced through the state machine and works: the flow sits in `OneWay` while the two SYNs cross, then the peer's SYN-ACK moves it to `TwoWay` and the ACK to `Established`. Not cited, because sitting in `OneWay` for two extra round trips interacts with the REQ-5 timeouts above, and citing it as conformant would overstate what was verified. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/config.toml | 3 + .../rfc/rfc5382/section-4.1.toml | 10 + .../rfc/rfc5382/section-4.2.toml | 20 + .../rfc/rfc5382/section-4.3.toml | 64 + .../rfc/rfc5382/section-5.toml | 46 + .../rfc/rfc5382/section-6.toml | 11 + .../rfc/rfc5382/section-7.1.toml | 10 + .../rfc/rfc5382/section-7.2.toml | 16 + .../rfc/rfc5382/section-7.3.toml | 17 + .../rfc/rfc5382/section-8.toml | 158 +++ .duvet/snapshot.txt | 124 ++ .../www.rfc-editor.org/rfc/rfc5382.txt | 1179 +++++++++++++++++ nat/src/masquerade/apalloc/mod.rs | 7 + nat/src/masquerade/fuzz.rs | 4 + nat/src/masquerade/nf.rs | 8 + nat/src/masquerade/protocol.rs | 3 + nat/src/masquerade/state_machine.rs | 4 + 17 files changed, 1684 insertions(+) create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.1.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.2.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.3.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-5.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-6.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.1.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.2.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.3.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-8.toml create mode 100644 .duvet/specifications/www.rfc-editor.org/rfc/rfc5382.txt diff --git a/.duvet/config.toml b/.duvet/config.toml index 464593eb4e..0aa726f4d7 100644 --- a/.duvet/config.toml +++ b/.duvet/config.toml @@ -6,6 +6,9 @@ pattern = "*/src/**/*.rs" [[specification]] source = "https://www.rfc-editor.org/rfc/rfc4884" +[[specification]] +source = "https://www.rfc-editor.org/rfc/rfc5382" + [report.html] enabled = true diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.1.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.1.toml new file mode 100644 index 0000000000..0c08bbaebc --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.1.toml @@ -0,0 +1,10 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-4.1" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior +for TCP. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.2.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.2.toml new file mode 100644 index 0000000000..020c6ad5fc --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.2.toml @@ -0,0 +1,20 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-4.2" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-2: A NAT MUST support all valid sequences of TCP packets +(defined in [RFC0793]) for connections initiated both internally +as well as externally when the connection is permitted by the NAT. +''' + +[[spec]] +level = "MUST" +quote = ''' +In particular: +a) In addition to handling the TCP 3-way handshake mode of +connection initiation, A NAT MUST handle the TCP simultaneous- +open mode of connection initiation. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.3.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.3.toml new file mode 100644 index 0000000000..026ef9e8b6 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-4.3.toml @@ -0,0 +1,64 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-4.3" + + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-3: If application transparency is most important, it is +RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" +behavior for TCP. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +If a more stringent filtering behavior is most +important, it is RECOMMENDED that a NAT have an "Address-Dependent +Filtering" behavior. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The filtering behavior MAY be an option configurable by the +administrator of the NAT. +''' + +[[spec]] +level = "MAY" +quote = ''' +b) The filtering behavior for TCP MAY be independent of the +filtering behavior for UDP. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-4: A NAT MUST NOT respond to an unsolicited inbound SYN packet +for at least 6 seconds after the packet is received. +''' + +[[spec]] +level = "MUST" +quote = ''' +If during +this interval the NAT receives and translates an outbound SYN for +the connection the NAT MUST silently drop the original unsolicited +inbound SYN packet. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +Otherwise, the NAT SHOULD send an ICMP Port +Unreachable error (Type 3, Code 3) for the original SYN, unless +REQ-4a applies. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) The NAT MUST silently drop the original SYN packet if sending a +response violates the security policy of the NAT. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-5.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-5.toml new file mode 100644 index 0000000000..86fb5a2990 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-5.toml @@ -0,0 +1,46 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-5" + + +[[spec]] +level = "MAY" +quote = ''' +REQ-5: If a NAT cannot determine whether the endpoints of a TCP +connection are active, it MAY abandon the session if it has been +idle for some time. +''' + +[[spec]] +level = "MUST" +quote = ''' +In such cases, the value of the "established +connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. +''' + +[[spec]] +level = "MUST" +quote = ''' +The value of the "transitory connection idle-timeout" MUST NOT be +less than 4 minutes. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The value of the NAT idle-timeouts MAY be configurable. +''' + +[[spec]] +level = "MAY" +quote = ''' +A NAT MAY hold state for a connection in +TIME_WAIT state to accommodate retransmissions of the last ACK. +''' + +[[spec]] +level = "MAY" +quote = ''' +When a NAT abandons a live connection, for +example due to a timeout expiring, the NAT MAY either send TCP RST +packets to the endpoints or MAY silently abandon the connection. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-6.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-6.toml new file mode 100644 index 0000000000..28b7dd9c07 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-6.toml @@ -0,0 +1,11 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-6" + + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-6: If a NAT includes ALGs that affect TCP, it is RECOMMENDED +that all of those ALGs (except for FTP [RFC0959]) be disabled by +default. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.1.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.1.toml new file mode 100644 index 0000000000..abf16bea56 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.1.toml @@ -0,0 +1,10 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-7.1" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port +overloading" for TCP. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.2.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.2.toml new file mode 100644 index 0000000000..d1cbd7b746 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.2.toml @@ -0,0 +1,16 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-7.2" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-8: A NAT MUST support "hairpinning" for TCP. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) A NAT's hairpinning behavior MUST be of type "External source +IP address and port". +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.3.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.3.toml new file mode 100644 index 0000000000..8752979bda --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-7.3.toml @@ -0,0 +1,17 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-7.3" + + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-9: If a NAT translates TCP, it SHOULD translate ICMP Destination +Unreachable (Type 3) messages. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the +NAT mapping or TCP connection for which the ICMP was generated. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-8.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-8.toml new file mode 100644 index 0000000000..cd2b984c17 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc5382/section-8.toml @@ -0,0 +1,158 @@ +target = "https://www.rfc-editor.org/rfc/rfc5382#section-8" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior +for TCP. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-2: A NAT MUST support all valid sequences of TCP packets +(defined in [RFC0793]) for connections initiated both internally +as well as externally when the connection is permitted by the NAT. +''' + +[[spec]] +level = "MUST" +quote = ''' +In particular: +a) In addition to handling the TCP 3-way handshake mode of +connection initiation, A NAT MUST handle the TCP simultaneous- +open mode of connection initiation. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-3: If application transparency is most important, it is +RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" +behavior for TCP. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +If a more stringent filtering behavior is most +important, it is RECOMMENDED that a NAT have an "Address-Dependent +Filtering" behavior. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The filtering behavior MAY be an option configurable by the +administrator of the NAT. +''' + +[[spec]] +level = "MAY" +quote = ''' +b) The filtering behavior for TCP MAY be independent of the +filtering behavior for UDP. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-4: A NAT MUST NOT respond to an unsolicited inbound SYN packet +for at least 6 seconds after the packet is received. +''' + +[[spec]] +level = "MUST" +quote = ''' +If during +this interval the NAT receives and translates an outbound SYN for +the connection the NAT MUST silently drop the original unsolicited +inbound SYN packet. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +Otherwise, the NAT SHOULD send an ICMP Port +Unreachable error (Type 3, Code 3) for the original SYN, unless +REQ-4a applies. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) The NAT MUST silently drop the original SYN packet if sending a +response violates the security policy of the NAT. +''' + +[[spec]] +level = "MAY" +quote = ''' +REQ-5: If a NAT cannot determine whether the endpoints of a TCP +connection are active, it MAY abandon the session if it has been +idle for some time. +''' + +[[spec]] +level = "MUST" +quote = ''' +In such cases, the value of the "established +connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. +''' + +[[spec]] +level = "MUST" +quote = ''' +The value of the "transitory connection idle-timeout" MUST NOT be +less than 4 minutes. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The value of the NAT idle-timeouts MAY be configurable. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-6: If a NAT includes ALGs that affect TCP, it is RECOMMENDED +that all of those ALGs (except for FTP [RFC0959]) be disabled by +default. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port +overloading" for TCP. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-8: A NAT MUST support "hairpinning" for TCP. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) A NAT's hairpinning behavior MUST be of type "External source +IP address and port". +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-9: If a NAT translates TCP, it SHOULD translate ICMP Destination +Unreachable (Type 3) messages. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the +NAT mapping or TCP connection for which the ICMP was generated. +''' + diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index 3fe31d4b2c..3881a4c39b 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -57,3 +57,127 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4884 TEXT[!MAY]: while ignoring others. TEXT[!MUST]: As stated above, the total length of the ICMP message, including TEXT[!MUST]: extensions, MUST NOT exceed the minimum reassembly buffer size. + +SPECIFICATION: https://www.rfc-editor.org/rfc/rfc5382 + SECTION: [Address and Port Mapping Behavior](#section-4.1) + TEXT[!MUST]: REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior + TEXT[!MUST]: for TCP. + + SECTION: [Internally Initiated Connections](#section-4.2) + TEXT[!MUST]: REQ-2: A NAT MUST support all valid sequences of TCP packets + TEXT[!MUST]: (defined in [RFC0793]) for connections initiated both internally + TEXT[!MUST]: as well as externally when the connection is permitted by the NAT. + TEXT[!MUST]: In particular: + TEXT[!MUST]: a) In addition to handling the TCP 3-way handshake mode of + TEXT[!MUST]: connection initiation, A NAT MUST handle the TCP simultaneous- + TEXT[!MUST]: open mode of connection initiation. + + SECTION: [Externally Initiated Connections](#section-4.3) + TEXT[!SHOULD]: REQ-3: If application transparency is most important, it is + TEXT[!SHOULD]: RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + TEXT[!SHOULD]: behavior for TCP. + TEXT[!SHOULD]: If a more stringent filtering behavior is most + TEXT[!SHOULD]: important, it is RECOMMENDED that a NAT have an "Address-Dependent + TEXT[!SHOULD]: Filtering" behavior. + TEXT[!MAY]: a) The filtering behavior MAY be an option configurable by the + TEXT[!MAY]: administrator of the NAT. + TEXT[!MAY]: b) The filtering behavior for TCP MAY be independent of the + TEXT[!MAY]: filtering behavior for UDP. + TEXT[!MUST]: REQ-4: A NAT MUST NOT respond to an unsolicited inbound SYN packet + TEXT[!MUST]: for at least 6 seconds after the packet is received. + TEXT[!MUST]: If during + TEXT[!MUST]: this interval the NAT receives and translates an outbound SYN for + TEXT[!MUST]: the connection the NAT MUST silently drop the original unsolicited + TEXT[!MUST]: inbound SYN packet. + TEXT[!SHOULD]: Otherwise, the NAT SHOULD send an ICMP Port + TEXT[!SHOULD]: Unreachable error (Type 3, Code 3) for the original SYN, unless + TEXT[!SHOULD]: REQ-4a applies. + TEXT[!MUST]: a) The NAT MUST silently drop the original SYN packet if sending a + TEXT[!MUST]: response violates the security policy of the NAT. + + SECTION: [NAT Session Refresh](#section-5) + TEXT[!MAY]: REQ-5: If a NAT cannot determine whether the endpoints of a TCP + TEXT[!MAY]: connection are active, it MAY abandon the session if it has been + TEXT[!MAY]: idle for some time. + TEXT[!MUST]: In such cases, the value of the "established + TEXT[!MUST]: connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. + TEXT[!MUST]: The value of the "transitory connection idle-timeout" MUST NOT be + TEXT[!MUST]: less than 4 minutes. + TEXT[!MAY]: a) The value of the NAT idle-timeouts MAY be configurable. + TEXT[!MAY]: A NAT MAY hold state for a connection in + TEXT[!MAY]: TIME_WAIT state to accommodate retransmissions of the last ACK. + TEXT[!MAY]: When a NAT abandons a live connection, for + TEXT[!MAY]: example due to a timeout expiring, the NAT MAY either send TCP RST + TEXT[!MAY]: packets to the endpoints or MAY silently abandon the connection. + + SECTION: [Application Level Gateways](#section-6) + TEXT[!SHOULD]: REQ-6: If a NAT includes ALGs that affect TCP, it is RECOMMENDED + TEXT[!SHOULD]: that all of those ALGs (except for FTP [RFC0959]) be disabled by + TEXT[!SHOULD]: default. + + SECTION: [Port Assignment](#section-7.1) + TEXT[!MUST]: REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port + TEXT[!MUST]: overloading" for TCP. + + SECTION: [Hairpinning Behavior](#section-7.2) + TEXT[!MUST]: REQ-8: A NAT MUST support "hairpinning" for TCP. + TEXT[!MUST]: a) A NAT's hairpinning behavior MUST be of type "External source + TEXT[!MUST]: IP address and port". + + SECTION: [ICMP Responses to TCP Packets](#section-7.3) + TEXT[!SHOULD]: REQ-9: If a NAT translates TCP, it SHOULD translate ICMP Destination + TEXT[!SHOULD]: Unreachable (Type 3) messages. + TEXT[!MUST]: REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the + TEXT[!MUST]: NAT mapping or TCP connection for which the ICMP was generated. + + SECTION: [Requirements](#section-8) + TEXT[!MUST,todo]: REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior + TEXT[!MUST,todo]: for TCP. + TEXT[!MUST]: REQ-2: A NAT MUST support all valid sequences of TCP packets + TEXT[!MUST]: (defined in [RFC0793]) for connections initiated both internally + TEXT[!MUST]: as well as externally when the connection is permitted by the NAT. + TEXT[!MUST]: In particular: + TEXT[!MUST]: a) In addition to handling the TCP 3-way handshake mode of + TEXT[!MUST]: connection initiation, A NAT MUST handle the TCP simultaneous- + TEXT[!MUST]: open mode of connection initiation. + TEXT[!SHOULD]: REQ-3: If application transparency is most important, it is + TEXT[!SHOULD]: RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + TEXT[!SHOULD]: behavior for TCP. + TEXT[!SHOULD]: If a more stringent filtering behavior is most + TEXT[!SHOULD]: important, it is RECOMMENDED that a NAT have an "Address-Dependent + TEXT[!SHOULD]: Filtering" behavior. + TEXT[!MAY]: a) The filtering behavior MAY be an option configurable by the + TEXT[!MAY]: administrator of the NAT. + TEXT[!MAY]: b) The filtering behavior for TCP MAY be independent of the + TEXT[!MAY]: filtering behavior for UDP. + TEXT[!MUST]: REQ-4: A NAT MUST NOT respond to an unsolicited inbound SYN packet + TEXT[!MUST]: for at least 6 seconds after the packet is received. + TEXT[!MUST]: If during + TEXT[!MUST]: this interval the NAT receives and translates an outbound SYN for + TEXT[!MUST]: the connection the NAT MUST silently drop the original unsolicited + TEXT[!MUST]: inbound SYN packet. + TEXT[!SHOULD]: Otherwise, the NAT SHOULD send an ICMP Port + TEXT[!SHOULD]: Unreachable error (Type 3, Code 3) for the original SYN, unless + TEXT[!SHOULD]: REQ-4a applies. + TEXT[!MUST]: a) The NAT MUST silently drop the original SYN packet if sending a + TEXT[!MUST]: response violates the security policy of the NAT. + TEXT[!MAY,todo]: REQ-5: If a NAT cannot determine whether the endpoints of a TCP + TEXT[!MAY,todo]: connection are active, it MAY abandon the session if it has been + TEXT[!MAY,todo]: idle for some time. + TEXT[!MUST,todo]: In such cases, the value of the "established + TEXT[!MUST,todo]: connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. + TEXT[!MUST,todo]: The value of the "transitory connection idle-timeout" MUST NOT be + TEXT[!MUST,todo]: less than 4 minutes. + TEXT[!MAY]: a) The value of the NAT idle-timeouts MAY be configurable. + TEXT[!SHOULD]: REQ-6: If a NAT includes ALGs that affect TCP, it is RECOMMENDED + TEXT[!SHOULD]: that all of those ALGs (except for FTP [RFC0959]) be disabled by + TEXT[!SHOULD]: default. + TEXT[!MUST,implementation,test]: REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port + TEXT[!MUST,implementation,test]: overloading" for TCP. + TEXT[!MUST]: REQ-8: A NAT MUST support "hairpinning" for TCP. + TEXT[!MUST]: a) A NAT's hairpinning behavior MUST be of type "External source + TEXT[!MUST]: IP address and port". + TEXT[!SHOULD]: REQ-9: If a NAT translates TCP, it SHOULD translate ICMP Destination + TEXT[!SHOULD]: Unreachable (Type 3) messages. + TEXT[!MUST,implementation,test]: REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the + TEXT[!MUST,implementation,test]: NAT mapping or TCP connection for which the ICMP was generated. diff --git a/.duvet/specifications/www.rfc-editor.org/rfc/rfc5382.txt b/.duvet/specifications/www.rfc-editor.org/rfc/rfc5382.txt new file mode 100644 index 0000000000..995986c3ce --- /dev/null +++ b/.duvet/specifications/www.rfc-editor.org/rfc/rfc5382.txt @@ -0,0 +1,1179 @@ + + + + + + +Network Working Group S. Guha, Ed. +Request for Comments: 5382 Cornell U. +BCP: 142 K. Biswas +Category: Best Current Practice Cisco Systems + B. Ford + MPI-SWS + S. Sivakumar + Cisco Systems + P. Srisuresh + Kazeon Systems + October 2008 + + + NAT Behavioral Requirements for TCP + +Status of This Memo + + This document specifies an Internet Best Current Practices for the + Internet Community, and requests discussion and suggestions for + improvements. Distribution of this memo is unlimited. + +Abstract + + This document defines a set of requirements for NATs that handle TCP + that would allow many applications, such as peer-to-peer applications + and online games to work consistently. Developing NATs that meet + this set of requirements will greatly increase the likelihood that + these applications will function properly. + + + + + + + + + + + + + + + + + + + + + + + +Guha, et al. Best Current Practice [Page 1] + +RFC 5382 NAT TCP Requirements October 2008 + + +Table of Contents + + 1. Applicability Statement . . . . . . . . . . . . . . . . . . . 3 + 2. Introduction . . . . . . . . . . . . . . . . . . . . . . . . . 3 + 3. Terminology . . . . . . . . . . . . . . . . . . . . . . . . . 4 + 4. TCP Connection Initiation . . . . . . . . . . . . . . . . . . 4 + 4.1. Address and Port Mapping Behavior . . . . . . . . . . . . 5 + 4.2. Internally Initiated Connections . . . . . . . . . . . . . 5 + 4.3. Externally Initiated Connections . . . . . . . . . . . . . 7 + 5. NAT Session Refresh . . . . . . . . . . . . . . . . . . . . . 10 + 6. Application Level Gateways . . . . . . . . . . . . . . . . . . 12 + 7. Other Requirements Applicable to TCP . . . . . . . . . . . . . 12 + 7.1. Port Assignment . . . . . . . . . . . . . . . . . . . . . 12 + 7.2. Hairpinning Behavior . . . . . . . . . . . . . . . . . . . 13 + 7.3. ICMP Responses to TCP Packets . . . . . . . . . . . . . . 13 + 8. Requirements . . . . . . . . . . . . . . . . . . . . . . . . . 14 + 9. Security Considerations . . . . . . . . . . . . . . . . . . . 16 + 10. Acknowledgments . . . . . . . . . . . . . . . . . . . . . . . 17 + 11. References . . . . . . . . . . . . . . . . . . . . . . . . . . 18 + 11.1. Normative References . . . . . . . . . . . . . . . . . . . 18 + 11.2. Informational References . . . . . . . . . . . . . . . . . 18 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Guha, et al. Best Current Practice [Page 2] + +RFC 5382 NAT TCP Requirements October 2008 + + +1. Applicability Statement + + This document is adjunct to [BEHAVE-UDP], which defines many terms + relating to NATs, lays out general requirements for all NATs, and + sets requirements for NATs that handle IP and unicast UDP traffic. + The purpose of this document is to set requirements for NATs that + handle TCP traffic. + + The requirements of this specification apply to traditional NATs as + described in [RFC2663]. + + This document only covers the TCP aspects of NAT traversal. + Middlebox behavior that is not necessary for network address + translation of TCP is out of scope. Packet inspection above the TCP + layer and firewalls are out of scope except for Application Level + Gateway (ALG) behavior that may interfere with NAT traversal. + Application and OS aspects of TCP NAT traversal are out of scope. + Signaling-based approaches to NAT traversal, such as Middlebox + Communication (MIDCOM) and Universal Plug and Play (UPnP), that + directly control the NAT are out of scope. Finally, TCP connections + intended for the NAT (e.g., an HTTP or Secure Shell Protocol (SSH) + management interface) and TCP connections initiated by the NAT (e.g., + reliable syslog client) are out of scope. + +2. Introduction + + Network Address Translators (NATs) hinder connectivity in + applications where sessions may be initiated to internal hosts. + Readers may refer to [RFC3022] for detailed information on + traditional NATs. [BEHAVE-UDP] lays out the terminology and + requirements for NATs in the context of IP and UDP. This document + supplements these by setting requirements for NATs that handle TCP + traffic. All definitions and requirements in [BEHAVE-UDP] are + inherited here. + + [RFC4614] chronicles the evolution of TCP from the original + definition [RFC0793] to present-day implementations. While much has + changed in TCP with regards to congestion control and flow control, + security, and support for high-bandwidth networks, the process of + initiating a connection (i.e., the 3-way handshake or simultaneous- + open) has changed little. It is the process of connection initiation + that NATs affect the most. Experimental approaches such as T/TCP + [RFC1644] have proposed alternate connection initiation approaches, + but have been found to be complex and susceptible to denial-of- + service attacks. Modern operating systems and NATs consequently + primarily support the 3-way handshake and simultaneous-open modes of + connection initiation as described in [RFC0793]. + + + + +Guha, et al. Best Current Practice [Page 3] + +RFC 5382 NAT TCP Requirements October 2008 + + + Recently, many techniques have been devised to make peer-to-peer TCP + applications work across NATs. [STUNT], [NATBLASTER], and [P2PNAT] + describe Unilateral Self-Address Fixing (UNSAF) mechanisms that allow + peer-to-peer applications to establish TCP through NATs. These + approaches require only endpoint applications to be modified and work + with standards compliant OS stacks. The approaches, however, depend + on specific NAT behavior that is usually, but not always, supported + by NATs (see [TCPTRAV] and [P2PNAT] for details). Consequently, a + complete TCP NAT traversal solution is sometimes forced to rely on + public TCP relays to traverse NATs that do not cooperate. This + document defines requirements that ensure that TCP NAT traversal + approaches are not forced to use data relays. + +3. Terminology + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and "OPTIONAL" in this + document are to be interpreted as described in [RFC2119]. + + "NAT" in this specification includes both "Basic NAT" and "Network + Address/Port Translator (NAPT)" [RFC2663]. The term "NAT Session" is + adapted from [NAT-MIB] and is defined as follows. + + NAT Session - A NAT session is an association between a TCP session + as seen in the internal realm and a TCP session as seen in the + external realm, by virtue of NAT translation. The NAT session will + provide the translation glue between the two session representations. + + This document uses the term "TCP connection" (or just "connection") + to refer to individual TCP flows identified by the 4-tuple (source + and destination IP address and TCP port) and the initial sequence + numbers (ISN). + + This document uses the term "address and port mapping" (or just + "mapping") as defined in [BEHAVE-UDP] to refer to state at the NAT + necessary for network address and port translation of TCP + connections. This document also uses the terms "Endpoint-Independent + Mapping", "Address-Dependent Mapping", "Address and Port-Dependent + Mapping", "filtering behavior", "Endpoint-Independent Filtering", + "Address-Dependent Filtering", "Address and Port-Dependent + Filtering", "Port assignment", "Port overloading", "hairpinning", and + "External source IP address and port" as defined in [BEHAVE-UDP]. + +4. TCP Connection Initiation + + This section describes various NAT behaviors applicable to TCP + connection initiation. + + + + +Guha, et al. Best Current Practice [Page 4] + +RFC 5382 NAT TCP Requirements October 2008 + + +4.1. Address and Port Mapping Behavior + + A NAT uses a mapping to translate packets for each TCP connection. A + mapping is dynamically allocated for connections initiated from the + internal side, and potentially reused for certain subsequent + connections. NAT behavior regarding when a mapping can be reused + differs for different NATs as described in [BEHAVE-UDP]. + + Consider an internal IP address and TCP port (X:x) that initiates a + TCP connection to an external (Y1:y1) tuple. Let the mapping + allocated by the NAT for this connection be (X1':x1'). Shortly + thereafter, the endpoint initiates a connection from the same (X:x) + to an external address (Y2:y2) and gets the mapping (X2':x2') on the + NAT. As per [BEHAVE-UDP], if (X1':x1') equals (X2':x2') for all + values of (Y2:y2), then the NAT is defined to have "Endpoint- + Independent Mapping" behavior. If (X1':x1') equals (X2':x2') only + when Y2 equals Y1, then the NAT is defined to have "Address-Dependent + Mapping" behavior. If (X1':x1') equals (X2':x2') only when (Y2:y2) + equals (Y1:y1), possible only for consecutive connections to the same + external address shortly after the first is terminated and if the NAT + retains state for connections in TIME_WAIT state, then the NAT is + defined to have "Address and Port-Dependent Mapping" behavior. This + document introduces one additional behavior where (X1':x1') never + equals (X2':x2'), that is, for each connection a new mapping is + allocated; in such a case, the NAT is defined to have "Connection- + Dependent Mapping" behavior. + + REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior + for TCP. + + Justification: REQ-1 is necessary for UNSAF methods to work. + Endpoint-Independent Mapping behavior allows peer-to-peer + applications to learn and advertise the external IP address and + port allocated to an internal endpoint such that external peers + can contact it (subject to the NAT's security policy). The + security policy of a NAT is independent of its mapping behavior + and is discussed later in Section 4.3. Having Endpoint- + Independent Mapping behavior allows peer-to-peer applications to + work consistently without compromising the security benefits of + the NAT. + +4.2. Internally Initiated Connections + + An internal endpoint initiates a TCP connection through a NAT by + sending a SYN packet. The NAT allocates (or reuses) a mapping for + the connection, as described in the previous section. The mapping + defines the external IP address and port used for translation of all + packets for that connection. In particular, for client-server + + + +Guha, et al. Best Current Practice [Page 5] + +RFC 5382 NAT TCP Requirements October 2008 + + + applications where an internal client initiates the connection to an + external server, the mapping is used to translate the outbound SYN, + the resulting inbound SYN-ACK response, the subsequent outbound ACK, + and other packets for the connection. This method of connection + initiation corresponds to the 3-way handshake (defined in [RFC0793]) + and is supported by all NATs. + + Peer-to-peer applications use an alternate method of connection + initiation termed simultaneous-open (Fig. 8, [RFC0793]) to traverse + NATs. In the simultaneous-open mode of operation, both peers send + SYN packets for the same TCP connection. The SYN packets cross in + the network. Upon receiving the other end's SYN packet, each end + responds with a SYN-ACK packet, which also cross in the network. The + connection is considered established once the SYN-ACKs are received. + From the perspective of the NAT, the internal host's SYN packet is + met by an inbound SYN packet for the same connection (as opposed to a + SYN-ACK packet during a 3-way handshake). Subsequent to this + exchange, both an outbound and an inbound SYN-ACK are seen for the + connection. Some NATs erroneously block the inbound SYN for the + connection in progress. Some NATs block or incorrectly translate the + outbound SYN-ACK. Such behavior breaks TCP simultaneous-open and + prevents peer-to-peer applications from functioning correctly behind + a NAT. + + In order to provide network address translation service for TCP, it + is necessary for a NAT to correctly receive, translate, and forward + all packets for a connection that conform to valid transitions of the + TCP State-Machine (Fig. 6, [RFC0793]). + + REQ-2: A NAT MUST support all valid sequences of TCP packets + (defined in [RFC0793]) for connections initiated both internally + as well as externally when the connection is permitted by the NAT. + In particular: + a) In addition to handling the TCP 3-way handshake mode of + connection initiation, A NAT MUST handle the TCP simultaneous- + open mode of connection initiation. + + Justification: The intent of this requirement is to allow standards + compliant TCP stacks to traverse NATs no matter what path the + stacks take through the TCP state-machine and no matter which end + initiates the connection as long as the connection is permitted by + the filtering policy of the NAT (filtering policy is described in + the following section). + a) In addition to TCP packets for a 3-way handshake, A NAT must be + prepared to accept an inbound SYN and an outbound SYN-ACK for + an internally initiated connection in order to support + simultaneous-open. + + + + +Guha, et al. Best Current Practice [Page 6] + +RFC 5382 NAT TCP Requirements October 2008 + + +4.3. Externally Initiated Connections + + The NAT allocates a mapping for the first connection initiated by an + internal endpoint to an external endpoint. In some scenarios, the + NAT's policy may allow this mapping to be reused for connections + initiated from the external side to the internal endpoint. Consider + as before an internal IP address and port (X:x) that is assigned (or + reuses) a mapping (X1':x1') when it initiates a connection to an + external (Y1:y1). An external endpoint (Y2:y2) attempts to initiate + a connection with the internal endpoint by sending a SYN to + (X1':x1'). A NAT can choose to either allow the connection to be + established, or to disallow the connection. If the NAT chooses to + allow the connection, it translates the inbound SYN and routes it to + (X:x) as per the existing mapping. It also translates the SYN-ACK + generated by (X:x) in response and routes it to (Y2:y2), and so on. + Alternately, the NAT can disallow the connection by filtering the + inbound SYN. + + A NAT may allow an existing mapping to be reused by an externally + initiated connection if its security policy permits. Several + different policies are possible as described in [BEHAVE-UDP]. If a + NAT allows the connection initiation from all (Y2:y2), then it is + defined to have "Endpoint-Independent Filtering" behavior. If the + NAT allows connection initiations only when Y2 equals Y1, then the + NAT is defined to have "Address-Dependent Filtering" behavior. If + the NAT allows connection initiations only when (Y2:y2) equals + (Y1:y1), then the NAT is defined to have "Address and Port-Dependent + Filtering" behavior (possible only shortly after the first connection + has been terminated but the mapping is still active). One additional + filtering behavior defined in this document is when the NAT does not + allow any connection initiations from the external side; in such + cases, the NAT is defined to have "Connection-Dependent Filtering" + behavior. The difference between "Address and Port-Dependent + Filtering" and "Connection-Dependent Filtering" behavior is that the + former permits an inbound SYN during the TIME_WAIT state of the first + connection to initiate a new connection while the latter does not. + + REQ-3: If application transparency is most important, it is + RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + behavior for TCP. If a more stringent filtering behavior is most + important, it is RECOMMENDED that a NAT have an "Address-Dependent + Filtering" behavior. + a) The filtering behavior MAY be an option configurable by the + administrator of the NAT. + b) The filtering behavior for TCP MAY be independent of the + filtering behavior for UDP. + + + + + +Guha, et al. Best Current Practice [Page 7] + +RFC 5382 NAT TCP Requirements October 2008 + + + Justification: The intent of this requirement is to allow peer-to- + peer applications that do not always initiate connections from the + internal side of the NAT to continue to work in the presence of + NATs. This behavior also allows applications behind a BEHAVE + compliant NAT to inter-operate with remote endpoints that are + behind non-BEHAVE compliant (legacy) NATs. If the remote + endpoint's NAT does not have Endpoint-Independent Mapping behavior + but has only one external IP address, then an application can + still traverse the combination of the two NATs if the local NAT + has Address-Dependent Filtering. Section 9 contains a detailed + discussion on the security implications of this requirement. + + If the inbound SYN packet is filtered, either because a corresponding + mapping does not exist or because of the NAT's filtering behavior, a + NAT has two basic choices: to ignore the packet silently, or to + signal an error to the sender. Signaling an error through ICMP + messages allows the sender to quickly detect that the SYN did not + reach the intended destination. Silently dropping the packet, on the + other hand, allows applications to perform simultaneous-open more + reliably. + + Silently dropping the SYN aids simultaneous-open as follows. + Consider that the application is attempting a simultaneous-open and + the outbound SYN from the internal endpoint has not yet crossed the + NAT (due to network congestion or clock skew between the two + endpoints); this outbound SYN would otherwise have created the + necessary mapping at the NAT to allow translation of the inbound SYN. + Since the outbound SYN did not reach the NAT in time, the inbound SYN + cannot be processed. If a NAT responds to the premature inbound SYN + with an error message that forces the external endpoint to abandon + the connection attempt, it hinders applications performing a TCP + simultaneous-open. If instead the NAT silently ignores the inbound + SYN, the external endpoint retransmits the SYN after a TCP timeout. + In the meantime, the NAT creates the mapping in response to the + (delayed) outbound SYN such that the retransmitted inbound SYN can be + routed and simultaneous-open can succeed. The downside to this + behavior is that in the event the inbound SYN is erroneous, the + remote side does not learn of the error until after several TCP + timeouts. + + NAT support for simultaneous-open as well as quickly signaling errors + are both important for applications. Unfortunately, there is no way + for a NAT to signal an error without forcing the endpoint to abort a + potential simultaneous-open: TCP RST and ICMP Port Unreachable + packets require the endpoint to abort the attempt while the ICMP Host + and Network Unreachable errors may adversely affect other connections + to the same host or network [RFC1122]. + + + + +Guha, et al. Best Current Practice [Page 8] + +RFC 5382 NAT TCP Requirements October 2008 + + + In addition, when an unsolicited SYN is received by the NAT, the NAT + may not know whether the application is attempting a simultaneous- + open (and that it should therefore silently drop the SYN) or whether + the SYN is in error (and that it should notify the sender). + + REQ-4: A NAT MUST NOT respond to an unsolicited inbound SYN packet + for at least 6 seconds after the packet is received. If during + this interval the NAT receives and translates an outbound SYN for + the connection the NAT MUST silently drop the original unsolicited + inbound SYN packet. Otherwise, the NAT SHOULD send an ICMP Port + Unreachable error (Type 3, Code 3) for the original SYN, unless + REQ-4a applies. + a) The NAT MUST silently drop the original SYN packet if sending a + response violates the security policy of the NAT. + + Justification: The intent of this requirement is to allow + simultaneous-open to work reliably in the presence of NATs as well + as to quickly signal an error in case the unsolicited SYN is in + error. As of writing this memo, it is not possible to achieve + both; the requirement therefore represents a compromise. The NAT + should tolerate some delay in the outbound SYN for a TCP + simultaneous-open, which may be due to network congestion or loose + synchronization between the endpoints. If the unsolicited SYN is + not part of a simultaneous-open attempt and is in error, the NAT + should endeavor to signal the error in accordance with [RFC1122]. + a) There may, however, be reasons for the NAT to rate-limit or + omit such error notifications, for example, in the case of an + attack. Silently dropping the SYN packet when under attack + allows simultaneous-open to work without consuming any extra + network bandwidth or revealing the presence of the NAT to + attackers. Section 9 mentions the security considerations for + this requirement. + + For NATs that combine NAT functionality with end-host functionality + (e.g., an end-host that also serves as a NAT for other hosts behind + it), REQ-4 above applies only to SYNs intended for the NAT'ed hosts + and not to SYNs intended for the NAT itself. One way to determine + whether the inbound SYN is intended for a NAT'ed host is to allocate + NAT mappings from one port range, and allocate ports for local + endpoints from a different non-overlapping port range. More dynamic + implementations can be imagined. + + + + + + + + + + +Guha, et al. Best Current Practice [Page 9] + +RFC 5382 NAT TCP Requirements October 2008 + + +5. NAT Session Refresh + + A NAT maintains state associated with in-progress and established + connections. Because of this, a NAT is susceptible to a resource- + exhaustion attack whereby an attacker (or virus) on the internal side + attempts to cause the NAT to create more state than for which it has + resources. To prevent such an attack, a NAT needs to abandon + sessions in order to free the state resources. + + A common method that is applicable only to TCP is to preferentially + abandon sessions for crashed endpoints, followed by closed TCP + connections and partially open connections. A NAT can check if an + endpoint for a session has crashed by sending a TCP keep-alive packet + and receiving a TCP RST packet in response. If the NAT cannot + determine whether the endpoint is active, it should not abandon the + session until the TCP connection has been idle for some time. Note + that an established TCP connection can stay idle (but live) + indefinitely; hence, there is no fixed value for an idle-timeout that + accommodates all applications. However, a large idle-timeout + motivated by recommendations in [RFC1122] can reduce the chances of + abandoning a live session. + + A TCP connection passes through three phases: partially open, + established, and closing. During the partially open phase, endpoints + synchronize initial sequence numbers. The phase is initiated by the + first SYN for the connection and extends until both endpoints have + sent a packet with the ACK flag set (TCP states: SYN_SENT and + SYN_RCVD). ACKs in both directions mark the beginning of the + established phase where application data can be exchanged + indefinitely (TCP states: ESTABLISHED, FIN_WAIT_1, FIN_WAIT_2, and + CLOSE_WAIT). The closing phase begins when both endpoints have + terminated their half of the connection by sending a FIN packet. + Once FIN packets are seen in both directions, application data can no + longer be exchanged, but the stacks still need to ensure that the FIN + packets are received (TCP states: CLOSING and LAST_ACK). + + TCP connections can stay in established phase indefinitely without + exchanging any packets. Some end-hosts can be configured to send + keep-alive packets on such idle connections; by default, such keep- + alive packets are sent every 2 hours if enabled [RFC1122]. + Consequently, a NAT that waits for slightly over 2 hours can detect + idle connections with keep-alive packets being sent at the default + rate. TCP connections in the partially open or closing phases, on + the other hand, can stay idle for at most 4 minutes while waiting for + in-flight packets to be delivered [RFC1122]. + + + + + + +Guha, et al. Best Current Practice [Page 10] + +RFC 5382 NAT TCP Requirements October 2008 + + + The "established connection idle-timeout" for a NAT is defined as the + minimum time a TCP connection in the established phase must remain + idle before the NAT considers the associated session a candidate for + removal. The "transitory connection idle-timeout" for a NAT is + defined as the minimum time a TCP connection in the partially open or + closing phases must remain idle before the NAT considers the + associated session a candidate for removal. TCP connections in the + TIME_WAIT state are not affected by the "transitory connection idle- + timeout". + + REQ-5: If a NAT cannot determine whether the endpoints of a TCP + connection are active, it MAY abandon the session if it has been + idle for some time. In such cases, the value of the "established + connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. + The value of the "transitory connection idle-timeout" MUST NOT be + less than 4 minutes. + a) The value of the NAT idle-timeouts MAY be configurable. + + Justification: The intent of this requirement is to minimize the + cases where a NAT abandons session state for a live connection. + While some NATs may choose to abandon sessions reactively in + response to new connection initiations (allowing idle connections + to stay up indefinitely in the absence of new initiations), other + NATs may choose to proactively reap idle sessions. In cases where + the NAT cannot actively determine if the connection is alive, this + requirement ensures that applications can send keep-alive packets + at the default rate (every 2 hours) such that the NAT can + passively determine that the connection is alive. The additional + 4 minutes allows time for in-flight packets to cross the NAT. + + NAT behavior for handling RST packets, or connections in TIME_WAIT + state is left unspecified. A NAT MAY hold state for a connection in + TIME_WAIT state to accommodate retransmissions of the last ACK. + However, since the TIME_WAIT state is commonly encountered by + internal endpoints properly closing the TCP connection, holding state + for a closed connection may limit the throughput of connections + through a NAT with limited resources. [RFC1337] describes hazards + associated with TIME_WAIT assassination. + + The handling of non-SYN packets for connections for which there is no + active mapping is left unspecified. Such packets may be received if + the NAT silently abandons a live connection, or abandons a connection + in TIME_WAIT state before the 4 minute TIME_WAIT period expires. The + decision to either silently drop such packets or to respond with a + TCP RST packet is left up to the implementation. + + + + + + +Guha, et al. Best Current Practice [Page 11] + +RFC 5382 NAT TCP Requirements October 2008 + + + NAT behavior for notifying endpoints when abandoning live connections + is left unspecified. When a NAT abandons a live connection, for + example due to a timeout expiring, the NAT MAY either send TCP RST + packets to the endpoints or MAY silently abandon the connection. + + Sending a RST notification allows endpoint applications to recover + more quickly; however, notifying the endpoints may not always be + possible if, for example, session state is lost due to a power + failure. + +6. Application Level Gateways + + Application Level Gateways (ALGs) in certain NATs modify IP addresses + and TCP ports embedded inside application protocols. Such ALGs may + interfere with UNSAF methods or protocols that try to be NAT-aware + and must therefore be used with extreme caution. + + REQ-6: If a NAT includes ALGs that affect TCP, it is RECOMMENDED + that all of those ALGs (except for FTP [RFC0959]) be disabled by + default. + + Justification: The intent of this requirement is to prevent ALGs + from interfering with UNSAF methods. The default state of an FTP + ALG is left unspecified because of legacy concerns: as of writing + this memo, a large fraction of legacy FTP clients do not enable + passive (PASV) mode by default and require an ALG to traverse + NATs. + +7. Other Requirements Applicable to TCP + + A list of general and UDP-specific NAT behavioral requirements are + described in [BEHAVE-UDP]. A list of ICMP-specific NAT behavioral + requirements are described in [BEHAVE-ICMP]. The requirements listed + below reiterate the requirements from these two documents that + directly affect TCP. The following requirements do not relax any + requirements in [BEHAVE-UDP] or [BEHAVE-ICMP]. + +7.1. Port Assignment + + NATs that allow different internal endpoints to simultaneously use + the same mapping are defined in [BEHAVE-UDP] to have a "Port + assignment" behavior of "Port overloading". Such behavior is + undesirable, as it prevents two internal endpoints sharing the same + mapping from establishing simultaneous connections to a common + external endpoint. + + REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port + overloading" for TCP. + + + +Guha, et al. Best Current Practice [Page 12] + +RFC 5382 NAT TCP Requirements October 2008 + + + Justification: This requirement allows two applications on the + internal side of the NAT to consistently communicate with the same + destination. + + NAT behavior for preserving the source TCP port range for connections + is left unspecified. Some applications expect the source TCP port to + be in the well-known range (TCP ports from 0 to 1023). The "r" + series of commands (rsh, rcp, rlogin, etc.) are an example. NATs + that preserve the range from which the source port is picked allow + such applications to function properly through the NAT; however, by + doing so the NAT may compromise the security of the application in + certain situations; applications that depend only on the IP address + and source TCP port range for security (the "r" commands, for + example) cannot distinguish between an attacker and a legitimate user + behind the same NAT. + +7.2. Hairpinning Behavior + + NATs that forward packets originating from an internal address, + destined for an external address that matches the active mapping for + an internal address, back to that internal address are defined in + [BEHAVE-UDP] as supporting "hairpinning". If the NAT presents the + hairpinned packet with an external source IP address and port (i.e., + the mapped source address and port of the originating internal + endpoint), then it is defined to have "External source IP address and + port" for hairpinning. Hairpinning is necessary to allow two + internal endpoints (known to each other only by their external mapped + addresses) to communicate with each other. "External source IP + address and port" behavior for hairpinning avoids confusing + implementations that expect the external source IP address and port. + + REQ-8: A NAT MUST support "hairpinning" for TCP. + a) A NAT's hairpinning behavior MUST be of type "External source + IP address and port". + + Justification: This requirement allows two applications behind the + same NAT that are trying to communicate with each other using + their external addresses. + a) Using the external source address and port for the hairpinned + packet is necessary for applications that do not expect to + receive a packet from a different address than the external + address they are trying to communicate with. + +7.3. ICMP Responses to TCP Packets + + Several TCP mechanisms depend on the reception of ICMP error messages + triggered by the transmission of TCP segments. One such mechanism is + path MTU discovery [RFC1191], which is required for the correct + + + +Guha, et al. Best Current Practice [Page 13] + +RFC 5382 NAT TCP Requirements October 2008 + + + operation of TCP. The current path MTU discovery mechanism requires + the sender of TCP segments to be notified of ICMP "Datagram Too Big" + responses. + + REQ-9: If a NAT translates TCP, it SHOULD translate ICMP Destination + Unreachable (Type 3) messages. + + Justification: Translating ICMP Destination Unreachable messages, + particularly the "Fragmentation Needed and Don't Fragment was Set" + (Type 3, Code 4) message avoids communication failures ("black + holes" [RFC2923]). Furthermore, TCP's connection establishment + and maintenance mechanisms also behave much more efficiently when + ICMP Destination Unreachable messages arrive in response to + outgoing TCP segments. + + REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the + NAT mapping or TCP connection for which the ICMP was generated. + + Justification: This is necessary for reliably performing TCP + simultaneous-open where a remote NAT may temporarily signal an + ICMP error. + +8. Requirements + + A NAT that supports all of the mandatory requirements of this + specification (i.e., the "MUST") and is compliant with [BEHAVE-UDP], + is "compliant with this specification". A NAT that supports all of + the requirements of this specification (i.e., included the + "RECOMMENDED") and is fully compliant with [BEHAVE-UDP] is "fully + compliant with all the mandatory and recommended requirements of this + specification". + + REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior + for TCP. + + REQ-2: A NAT MUST support all valid sequences of TCP packets + (defined in [RFC0793]) for connections initiated both internally + as well as externally when the connection is permitted by the NAT. + In particular: + a) In addition to handling the TCP 3-way handshake mode of + connection initiation, A NAT MUST handle the TCP simultaneous- + open mode of connection initiation. + + REQ-3: If application transparency is most important, it is + RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + behavior for TCP. If a more stringent filtering behavior is most + important, it is RECOMMENDED that a NAT have an "Address-Dependent + Filtering" behavior. + + + +Guha, et al. Best Current Practice [Page 14] + +RFC 5382 NAT TCP Requirements October 2008 + + + a) The filtering behavior MAY be an option configurable by the + administrator of the NAT. + b) The filtering behavior for TCP MAY be independent of the + filtering behavior for UDP. + + REQ-4: A NAT MUST NOT respond to an unsolicited inbound SYN packet + for at least 6 seconds after the packet is received. If during + this interval the NAT receives and translates an outbound SYN for + the connection the NAT MUST silently drop the original unsolicited + inbound SYN packet. Otherwise, the NAT SHOULD send an ICMP Port + Unreachable error (Type 3, Code 3) for the original SYN, unless + REQ-4a applies. + a) The NAT MUST silently drop the original SYN packet if sending a + response violates the security policy of the NAT. + + REQ-5: If a NAT cannot determine whether the endpoints of a TCP + connection are active, it MAY abandon the session if it has been + idle for some time. In such cases, the value of the "established + connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. + The value of the "transitory connection idle-timeout" MUST NOT be + less than 4 minutes. + a) The value of the NAT idle-timeouts MAY be configurable. + + REQ-6: If a NAT includes ALGs that affect TCP, it is RECOMMENDED + that all of those ALGs (except for FTP [RFC0959]) be disabled by + default. + + The following requirements reiterate requirements from [BEHAVE-UDP] + or [BEHAVE-ICMP] that directly affect TCP. This document does not + relax any requirements in [BEHAVE-UDP] or [BEHAVE-ICMP]. + + REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port + overloading" for TCP. + + REQ-8: A NAT MUST support "hairpinning" for TCP. + a) A NAT's hairpinning behavior MUST be of type "External source + IP address and port". + + REQ-9: If a NAT translates TCP, it SHOULD translate ICMP Destination + Unreachable (Type 3) messages. + + REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the + NAT mapping or TCP connection for which the ICMP was generated. + + + + + + + + +Guha, et al. Best Current Practice [Page 15] + +RFC 5382 NAT TCP Requirements October 2008 + + +9. Security Considerations + + [BEHAVE-UDP] discusses security considerations for NATs that handle + IP and unicast UDP traffic. Security concerns specific to handling + TCP packets are discussed in this section. + + Security considerations for REQ-1: This requirement does not + introduce any TCP-specific security concerns. + + Security considerations for REQ-2: This requirement does not + introduce any TCP-specific security concerns. Simultaneous-open + and other transitions in the TCP state machine are by-design and + necessary for TCP to work correctly in all scenarios. Further, + this requirement only affects connections already in progress as + authorized by the NAT in accordance with its policy. + + Security considerations for REQ-3: The security provided by the NAT + is governed by its filtering behavior as addressed in + [BEHAVE-UDP]. Connection-Dependent Filtering behavior is most + secure from a firewall perspective, but severely restricts + connection initiations through a NAT. Endpoint-Independent + Filtering behavior, which is most transparent to applications, + requires an attacker to guess the IP address and port of an active + mapping in order to get his packet to an internal host. Address- + Dependent Filtering, on the other hand, is less transparent than + Endpoint-Independent Filtering but more transparent than + Connection-Dependent Filtering; it is more secure than Endpoint- + Independent Filtering as it requires an attacker to additionally + guess the address of the external endpoint for a NAT session + associated with the mapping and be able to receive packets + addressed to the same. While this protects against most attackers + on the Internet, it does not necessarily protect against attacks + that originate from behind a remote NAT with a single IP address + that is also translating a legitimate connection to the victim. + + Security considerations for REQ-4: This document recommends that a + NAT respond to unsolicited inbound SYN packets with an ICMP error + delayed by a few seconds. Doing so may reveal the presence of a + NAT to an external attacker. Silently dropping the SYN makes it + harder to diagnose network problems and forces applications to + wait for the TCP stack to finish several retransmissions before + reporting an error. An implementer must therefore understand and + carefully weigh the effects of not sending an ICMP error or rate- + limiting such ICMP errors to a very small number. + + + + + + + +Guha, et al. Best Current Practice [Page 16] + +RFC 5382 NAT TCP Requirements October 2008 + + + Security considerations for REQ-5: This document recommends that a + NAT that passively monitors TCP state keep idle sessions alive for + at least 2 hours 4 minutes or 4 minutes depending on the state of + the connection. If a NAT is under attack, it may attempt to + actively determine the liveliness of a TCP connection or let the + NAT administrator configure more conservative timeouts. + + Security considerations for REQ-6: This requirement does not + introduce any TCP-specific security concerns. + + Security considerations for REQ-7: This requirement does not + introduce any TCP-specific security concerns. + + Security considerations for REQ-8: This requirement does not + introduce any TCP-specific security concerns. + + Security considerations for REQ-9: This requirement does not + introduce any TCP-specific security concerns. + + Security considerations for REQ-10: This requirement does not + introduce any TCP-specific security concerns. + + NAT implementations that modify TCP sequence numbers (e.g., for + privacy reasons or for ALG support) must ensure that TCP packets with + Selective Acknowledgement (SACK) notifications [RFC2018] are properly + handled. + + NAT implementations that modify local state based on TCP flags in + packets must ensure that out-of-window TCP packets are properly + handled. [RFC4953] summarizes and discusses a variety of solutions + designed to prevent attackers from affecting TCP connections. + +10. Acknowledgments + + Joe Touch contributed the mechanism for handling unsolicited inbound + SYNs. Thanks to Mark Allman, Francois Audet, Lars Eggert, Paul + Francis, Fernando Gont, Sam Hartman, Paul Hoffman, Dave Hudson, + Cullen Jennings, Philip Matthews, Tom Petch, Magnus Westerlund, and + Dan Wing for their many contributions, comments, and suggestions. + + + + + + + + + + + + +Guha, et al. Best Current Practice [Page 17] + +RFC 5382 NAT TCP Requirements October 2008 + + +11. References + +11.1. Normative References + + [BEHAVE-UDP] Audet, F. and C. Jennings, "Network Address + Translation (NAT) Behavioral Requirements for Unicast + UDP", BCP 127, RFC 4787, January 2007. + + [RFC0793] Postel, J., "Transmission Control Protocol", STD 7, + RFC 793, September 1981. + + [RFC0959] Postel, J. and J. Reynolds, "File Transfer Protocol", + STD 9, RFC 959, October 1985. + + [RFC1122] Braden, R., "Requirements for Internet Hosts - + Communication Layers", STD 3, RFC 1122, October 1989. + + [RFC1191] Mogul, J. and S. Deering, "Path MTU discovery", + RFC 1191, November 1990. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, March 1997. + +11.2. Informational References + + [BEHAVE-ICMP] Srisuresh, P., Ford, B., Sivakumar, S., and S. Guha, + "NAT Behavioral Requirements for ICMP protocol", Work + in Progress, June 2008. + + [NAT-MIB] Rohit, R., Srisuresh, P., Raghunarayan, R., Pai, N., + and C. Wang, "Definitions of Managed Objects for + Network Address Translators (NAT)", RFC 4008, + March 2005. + + [NATBLASTER] Biggadike, A., Ferullo, D., Wilson, G., and A. Perrig, + "NATBLASTER: Establishing TCP connections between + hosts behind NATs", Proceedings of the ACM SIGCOMM + Asia Workshop (Beijing, China), April 2005. + + [P2PNAT] Ford, B., Srisuresh, P., and D. Kegel, "Peer-to-peer + communication across network address translators", + Proceedings of the USENIX Annual Technical + Conference (Anaheim, CA), April 2005. + + [RFC1337] Braden, B., "TIME-WAIT Assassination Hazards in TCP", + RFC 1337, May 1992. + + + + + +Guha, et al. Best Current Practice [Page 18] + +RFC 5382 NAT TCP Requirements October 2008 + + + [RFC1644] Braden, B., "T/TCP -- TCP Extensions for Transactions + Functional Specification", RFC 1644, July 1994. + + [RFC2018] Mathis, M., Mahdavi, J., Floyd, S., and A. Romanow, + "TCP Selective Acknowledgment Options", RFC 2018, + October 1996. + + [RFC2663] Srisuresh, P. and M. Holdrege, "IP Network Address + Translator (NAT) Terminology and Considerations", + RFC 2663, August 1999. + + [RFC2923] Lahey, K., "TCP Problems with Path MTU Discovery", + RFC 2923, September 2000. + + [RFC3022] Srisuresh, P. and K. Egevang, "Traditional IP Network + Address Translator (Traditional NAT)", RFC 3022, + January 2001. + + [RFC4614] Duke, M., Braden, R., Eddy, W., and E. Blanton, "A + Roadmap for Transmission Control Protocol (TCP) + Specification Documents", RFC 4614, September 2006. + + [RFC4953] Touch, J., "Defending TCP Against Spoofing Attacks", + RFC 4953, July 2007. + + [STUNT] Guha, S. and P. Francis, "NUTSS: A SIP based approach + to UDP and TCP connectivity", Proceedings of the ACM + SIGCOMM Workshop on Future Directions in Network + Architecture (Portland, OR), August 2004. + + [TCPTRAV] Guha, S. and P. Francis, "Characterization and + Measurement of TCP Traversal through NATs and + Firewalls", Proceedings of the Internet Measurement + Conference (Berkeley, CA), October 2005. + + + + + + + + + + + + + + + + + +Guha, et al. Best Current Practice [Page 19] + +RFC 5382 NAT TCP Requirements October 2008 + + +Authors' Addresses + + Saikat Guha (editor) + Cornell University + 331 Upson Hall + Ithaca, NY 14853 + US + Phone: +1 607 255 1008 + EMail: saikat@cs.cornell.edu + + Kaushik Biswas + Cisco Systems, Inc. + 170 West Tasman Dr. + San Jose, CA 95134 + US + Phone: +1 408 525 5134 + EMail: kbiswas@cisco.com + + Bryan Ford + Max Planck Institute for Software Systems + Campus Building E1 4 + D-66123 Saarbruecken + Germany + Phone: +49-681-9325657 + EMail: baford@mpi-sws.org + + Senthil Sivakumar + Cisco Systems, Inc. + 7100-8 Kit Creek Road + PO Box 14987 + Research Triangle Park, NC 27709-4987 + US + Phone: +1 919 392 5158 + EMail: ssenthil@cisco.com + + Pyda Srisuresh + Kazeon Systems, Inc. + 1161 San Antonio Rd. + Mountain View, CA 94043 + US + Phone: +1 408 836 4773 + EMail: srisuresh@yahoo.com + + + + + + + + + +Guha, et al. Best Current Practice [Page 20] + +RFC 5382 NAT TCP Requirements October 2008 + + +Full Copyright Statement + + Copyright (C) The IETF Trust (2008). + + This document is subject to the rights, licenses and restrictions + contained in BCP 78, and except as set forth therein, the authors + retain all their rights. + + This document and the information contained herein are provided on an + "AS IS" basis and THE CONTRIBUTOR, THE ORGANIZATION HE/SHE REPRESENTS + OR IS SPONSORED BY (IF ANY), THE INTERNET SOCIETY, THE IETF TRUST AND + THE INTERNET ENGINEERING TASK FORCE DISCLAIM ALL WARRANTIES, EXPRESS + OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTY THAT THE USE OF + THE INFORMATION HEREIN WILL NOT INFRINGE ANY RIGHTS OR ANY IMPLIED + WARRANTIES OF MERCHANTABILITY OR FITNESS FOR A PARTICULAR PURPOSE. + +Intellectual Property + + The IETF takes no position regarding the validity or scope of any + Intellectual Property Rights or other rights that might be claimed to + pertain to the implementation or use of the technology described in + this document or the extent to which any license under such rights + might or might not be available; nor does it represent that it has + made any independent effort to identify any such rights. Information + on the procedures with respect to rights in RFC documents can be + found in BCP 78 and BCP 79. + + Copies of IPR disclosures made to the IETF Secretariat and any + assurances of licenses to be made available, or the result of an + attempt made to obtain a general license or permission for the use of + such proprietary rights by implementers or users of this + specification can be obtained from the IETF on-line IPR repository at + http://www.ietf.org/ipr. + + The IETF invites any interested party to bring to its attention any + copyrights, patents or patent applications, or other proprietary + rights that may cover technology that may be required to implement + this standard. Please address the information to the IETF at + ietf-ipr@ietf.org. + + + + + + + + + + + + +Guha, et al. Best Current Practice [Page 21] + diff --git a/nat/src/masquerade/apalloc/mod.rs b/nat/src/masquerade/apalloc/mod.rs index 05b9204473..3e3822421f 100644 --- a/nat/src/masquerade/apalloc/mod.rs +++ b/nat/src/masquerade/apalloc/mod.rs @@ -281,6 +281,13 @@ impl NatAllocator { self.genid.store(genid, Ordering::Relaxed); } + //= https://www.rfc-editor.org/rfc/rfc5382#section-8 + //= type=todo + //# REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior + //# for TCP. + //= https://www.rfc-editor.org/rfc/rfc5382#section-8 + //# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port + //# overloading" for TCP. fn allocate_v4( &self, src_vpcd: VpcDiscriminant, diff --git a/nat/src/masquerade/fuzz.rs b/nat/src/masquerade/fuzz.rs index ca7d004bc9..cf183b1007 100644 --- a/nat/src/masquerade/fuzz.rs +++ b/nat/src/masquerade/fuzz.rs @@ -200,6 +200,10 @@ fn out_unchanged(out: &[Packet], before: (IpAddr, u16)) -> bool { out[0].is_done() || source_of(&out[0]) == before } +//= https://www.rfc-editor.org/rfc/rfc5382#section-8 +//= type=test +//# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port +//# overloading" for TCP. #[test] fn distinct_flows_do_not_share_a_translation() { let tally = Tally::default(); diff --git a/nat/src/masquerade/nf.rs b/nat/src/masquerade/nf.rs index 84974c52d0..e3b0313cfa 100644 --- a/nat/src/masquerade/nf.rs +++ b/nat/src/masquerade/nf.rs @@ -97,6 +97,14 @@ impl Masquerade { }; // Internal flow timeouts for masquerading + //= https://www.rfc-editor.org/rfc/rfc5382#section-8 + //= type=todo + //# REQ-5: If a NAT cannot determine whether the endpoints of a TCP + //# connection are active, it MAY abandon the session if it has been + //# idle for some time. In such cases, the value of the "established + //# connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. + //# The value of the "transitory connection idle-timeout" MUST NOT be + //# less than 4 minutes. pub const MASQUERADE_ONEWAY_TIMEOUT: Duration = Duration::from_secs(5 * Self::TIMEOUT_SCALE); pub const MASQUERADE_TWOWAY_TIMEOUT: Duration = Duration::from_secs(3 * Self::TIMEOUT_SCALE); pub const MASQUERADE_CLOSING_TIMEOUT: Duration = Duration::from_secs(2 * Self::TIMEOUT_SCALE); diff --git a/nat/src/masquerade/protocol.rs b/nat/src/masquerade/protocol.rs index 1e101a3ee9..fee71b3b5e 100644 --- a/nat/src/masquerade/protocol.rs +++ b/nat/src/masquerade/protocol.rs @@ -50,6 +50,9 @@ fn next_flow_status_udp(action: NatAction, status: NatFlowStatus) -> NatFlowStat } } +//= https://www.rfc-editor.org/rfc/rfc5382#section-8 +//# REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the +//# NAT mapping or TCP connection for which the ICMP was generated. #[allow(clippy::match_single_binding)] fn next_flow_status_icmp(action: NatAction, status: NatFlowStatus) -> NatFlowStatus { match action { diff --git a/nat/src/masquerade/state_machine.rs b/nat/src/masquerade/state_machine.rs index d7616493b9..ca85d854c2 100644 --- a/nat/src/masquerade/state_machine.rs +++ b/nat/src/masquerade/state_machine.rs @@ -195,6 +195,10 @@ fn ordinary_udp_opens_and_settles() { } } +//= https://www.rfc-editor.org/rfc/rfc5382#section-8 +//= type=test +//# REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the +//# NAT mapping or TCP connection for which the ICMP was generated. #[test] fn an_icmp_reply_makes_a_flow_two_way_and_nothing_more() { let packet = build_test_icmp4_echo( From fccaeda44069afbba185cd8afb00c2b7f8fe78e5 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Tue, 18 Aug 2026 22:49:54 -0600 Subject: [PATCH 04/21] build(duvet): Track RFC 4787, and record where masquerade departs from it A trial run, chosen to answer two questions: whether a second specification is cheaper than the first, and where the procedure hurts. RFC 4787 is the UDP counterpart of RFC 5382, so it lands on code that already carries citations. The first answer is much cheaper, because the specifications overlap. Four of the nine requirements cited here are restatements of RFC 5382 clauses already cited on the same lines -- REQ-1 is RFC 5382 REQ-1 without the "for TCP" qualifier, REQ-3 is REQ-7, REQ-12 is REQ-10. Those took minutes and mostly consist of noting that one decision settles two citations. The second answer is that the requirements which do not overlap are where the findings are. REQ-5, the UDP mapping timer, is a materially worse deviation than its TCP sibling and the reason is structural. RFC 5382 distinguishes "established" from "transitory" connections, which is what makes our five- and three-second constants arguable -- they govern states where a connection is opening or closing. RFC 4787 draws no such distinction. A UDP mapping is a UDP mapping and the timer is "the time a mapping will stay active without packets traversing the NAT", against a floor of two minutes. Traced through `next_flow_status_udp`, a plain request/response exchange creates the flow at `OneWay` (five seconds), the reply moves it to `TwoWay` (three), and only a second outbound packet reaches `Established` and the two-minute `idle_timeout`. One round trip and a four-second pause loses the mapping. That is short of the floor by a factor of forty, for all UDP that is not a resolver exchange, and REQ-5a does not cover it: that exemption is for timers specific to one IANA-registered application on one well-known port, not a blanket rule. The resolver fast-close in `protocol.rs` turns out to be exactly what REQ-5a describes, and is cited as such -- with the caveat that 8853 sits above 1023, so the exemption reaches 53 and 853 but not it. REQ-14 is cited onto an existing bare `TODO: Check whether the packet is fragmented`. The TODO was already right; it now says which BCP it is a TODO about, and records that REQ-14a wants out-of-order fragment handling that cannot become a denial of service vector. REQ-9, hairpinning, is a MUST with no implementation anywhere -- "hairpin" does not appear in the workspace. There is a real argument that it does not apply under masquerade, where a public tuple exists only for the lifetime of an outbound flow and is not something a peer can learn and dial. The argument is recorded next to the citation and the citation is still `todo`, because nobody who owns that decision has made it. REQ-3a is the first `exception` in the tree, and is what an exception is for: `setup.rs` excludes well-known ports for TCP and UDP alike, so a host sourcing from a port below 1024 is always translated above it and the requirement can never be met. That was decided -- the range is named, a flag carries it, tests hold it -- so it is recorded as a decision rather than as an omission. Cited as the individual RFC. BCP 127 concatenates RFC 4787, 6888 and 7857, whose section numbers collide, and duvet silently keeps only the last. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/config.toml | 3 + .../rfc/rfc4787/section-10.toml | 18 + .../rfc/rfc4787/section-11.toml | 20 + .../rfc/rfc4787/section-12.toml | 212 +++ .../rfc/rfc4787/section-4.1.toml | 16 + .../rfc/rfc4787/section-4.2.1.toml | 25 + .../rfc/rfc4787/section-4.2.2.toml | 10 + .../rfc/rfc4787/section-4.3.toml | 46 + .../rfc/rfc4787/section-4.4.toml | 14 + .../rfc/rfc4787/section-5.toml | 26 + .../rfc/rfc4787/section-6.toml | 16 + .../rfc/rfc4787/section-7.toml | 18 + .../rfc/rfc4787/section-8.toml | 12 + .../rfc/rfc4787/section-9.toml | 47 + .duvet/snapshot.txt | 173 ++ .../www.rfc-editor.org/rfc/rfc4787.txt | 1627 +++++++++++++++++ nat/src/masquerade/apalloc/mod.rs | 6 + nat/src/masquerade/apalloc/port_alloc.rs | 4 + nat/src/masquerade/fuzz.rs | 4 + nat/src/masquerade/mod.rs | 3 + nat/src/masquerade/nf.rs | 13 + nat/src/masquerade/protocol.rs | 8 + 22 files changed, 2321 insertions(+) create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-10.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-11.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-12.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.1.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.1.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.2.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.3.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.4.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-5.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-6.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-7.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-8.toml create mode 100644 .duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-9.toml create mode 100644 .duvet/specifications/www.rfc-editor.org/rfc/rfc4787.txt diff --git a/.duvet/config.toml b/.duvet/config.toml index 0aa726f4d7..53f123a632 100644 --- a/.duvet/config.toml +++ b/.duvet/config.toml @@ -14,3 +14,6 @@ enabled = true [report.snapshot] enabled = true + +[[specification]] +source = "https://www.rfc-editor.org/rfc/rfc4787" diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-10.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-10.toml new file mode 100644 index 0000000000..b6605f51fa --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-10.toml @@ -0,0 +1,18 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-10" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-13: If the packet received on an internal IP address has DF=1, +the NAT MUST send back an ICMP message "Fragmentation needed and +DF set" to the host, as described in [RFC0792]. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) If the packet has DF=0, the NAT MUST fragment the packet and +SHOULD send the fragments in order. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-11.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-11.toml new file mode 100644 index 0000000000..94f0cb1e06 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-11.toml @@ -0,0 +1,20 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-11" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-14: A NAT MUST support receiving in-order and out-of-order +fragments, so it MUST have "Received Fragment Out of Order" +behavior. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) A NAT's out-of-order fragment processing mechanism MUST be +designed so that fragmentation-based DoS attacks do not +compromise the NAT's ability to process in-order and +unfragmented IP packets. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-12.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-12.toml new file mode 100644 index 0000000000..7315282fee --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-12.toml @@ -0,0 +1,212 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-12" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" +behavior of "Paired". +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port +overloading". +''' + +[[spec]] +level = "SHOULD" +quote = ''' +a) If the host's source port was in the range 0-1023, it is +RECOMMENDED the NAT's source port be in the same range. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +If the +host's source port was in the range 1024-65535, it is +RECOMMENDED that the NAT's source port be in that range. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-4: It is RECOMMENDED that a NAT have a "Port parity +preservation" behavior of "Yes". +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two +minutes, unless REQ-5a applies. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) For specific destination ports in the well-known port range +(ports 0-1023), a NAT MAY have shorter UDP mapping timers that +are specific to the IANA-registered application running over +that specific destination port. +''' + +[[spec]] +level = "MAY" +quote = ''' +b) The value of the NAT UDP mapping timer MAY be configurable. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +c) A default value of five minutes or more for the NAT UDP mapping +timer is RECOMMENDED. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound +refresh behavior" of "True". +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The NAT mapping Refresh Direction MAY have a "NAT Inbound +refresh behavior" of "True". +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-7 A NAT device whose external IP interface can be configured +dynamically MUST either (1) Automatically ensure that its internal +network uses IP addresses that do not conflict with its external +network, or (2) Be able to translate and forward traffic between +all internal nodes and all external nodes whose IP addresses +numerically conflict with the internal network. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-8: If application transparency is most important, it is +RECOMMENDED that a NAT have "Endpoint-Independent Filtering" +behavior. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +If a more stringent filtering behavior is most +important, it is RECOMMENDED that a NAT have "Address-Dependent +Filtering" behavior. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The filtering behavior MAY be an option configurable by the +administrator of the NAT. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-9: A NAT MUST support "Hairpinning". +''' + +[[spec]] +level = "MUST" +quote = ''' +a) A NAT Hairpinning behavior MUST be "External source IP address +and port". +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-10: To eliminate interference with UNSAF NAT traversal +mechanisms and allow integrity protection of UDP communications, +NAT ALGs for UDP-based protocols SHOULD be turned off. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +a) If a NAT includes ALGs, it is RECOMMENDED that the NAT allow +the NAT administrator to enable or disable each ALG separately. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT +change the NAT translation (Section 4) or the Filtering +(Section 5) Behavior at any point in time, or under any particular +conditions. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the +NAT mapping. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +a) The NAT's default configuration SHOULD NOT filter ICMP messages +based on their source IP address. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +b) It is RECOMMENDED that a NAT support ICMP Destination +Unreachable messages. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-13 If the packet received on an internal IP address has DF=1, +the NAT MUST send back an ICMP message "Fragmentation needed and +DF set" to the host, as described in [RFC0792]. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) If the packet has DF=0, the NAT MUST fragment the packet and +SHOULD send the fragments in order. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-14: A NAT MUST support receiving in-order and out-of-order +fragments, so it MUST have "Received Fragment Out of Order" +behavior. +''' + +[[spec]] +level = "MUST" +quote = ''' +a) A NAT's out-of-order fragment processing mechanism MUST be +designed so that fragmentation-based DoS attacks do not +compromise the NAT's ability to process in-order and +unfragmented IP packets. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.1.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.1.toml new file mode 100644 index 0000000000..0710d03721 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.1.toml @@ -0,0 +1,16 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-4.1" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" +behavior of "Paired". +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.1.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.1.toml new file mode 100644 index 0000000000..f484a238bd --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.1.toml @@ -0,0 +1,25 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port +overloading". +''' + +[[spec]] +level = "SHOULD" +quote = ''' +a) If the host's source port was in the range 0-1023, it is +RECOMMENDED the NAT's source port be in the same range. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +If the +host's source port was in the range 1024-65535, it is +RECOMMENDED that the NAT's source port be in that range. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.2.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.2.toml new file mode 100644 index 0000000000..ef7bf9b0ca --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.2.2.toml @@ -0,0 +1,10 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-4.2.2" + + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-4: It is RECOMMENDED that a NAT have a "Port parity +preservation" behavior of "Yes". +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.3.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.3.toml new file mode 100644 index 0000000000..6ddf03449e --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.3.toml @@ -0,0 +1,46 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-4.3" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two +minutes, unless REQ-5a applies. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) For specific destination ports in the well-known port range +(ports 0-1023), a NAT MAY have shorter UDP mapping timers that +are specific to the IANA-registered application running over +that specific destination port. +''' + +[[spec]] +level = "MAY" +quote = ''' +b) The value of the NAT UDP mapping timer MAY be configurable. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +c) A default value of five minutes or more for the NAT UDP mapping +timer is RECOMMENDED. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound +refresh behavior" of "True". +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The NAT mapping Refresh Direction MAY have a "NAT Inbound +refresh behavior" of "True". +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.4.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.4.toml new file mode 100644 index 0000000000..a2f1d56905 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-4.4.toml @@ -0,0 +1,14 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-4.4" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-7: A NAT device whose external IP interface can be configured +dynamically MUST either (1) automatically ensure that its internal +network uses IP addresses that do not conflict with its external +network, or (2) be able to translate and forward traffic between +all internal nodes and all external nodes whose IP addresses +numerically conflict with the internal network. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-5.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-5.toml new file mode 100644 index 0000000000..0d45687d99 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-5.toml @@ -0,0 +1,26 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-5" + + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-8: If application transparency is most important, it is +RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" +behavior. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +If a more stringent filtering behavior is most +important, it is RECOMMENDED that a NAT have an "Address-Dependent +Filtering" behavior. +''' + +[[spec]] +level = "MAY" +quote = ''' +a) The filtering behavior MAY be an option configurable by the +administrator of the NAT. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-6.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-6.toml new file mode 100644 index 0000000000..4c2f32febb --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-6.toml @@ -0,0 +1,16 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-6" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-9: A NAT MUST support "Hairpinning". +''' + +[[spec]] +level = "MUST" +quote = ''' +a) A NAT Hairpinning behavior MUST be "External source IP address +and port". +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-7.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-7.toml new file mode 100644 index 0000000000..80957b772c --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-7.toml @@ -0,0 +1,18 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-7" + + +[[spec]] +level = "SHOULD" +quote = ''' +REQ-10: To eliminate interference with UNSAF NAT traversal +mechanisms and allow integrity protection of UDP communications, +NAT ALGs for UDP-based protocols SHOULD be turned off. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +a) If a NAT includes ALGs, it is RECOMMENDED that the NAT allow +the NAT administrator to enable or disable each ALG separately. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-8.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-8.toml new file mode 100644 index 0000000000..5d23c95960 --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-8.toml @@ -0,0 +1,12 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-8" + + +[[spec]] +level = "MUST" +quote = ''' +REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT +change the NAT translation (Section 4) or the Filtering +(Section 5) Behavior at any point in time, or under any particular +conditions. +''' + diff --git a/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-9.toml b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-9.toml new file mode 100644 index 0000000000..38193443da --- /dev/null +++ b/.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-9.toml @@ -0,0 +1,47 @@ +target = "https://www.rfc-editor.org/rfc/rfc4787#section-9" + + +[[spec]] +level = "SHOULD" +quote = ''' +The NAT's default configuration SHOULD NOT filter +ICMP messages based on their source IP address. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +Such ICMP messages +SHOULD be rewritten by the NAT (specifically, the IP headers and the +ICMP payload) and forwarded to the appropriate internal or external +host. +''' + +[[spec]] +level = "MUST" +quote = ''' +Receipt of any sort of ICMP message MUST NOT +destroy the NAT mapping. +''' + +[[spec]] +level = "MUST" +quote = ''' +REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the +NAT mapping. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +a) The NAT's default configuration SHOULD NOT filter ICMP messages +based on their source IP address. +''' + +[[spec]] +level = "SHOULD" +quote = ''' +b) It is RECOMMENDED that a NAT support ICMP Destination +Unreachable messages. +''' + diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index 3881a4c39b..6e05273b0f 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -1,3 +1,176 @@ +SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 + SECTION: [Address and Port Mapping](#section-4.1) + TEXT[!MUST,todo]: REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. + TEXT[!SHOULD]: REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" + TEXT[!SHOULD]: behavior of "Paired". + + SECTION: [Port Assignment Behavior](#section-4.2.1) + TEXT[!MUST,implementation,test]: REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port + TEXT[!MUST,implementation,test]: overloading". + TEXT[!SHOULD,exception]: a) If the host's source port was in the range 0-1023, it is + TEXT[!SHOULD,exception]: RECOMMENDED the NAT's source port be in the same range. + TEXT[!SHOULD]: If the + TEXT[!SHOULD]: host's source port was in the range 1024-65535, it is + TEXT[!SHOULD]: RECOMMENDED that the NAT's source port be in that range. + + SECTION: [Port Parity](#section-4.2.2) + TEXT[!SHOULD]: REQ-4: It is RECOMMENDED that a NAT have a "Port parity + TEXT[!SHOULD]: preservation" behavior of "Yes". + + SECTION: [Mapping Refresh](#section-4.3) + TEXT[!MUST,todo]: REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two + TEXT[!MUST,todo]: minutes, unless REQ-5a applies. + TEXT[!MAY,implementation]: a) For specific destination ports in the well-known port range + TEXT[!MAY,implementation]: (ports 0-1023), a NAT MAY have shorter UDP mapping timers that + TEXT[!MAY,implementation]: are specific to the IANA-registered application running over + TEXT[!MAY,implementation]: that specific destination port. + TEXT[!MAY]: b) The value of the NAT UDP mapping timer MAY be configurable. + TEXT[!SHOULD,todo]: c) A default value of five minutes or more for the NAT UDP mapping + TEXT[!SHOULD,todo]: timer is RECOMMENDED. + TEXT[!MUST]: REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + TEXT[!MUST]: refresh behavior" of "True". + TEXT[!MAY]: a) The NAT mapping Refresh Direction MAY have a "NAT Inbound + TEXT[!MAY]: refresh behavior" of "True". + + SECTION: [Conflicting Internal and External IP Address Spaces](#section-4.4) + TEXT[!MUST]: REQ-7: A NAT device whose external IP interface can be configured + TEXT[!MUST]: dynamically MUST either (1) automatically ensure that its internal + TEXT[!MUST]: network uses IP addresses that do not conflict with its external + TEXT[!MUST]: network, or (2) be able to translate and forward traffic between + TEXT[!MUST]: all internal nodes and all external nodes whose IP addresses + TEXT[!MUST]: numerically conflict with the internal network. + + SECTION: [Filtering Behavior](#section-5) + TEXT[!SHOULD]: REQ-8: If application transparency is most important, it is + TEXT[!SHOULD]: RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + TEXT[!SHOULD]: behavior. + TEXT[!SHOULD]: If a more stringent filtering behavior is most + TEXT[!SHOULD]: important, it is RECOMMENDED that a NAT have an "Address-Dependent + TEXT[!SHOULD]: Filtering" behavior. + TEXT[!MAY]: a) The filtering behavior MAY be an option configurable by the + TEXT[!MAY]: administrator of the NAT. + + SECTION: [Hairpinning Behavior](#section-6) + TEXT[!MUST,todo]: REQ-9: A NAT MUST support "Hairpinning". + TEXT[!MUST]: a) A NAT Hairpinning behavior MUST be "External source IP address + TEXT[!MUST]: and port". + + SECTION: [Application Level Gateways](#section-7) + TEXT[!SHOULD]: REQ-10: To eliminate interference with UNSAF NAT traversal + TEXT[!SHOULD]: mechanisms and allow integrity protection of UDP communications, + TEXT[!SHOULD]: NAT ALGs for UDP-based protocols SHOULD be turned off. + TEXT[!SHOULD]: a) If a NAT includes ALGs, it is RECOMMENDED that the NAT allow + TEXT[!SHOULD]: the NAT administrator to enable or disable each ALG separately. + + SECTION: [Deterministic Properties](#section-8) + TEXT[!MUST]: REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT + TEXT[!MUST]: change the NAT translation (Section 4) or the Filtering + TEXT[!MUST]: (Section 5) Behavior at any point in time, or under any particular + TEXT[!MUST]: conditions. + + SECTION: [ICMP Destination Unreachable Behavior](#section-9) + TEXT[!SHOULD]: The NAT's default configuration SHOULD NOT filter + TEXT[!SHOULD]: ICMP messages based on their source IP address. + TEXT[!SHOULD]: Such ICMP messages + TEXT[!SHOULD]: SHOULD be rewritten by the NAT (specifically, the IP headers and the + TEXT[!SHOULD]: ICMP payload) and forwarded to the appropriate internal or external + TEXT[!SHOULD]: host. + TEXT[!MUST]: Receipt of any sort of ICMP message MUST NOT + TEXT[!MUST]: destroy the NAT mapping. + TEXT[!MUST,implementation]: REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the + TEXT[!MUST,implementation]: NAT mapping. + TEXT[!SHOULD]: a) The NAT's default configuration SHOULD NOT filter ICMP messages + TEXT[!SHOULD]: based on their source IP address. + TEXT[!SHOULD]: b) It is RECOMMENDED that a NAT support ICMP Destination + TEXT[!SHOULD]: Unreachable messages. + + SECTION: [Fragmentation of Outgoing Packets](#section-10) + TEXT[!MUST]: REQ-13: If the packet received on an internal IP address has DF=1, + TEXT[!MUST]: the NAT MUST send back an ICMP message "Fragmentation needed and + TEXT[!MUST]: DF set" to the host, as described in [RFC0792]. + TEXT[!MUST]: a) If the packet has DF=0, the NAT MUST fragment the packet and + TEXT[!MUST]: SHOULD send the fragments in order. + + SECTION: [Receiving Fragmented Packets](#section-11) + TEXT[!MUST,todo]: REQ-14: A NAT MUST support receiving in-order and out-of-order + TEXT[!MUST,todo]: fragments, so it MUST have "Received Fragment Out of Order" + TEXT[!MUST,todo]: behavior. + TEXT[!MUST]: a) A NAT's out-of-order fragment processing mechanism MUST be + TEXT[!MUST]: designed so that fragmentation-based DoS attacks do not + TEXT[!MUST]: compromise the NAT's ability to process in-order and + TEXT[!MUST]: unfragmented IP packets. + + SECTION: [Requirements](#section-12) + TEXT[!MUST]: REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. + TEXT[!SHOULD]: REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" + TEXT[!SHOULD]: behavior of "Paired". + TEXT[!MUST]: REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port + TEXT[!MUST]: overloading". + TEXT[!SHOULD]: a) If the host's source port was in the range 0-1023, it is + TEXT[!SHOULD]: RECOMMENDED the NAT's source port be in the same range. + TEXT[!SHOULD]: If the + TEXT[!SHOULD]: host's source port was in the range 1024-65535, it is + TEXT[!SHOULD]: RECOMMENDED that the NAT's source port be in that range. + TEXT[!SHOULD]: REQ-4: It is RECOMMENDED that a NAT have a "Port parity + TEXT[!SHOULD]: preservation" behavior of "Yes". + TEXT[!MUST]: REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two + TEXT[!MUST]: minutes, unless REQ-5a applies. + TEXT[!MAY]: a) For specific destination ports in the well-known port range + TEXT[!MAY]: (ports 0-1023), a NAT MAY have shorter UDP mapping timers that + TEXT[!MAY]: are specific to the IANA-registered application running over + TEXT[!MAY]: that specific destination port. + TEXT[!MAY]: b) The value of the NAT UDP mapping timer MAY be configurable. + TEXT[!SHOULD]: c) A default value of five minutes or more for the NAT UDP mapping + TEXT[!SHOULD]: timer is RECOMMENDED. + TEXT[!MUST]: REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + TEXT[!MUST]: refresh behavior" of "True". + TEXT[!MAY]: a) The NAT mapping Refresh Direction MAY have a "NAT Inbound + TEXT[!MAY]: refresh behavior" of "True". + TEXT[!MUST]: REQ-7 A NAT device whose external IP interface can be configured + TEXT[!MUST]: dynamically MUST either (1) Automatically ensure that its internal + TEXT[!MUST]: network uses IP addresses that do not conflict with its external + TEXT[!MUST]: network, or (2) Be able to translate and forward traffic between + TEXT[!MUST]: all internal nodes and all external nodes whose IP addresses + TEXT[!MUST]: numerically conflict with the internal network. + TEXT[!SHOULD]: REQ-8: If application transparency is most important, it is + TEXT[!SHOULD]: RECOMMENDED that a NAT have "Endpoint-Independent Filtering" + TEXT[!SHOULD]: behavior. + TEXT[!SHOULD]: If a more stringent filtering behavior is most + TEXT[!SHOULD]: important, it is RECOMMENDED that a NAT have "Address-Dependent + TEXT[!SHOULD]: Filtering" behavior. + TEXT[!MAY]: a) The filtering behavior MAY be an option configurable by the + TEXT[!MAY]: administrator of the NAT. + TEXT[!MUST]: REQ-9: A NAT MUST support "Hairpinning". + TEXT[!MUST]: a) A NAT Hairpinning behavior MUST be "External source IP address + TEXT[!MUST]: and port". + TEXT[!SHOULD]: REQ-10: To eliminate interference with UNSAF NAT traversal + TEXT[!SHOULD]: mechanisms and allow integrity protection of UDP communications, + TEXT[!SHOULD]: NAT ALGs for UDP-based protocols SHOULD be turned off. + TEXT[!SHOULD]: a) If a NAT includes ALGs, it is RECOMMENDED that the NAT allow + TEXT[!SHOULD]: the NAT administrator to enable or disable each ALG separately. + TEXT[!MUST]: REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT + TEXT[!MUST]: change the NAT translation (Section 4) or the Filtering + TEXT[!MUST]: (Section 5) Behavior at any point in time, or under any particular + TEXT[!MUST]: conditions. + TEXT[!MUST]: REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the + TEXT[!MUST]: NAT mapping. + TEXT[!SHOULD]: a) The NAT's default configuration SHOULD NOT filter ICMP messages + TEXT[!SHOULD]: based on their source IP address. + TEXT[!SHOULD]: b) It is RECOMMENDED that a NAT support ICMP Destination + TEXT[!SHOULD]: Unreachable messages. + TEXT[!MUST]: REQ-13 If the packet received on an internal IP address has DF=1, + TEXT[!MUST]: the NAT MUST send back an ICMP message "Fragmentation needed and + TEXT[!MUST]: DF set" to the host, as described in [RFC0792]. + TEXT[!MUST]: a) If the packet has DF=0, the NAT MUST fragment the packet and + TEXT[!MUST]: SHOULD send the fragments in order. + TEXT[!MUST]: REQ-14: A NAT MUST support receiving in-order and out-of-order + TEXT[!MUST]: fragments, so it MUST have "Received Fragment Out of Order" + TEXT[!MUST]: behavior. + TEXT[!MUST]: a) A NAT's out-of-order fragment processing mechanism MUST be + TEXT[!MUST]: designed so that fragmentation-based DoS attacks do not + TEXT[!MUST]: compromise the NAT's ability to process in-order and + TEXT[!MUST]: unfragmented IP packets. + SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4884 SECTION: [Summary of Changes to ICMP](#section-3) TEXT[!MAY]: An ICMP Extension Structure MAY be appended to ICMPv4 Destination diff --git a/.duvet/specifications/www.rfc-editor.org/rfc/rfc4787.txt b/.duvet/specifications/www.rfc-editor.org/rfc/rfc4787.txt new file mode 100644 index 0000000000..00219c79bf --- /dev/null +++ b/.duvet/specifications/www.rfc-editor.org/rfc/rfc4787.txt @@ -0,0 +1,1627 @@ + + + + + + +Network Working Group F. Audet, Ed. +Request for Comments: 4787 Nortel Networks +BCP: 127 C. Jennings +Category: Best Current Practice Cisco Systems + January 2007 + + + Network Address Translation (NAT) Behavioral Requirements + for Unicast UDP + +Status of This Memo + + This document specifies an Internet Best Current Practices for the + Internet Community, and requests discussion and suggestions for + improvements. Distribution of this memo is unlimited. + +Copyright Notice + + Copyright (C) The IETF Trust (2007). + +Abstract + + This document defines basic terminology for describing different + types of Network Address Translation (NAT) behavior when handling + Unicast UDP and also defines a set of requirements that would allow + many applications, such as multimedia communications or online + gaming, to work consistently. Developing NATs that meet this set of + requirements will greatly increase the likelihood that these + applications will function properly. + + + + + + + + + + + + + + + + + + + + + + +Audet & Jennings Best Current Practice [Page 1] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + +Table of Contents + + 1. Applicability Statement . . . . . . . . . . . . . . . . . . . 3 + 2. Introduction . . . . . . . . . . . . . . . . . . . . . . . . . 3 + 3. Terminology . . . . . . . . . . . . . . . . . . . . . . . . . 4 + 4. Network Address and Port Translation Behavior . . . . . . . . 5 + 4.1. Address and Port Mapping . . . . . . . . . . . . . . . . . 5 + 4.2. Port Assignment . . . . . . . . . . . . . . . . . . . . . 9 + 4.2.1. Port Assignment Behavior . . . . . . . . . . . . . . . 9 + 4.2.2. Port Parity . . . . . . . . . . . . . . . . . . . . . 11 + 4.2.3. Port Contiguity . . . . . . . . . . . . . . . . . . . 11 + 4.3. Mapping Refresh . . . . . . . . . . . . . . . . . . . . . 12 + 4.4. Conflicting Internal and External IP Address Spaces . . . 13 + 5. Filtering Behavior . . . . . . . . . . . . . . . . . . . . . . 15 + 6. Hairpinning Behavior . . . . . . . . . . . . . . . . . . . . . 16 + 7. Application Level Gateways . . . . . . . . . . . . . . . . . . 17 + 8. Deterministic Properties . . . . . . . . . . . . . . . . . . . 18 + 9. ICMP Destination Unreachable Behavior . . . . . . . . . . . . 19 + 10. Fragmentation of Outgoing Packets . . . . . . . . . . . . . . 20 + 11. Receiving Fragmented Packets . . . . . . . . . . . . . . . . . 20 + 12. Requirements . . . . . . . . . . . . . . . . . . . . . . . . . 21 + 13. Security Considerations . . . . . . . . . . . . . . . . . . . 24 + 14. IAB Considerations . . . . . . . . . . . . . . . . . . . . . . 25 + 15. Acknowledgments . . . . . . . . . . . . . . . . . . . . . . . 26 + 16. References . . . . . . . . . . . . . . . . . . . . . . . . . . 26 + 16.1. Normative References . . . . . . . . . . . . . . . . . . . 26 + 16.2. Informative References . . . . . . . . . . . . . . . . . . 26 + + + + + + + + + + + + + + + + + + + + + + + + +Audet & Jennings Best Current Practice [Page 2] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + +1. Applicability Statement + + The purpose of this specification is to define a set of requirements + for NATs that would allow many applications, such as multimedia + communications or online gaming, to work consistently. Developing + NATs that meet this set of requirements will greatly increase the + likelihood that these applications will function properly. + + The requirements of this specification apply to Traditional NATs as + described in [RFC2663]. + + This document is meant to cover NATs of any size, from small + residential NATs to large Enterprise NATs. However, it should be + understood that Enterprise NATs normally provide much more than just + NAT capabilities; for example, they typically provide firewall + functionalities. A comprehensive description of firewall behaviors + and associated requirements is specifically out-of-scope for this + specification. However, this specification does cover basic firewall + aspects present in NATs (see Section 5). + + Approaches using directly signaled control of middle boxes are out of + scope. + + UDP Relays (e.g., Traversal Using Relay NAT [TURN]) are out of scope. + + Application aspects are out of scope, as the focus here is strictly + on the NAT itself. + + This document only covers aspects of NAT traversal related to Unicast + UDP [RFC0768] over IP [RFC0791] and their dependencies on other + protocols. + +2. Introduction + + Network Address Translators (NATs) are well known to cause very + significant problems with applications that carry IP addresses in the + payload (see [RFC3027]). Applications that suffer from this problem + include Voice Over IP and Multimedia Over IP (e.g., SIP [RFC3261] and + H.323 [ITU.H323]), as well as online gaming. + + Many techniques are used to attempt to make realtime multimedia + applications, online games, and other applications work across NATs. + Application Level Gateways [RFC2663] are one such mechanism. STUN + [RFC3489bis] describes a UNilateral Self-Address Fixing (UNSAF) + mechanism [RFC3424]. Teredo [RFC4380] describes an UNSAF mechanism + consisting of tunnelling IPv6 [RFC2460] over UDP/IPv4. UDP Relays + have also been used to enable applications across NATs, but these are + generally seen as a solution of last resort. Interactive + + + +Audet & Jennings Best Current Practice [Page 3] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + Connectivity Establishment [ICE] describes a methodology for using + many of these techniques and avoiding a UDP relay, unless the type of + NAT is such that it forces the use of such a UDP relay. This + specification defines requirements for improving NATs. Meeting these + requirements ensures that applications will not be forced to use UDP + relay. + + As pointed out in UNSAF [RFC3424], "From observations of deployed + networks, it is clear that different NAT box implementations vary + widely in terms of how they handle different traffic and addressing + cases". This wide degree of variability is one factor in the overall + brittleness introduced by NATs and makes it extremely difficult to + predict how any given protocol will behave on a network traversing + NAT. Discussions with many of the major NAT vendors have made it + clear that they would prefer to deploy NATs that were deterministic + and caused the least harm to applications while still meeting the + requirements that caused their customers to deploy NATs in the first + place. The problem NAT vendors face is that they are not sure how + best to do that or how to document their NATs' behavior. + + The goals of this document are to define a set of common terminology + for describing the behavior of NATs and to produce a set of + requirements on a specific set of behaviors for NATs. + + This document forms a common set of requirements that are simple and + useful for voice, video, and games, which can be implemented by NAT + vendors. This document will simplify the analysis of protocols for + deciding whether or not they work in this environment and will allow + providers of services that have NAT traversal issues to make + statements about where their applications will work and where they + will not, as well as to specify their own NAT requirements. + +3. Terminology + + The key words "MUST", "MUST NOT", "REQUIRED", "SHALL", "SHALL NOT", + "SHOULD", "SHOULD NOT", "RECOMMENDED", "MAY", and "OPTIONAL" in this + document are to be interpreted as described in [RFC2119]. + + Readers are urged to refer to [RFC2663] for information on NAT + taxonomy and terminology. Traditional NAT is the most common type of + NAT device deployed. Readers may refer to [RFC3022] for detailed + information on traditional NAT. Traditional NAT has two main + varieties -- Basic NAT and Network Address/Port Translator (NAPT). + + NAPT is by far the most commonly deployed NAT device. NAPT allows + multiple internal hosts to share a single public IP address + simultaneously. When an internal host opens an outgoing TCP or UDP + session through a NAPT, the NAPT assigns the session a public IP + + + +Audet & Jennings Best Current Practice [Page 4] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + address and port number, so that subsequent response packets from the + external endpoint can be received by the NAPT, translated, and + forwarded to the internal host. The effect is that the NAPT + establishes a NAT session to translate the (private IP address, + private port number) tuple to a (public IP address, public port + number) tuple, and vice versa, for the duration of the session. An + issue of relevance to peer-to-peer applications is how the NAT + behaves when an internal host initiates multiple simultaneous + sessions from a single (private IP, private port) endpoint to + multiple distinct endpoints on the external network. In this + specification, the term "NAT" refers to both "Basic NAT" and "Network + Address/Port Translator (NAPT)". + + This document uses the term "session" as defined in RFC 2663: "TCP/ + UDP sessions are uniquely identified by the tuple of (source IP + address, source TCP/UDP ports, target IP address, target TCP/UDP + Port)". + + This document uses the term "address and port mapping" as the + translation between an external address and port and an internal + address and port. Note that this is not the same as an "address + binding" as defined in RFC 2663. + + This document uses IANA terminology for port ranges, i.e., "Well + Known Ports" is 0-1023, "Registered" is 1024-49151, and "Dynamic + and/or Private" is 49152-65535, as defined in + http://www.iana.org/assignments/port-numbers. + + STUN [RFC3489] used the terms "Full Cone", "Restricted Cone", "Port + Restricted Cone", and "Symmetric" to refer to different variations of + NATs applicable to UDP only. Unfortunately, this terminology has + been the source of much confusion, as it has proven inadequate at + describing real-life NAT behavior. This specification therefore + refers to specific individual NAT behaviors instead of using the + Cone/Symmetric terminology. + +4. Network Address and Port Translation Behavior + + This section describes the various NAT behaviors applicable to NATs. + +4.1. Address and Port Mapping + + When an internal endpoint opens an outgoing session through a NAT, + the NAT assigns the session an external IP address and port number so + that subsequent response packets from the external endpoint can be + received by the NAT, translated, and forwarded to the internal + endpoint. This is a mapping between an internal IP address and port + IP:port and external IP:port tuple. It establishes the translation + + + +Audet & Jennings Best Current Practice [Page 5] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + that will be performed by the NAT for the duration of the session. + For many applications, it is important to distinguish the behavior of + the NAT when there are multiple simultaneous sessions established to + different external endpoints. + + The key behavior to describe is the criteria for reuse of a mapping + for new sessions to external endpoints, after establishing a first + mapping between an internal X:x address and port and an external + Y1:y1 address tuple. Let's assume that the internal IP address and + port X:x are mapped to X1':x1' for this first session. The endpoint + then sends from X:x to an external address Y2:y2 and gets a mapping + of X2':x2' on the NAT. The relationship between X1':x1' and X2':x2' + for various combinations of the relationship between Y1:y1 and Y2:y2 + is critical for describing the NAT behavior. This arrangement is + illustrated in the following diagram: + + E + +------+ +------+ x + | Y1 | | Y2 | t + +--+---+ +---+--+ e + | Y1:y1 Y2:y2 | r + +----------+ +----------+ n + | | a + X1':x1' | | X2':x2' l + +--+---+-+ + ...........| NAT |............... + +--+---+-+ I + | | n + X:x | | X:x t + ++---++ e + | X | r + +-----+ n + a + l + + Address and Port Mapping + + The following address and port mapping behavior are defined: + + Endpoint-Independent Mapping: + + The NAT reuses the port mapping for subsequent packets sent + from the same internal IP address and port (X:x) to any + external IP address and port. Specifically, X1':x1' equals + X2':x2' for all values of Y2:y2. + + + + + + +Audet & Jennings Best Current Practice [Page 6] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + Address-Dependent Mapping: + + The NAT reuses the port mapping for subsequent packets sent + from the same internal IP address and port (X:x) to the same + external IP address, regardless of the external port. + Specifically, X1':x1' equals X2':x2' if and only if, Y2 equals + Y1. + + Address and Port-Dependent Mapping: + + The NAT reuses the port mapping for subsequent packets sent + from the same internal IP address and port (X:x) to the same + external IP address and port while the mapping is still active. + Specifically, X1':x1' equals X2':x2' if and only if, Y2:y2 + equals Y1:y1. + + It is important to note that these three possible choices make no + difference to the security properties of the NAT. The security + properties are fully determined by which packets the NAT allows in + and which it does not. This is determined by the filtering behavior + in the filtering portions of the NAT. + + REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. + + Justification: In order for UNSAF methods to work, REQ-1 needs to be + met. Failure to meet REQ-1 will force the use of a UDP relay, + which is very often impractical. + + Some NATs are capable of assigning IP addresses from a pool of IP + addresses on the external side of the NAT, as opposed to just a + single IP address. This is especially common with larger NATs. Some + NATs use the external IP address mapping in an arbitrary fashion + (i.e., randomly): one internal IP address could have multiple + external IP address mappings active at the same time for different + sessions. These NATs have an "IP address pooling" behavior of + "Arbitrary". Some large Enterprise NATs use an IP address pooling + behavior of "Arbitrary" as a means of hiding the IP address assigned + to specific endpoints by making their assignment less predictable. + Other NATs use the same external IP address mapping for all sessions + associated with the same internal IP address. These NATs have an "IP + address pooling" behavior of "Paired". NATs that use an "IP address + pooling" behavior of "Arbitrary" can cause issues for applications + that use multiple ports from the same endpoint, but that do not + negotiate IP addresses individually (e.g., some applications using + RTP and RTCP). + + + + + + +Audet & Jennings Best Current Practice [Page 7] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" + behavior of "Paired". Note that this requirement is not + applicable to NATs that do not support IP address pooling. + + Justification: This will allow applications that use multiple ports + originating from the same internal IP address to also have the + same external IP address. This is to avoid breaking peer-to-peer + applications that are not capable of negotiating the IP address + for RTP and the IP address for RTCP separately. As such it is + envisioned that this requirement will become less important as + applications become NAT-friendlier with time. The main reason why + this requirement is here is that in a peer-to-peer application, + you are subject to the other peer's mistake. In particular, in + the context of SIP, if my application supports the extensions + defined in [RFC3605] for indicating RTP and RTCP addresses and + ports separately, but the other peer does not, there may still be + breakage in the form of the stream losing RTCP packets. This + requirement will avoid the loss of RTP in this context, although + the loss of RTCP may be inevitable in this particular example. It + is also worth noting that RFC 3605 is unfortunately not a + mandatory part of SIP [RFC3261]. Therefore, this requirement will + address a particularly nasty problem that will prevail for a + significant period of time. + + + + + + + + + + + + + + + + + + + + + + + + + + + + +Audet & Jennings Best Current Practice [Page 8] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + +4.2. Port Assignment + +4.2.1. Port Assignment Behavior + + This section uses the following diagram for reference. + + E + +-------+ +-------+ x + | Y1 | | Y2 | t + +---+---+ +---+---+ e + | Y1:y1 Y2:y2 | r + +---------+ +---------+ n + | | a + X1':x1' | | X2':x2' l + +--+---+--+ + ...........| NAT |............... + +--+---+--+ I + | | n + +---------+ +---------+ t + | X1:x1 X2:x2 | e + +---+---+ +---+---+ r + | X1 | | X2 | n + +-------+ +-------+ a + l + + Port Assignment + + Some NATs attempt to preserve the port number used internally when + assigning a mapping to an external IP address and port (e.g., x1=x1', + x2=x2'). This port assignment behavior is referred to as "port + preservation". In case of port collision, these NATs attempt a + variety of techniques for coping. For example, some NATs will + overridden the previous mapping to preserve the same port. Other + NATs will assign a different IP address from a pool of external IP + addresses; this is only possible as long as the NAT has enough + external IP addresses; if the port is already in use on all available + external IP addresses, then these NATs will pick a different port + (i.e., they don't do port preservation anymore). + + Some NATs use "Port overloading", i.e., they always use port + preservation even in the case of collision (i.e., X1'=X2' and + x1=x2=x1'=x2'). Most applications will fail if the NAT uses "Port + overloading". + + A NAT that does not attempt to make the external port numbers match + the internal port numbers in any case is referred to as "no port + preservation". + + + + +Audet & Jennings Best Current Practice [Page 9] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + When NATs do allocate a new source port, there is the issue of which + IANA-defined range of port to choose. The ranges are "well-known" + from 0 to 1023, "registered" from 1024 to 49151, and "dynamic/ + private" from 49152 through 65535. For most protocols, these are + destination ports and not source ports, so mapping a source port to a + source port that is already registered is unlikely to have any bad + effects. Some NATs may choose to use only the ports in the dynamic + range; the only downside of this practice is that it limits the + number of ports available. Other NAT devices may use everything but + the well-known range and may prefer to use the dynamic range first, + or possibly avoid the actual registered ports in the registered + range. Other NATs preserve the port range if it is in the well-known + range. [RFC0768] specifies that the source port is set to zero if no + reply packets are expected. In this case, it does not matter what + the NAT maps it to, as the source port will not be used. However, + many common OS APIs do not allow a user to send from port zero, + applications do not use port zero, and the behavior of various + existing NATs with regards to a packet with a source of port zero is + unknown. This document does not specify any normative behavior for a + NAT when handling a packet with a source port of zero which means + that applications cannot count on any sort of deterministic behavior + for these packets. + + REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port + overloading". + + a) If the host's source port was in the range 0-1023, it is + RECOMMENDED the NAT's source port be in the same range. If the + host's source port was in the range 1024-65535, it is + RECOMMENDED that the NAT's source port be in that range. + + Justification: This requirement must be met in order to enable two + applications on the internal side of the NAT both to use the same + port to try to communicate with the same destination. NATs that + implement port preservation have to deal with conflicts on ports, + and the multiple code paths this introduces often result in + nondeterministic behavior. However, it should be understood that + when a port is randomly assigned, it may just randomly happen to + be assigned the same port. Applications must, therefore, be able + to deal with both port preservation and no port preservation. + + a) Certain applications expect the source UDP port to be in the + well-known range. See the discussion of Network File System + port expectations in [RFC2623] for an example. + + + + + + + +Audet & Jennings Best Current Practice [Page 10] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + +4.2.2. Port Parity + + Some NATs preserve the parity of the UDP port, i.e., an even port + will be mapped to an even port, and an odd port will be mapped to an + odd port. This behavior respects the [RFC3550] rule that RTP use + even ports, and RTCP use odd ports. RFC 3550 allows any port numbers + to be used for RTP and RTCP if the two numbers are specified + separately; for example, using [RFC3605]. However, some + implementations do not include RFC 3605, and do not recognize when + the peer has specified the RTCP port separately using RFC 3605. If + such an implementation receives an odd RTP port number from the peer + (perhaps after having been translated by a NAT), and then follows the + RFC 3550 rule to change the RTP port to the next lower even number, + this would obviously result in the loss of RTP. NAT-friendly + application aspects are outside the scope of this document. It is + expected that this issue will fade away with time, as implementations + improve. Preserving the port parity allows for supporting + communication with peers that do not support explicit specification + of both RTP and RTCP port numbers. + + REQ-4: It is RECOMMENDED that a NAT have a "Port parity + preservation" behavior of "Yes". + + Justification: This is to avoid breaking peer-to-peer applications + that do not explicitly and separately specify RTP and RTCP port + numbers and that follow the RFC 3550 rule to decrement an odd RTP + port to make it even. The same considerations apply, as per the + IP address pooling requirement. + +4.2.3. Port Contiguity + + Some NATs attempt to preserve the port contiguity rule of RTCP=RTP+1. + These NATs do things like sequential assignment or port reservation. + Sequential port assignment assumes that the application will open a + mapping for RTP first and then open a mapping for RTCP. It is not + practical to enforce this requirement on all applications. + Furthermore, there is a problem with glare if many applications (or + endpoints) are trying to open mappings simultaneously. Port + preservation is also problematic since it is wasteful, especially + considering that a NAT cannot reliably distinguish between RTP over + UDP and other UDP packets where there is no contiguity rule. For + those reasons, it would be too complex to attempt to preserve the + contiguity rule by suggesting specific NAT behavior, and it would + certainly break the deterministic behavior rule. + + In order to support both RTP and RTCP, it will therefore be necessary + that applications follow rules to negotiate RTP and RTCP separately, + and account for the very real possibility that the RTCP=RTP+1 rule + + + +Audet & Jennings Best Current Practice [Page 11] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + will be broken. As this is an application requirement, it is outside + the scope of this document. + +4.3. Mapping Refresh + + NAT mapping timeout implementations vary, but include the timer's + value and the way the mapping timer is refreshed to keep the mapping + alive. + + The mapping timer is defined as the time a mapping will stay active + without packets traversing the NAT. There is great variation in the + values used by different NATs. + + REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two + minutes, unless REQ-5a applies. + + a) For specific destination ports in the well-known port range + (ports 0-1023), a NAT MAY have shorter UDP mapping timers that + are specific to the IANA-registered application running over + that specific destination port. + + b) The value of the NAT UDP mapping timer MAY be configurable. + + c) A default value of five minutes or more for the NAT UDP mapping + timer is RECOMMENDED. + + Justification: This requirement is to ensure that the timeout is + long enough to avoid too-frequent timer refresh packets. + + a) Some UDP protocols using UDP use very short-lived connections. + There can be very many such connections; keeping them all in a + connections table could cause considerable load on the NAT. + Having shorter timers for these specific applications is, + therefore, an optimization technique. It is important that the + shorter timers applied to specific protocols be used sparingly, + and only for protocols using well-known destination ports that + are known to have a shorter timer, and that are known not to be + used by any applications for other purposes. + + b) Configuration is desirable for adapting to specific networks + and troubleshooting. + + c) This default is to avoid too-frequent timer refresh packets. + + Some NATs keep the mapping active (i.e., refresh the timer value) + when a packet goes from the internal side of the NAT to the external + side of the NAT. This is referred to as having a NAT Outbound + refresh behavior of "True". + + + +Audet & Jennings Best Current Practice [Page 12] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + Some NATs keep the mapping active when a packet goes from the + external side of the NAT to the internal side of the NAT. This is + referred to as having a NAT Inbound Refresh Behavior of "True". + + Some NATs keep the mapping active on both, in which case, both + properties are "True". + + REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + refresh behavior" of "True". + + a) The NAT mapping Refresh Direction MAY have a "NAT Inbound + refresh behavior" of "True". + + Justification: Outbound refresh is necessary for allowing the client + to keep the mapping alive. + + a) Inbound refresh may be useful for applications with no outgoing + UDP traffic. However, allowing inbound refresh may allow an + external attacker or misbehaving application to keep a mapping + alive indefinitely. This may be a security risk. Also, if the + process is repeated with different ports, over time, it could + use up all the ports on the NAT. + +4.4. Conflicting Internal and External IP Address Spaces + + Many NATs, particularly consumer-level devices designed to be + deployed by nontechnical users, routinely obtain their external IP + address, default router, and other IP configuration information for + their external interface dynamically from an external network, such + as an upstream ISP. The NAT, in turn, automatically sets up its own + internal subnet in one of the private IP address spaces assigned to + this purpose in [RFC1918], typically providing dynamic IP + configuration services for hosts on this internal network. + + Auto-configuration of NATs and private networks can be problematic, + however, if the NAT's external network is also in RFC 1918 private + address space. In a common scenario, an ISP places its customers + behind a NAT and hands out private RFC 1918 addresses to them. Some + of these customers, in turn, deploy consumer-level NATs, which, in + effect, act as "second-level" NATs, multiplexing their own private + RFC 1918 IP subnets onto the single RFC 1918 IP address provided by + the ISP. There is no inherent guarantee, in this case, that the + ISP's "intermediate" privately-addressed network and the customer's + internal privately-addressed network will not use numerically + identical or overlapping RFC 1918 IP subnets. Furthermore, customers + of consumer-level NATs cannot be expected to have the technical + + + + + +Audet & Jennings Best Current Practice [Page 13] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + knowledge to prevent this scenario from occurring by manually + configuring their internal network with non-conflicting RFC 1918 + subnets. + + NAT vendors need to design their NATs to ensure that they function + correctly and robustly even in such problematic scenarios. One + possible solution is for the NAT to ensure that whenever its external + link is configured with an RFC 1918 private IP address, the NAT + automatically selects a different, non-conflicting RFC 1918 IP subnet + for its internal network. A disadvantage of this solution is that, + if the NAT's external interface is dynamically configured or re- + configured after its internal network is already in use, then the NAT + may have to renumber its entire internal network dynamically if it + detects a conflict. + + An alternative solution is for the NAT to be designed so that it can + translate and forward traffic correctly, even when its external and + internal interfaces are configured with numerically overlapping IP + subnets. In this scenario, for example, if the NAT's external + interface has been assigned an IP address P in RFC 1918 space, then + there might also be an internal node I having the same RFC 1918 + private IP address P. An IP packet with destination address P on the + external network is directed at the NAT, whereas an IP packet with + the same destination address P on the internal network is directed at + node I. The NAT therefore needs to maintain a clear operational + distinction between "external IP addresses" and "internal IP + addresses" to avoid confusing internal node I with its own external + interface. In general, the NAT needs to allow all internal nodes + (including I) to communicate with all external nodes having public + (non-RFC 1918) IP addresses, or having private IP addresses that do + not conflict with the addresses used by its internal network. + + REQ-7: A NAT device whose external IP interface can be configured + dynamically MUST either (1) automatically ensure that its internal + network uses IP addresses that do not conflict with its external + network, or (2) be able to translate and forward traffic between + all internal nodes and all external nodes whose IP addresses + numerically conflict with the internal network. + + Justification: If a NAT's external and internal interfaces are + configured with overlapping IP subnets, then there is, of course, + no way for an internal host with RFC 1918 IP address Q to initiate + a direct communication session to an external node having the same + RFC 1918 address Q, or to other external nodes with IP addresses + that numerically conflict with the internal subnet. Such nodes + can still open communication sessions indirectly via NAT traversal + techniques, however, with the help of a third-party server, such + as a STUN server having a public, non-RFC 1918 IP address. In + + + +Audet & Jennings Best Current Practice [Page 14] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + this case, nodes with conflicting private RFC 1918 addresses on + opposite sides of the second-level NAT can communicate with each + other via their respective temporary public endpoints on the main + Internet, as long as their common, first-level NAT (e.g., the + upstream ISP's NAT) supports hairpinning behavior, as described in + Section 6. + +5. Filtering Behavior + + This section describes various filtering behaviors observed in NATs. + + When an internal endpoint opens an outgoing session through a NAT, + the NAT assigns a filtering rule for the mapping between an internal + IP:port (X:x) and external IP:port (Y:y) tuple. + + The key behavior to describe is what criteria are used by the NAT to + filter packets originating from specific external endpoints. + + Endpoint-Independent Filtering: + + The NAT filters out only packets not destined to the internal + address and port X:x, regardless of the external IP address and + port source (Z:z). The NAT forwards any packets destined to + X:x. In other words, sending packets from the internal side of + the NAT to any external IP address is sufficient to allow any + packets back to the internal endpoint. + + Address-Dependent Filtering: + + The NAT filters out packets not destined to the internal + address X:x. Additionally, the NAT will filter out packets + from Y:y destined for the internal endpoint X:x if X:x has not + sent packets to Y:any previously (independently of the port + used by Y). In other words, for receiving packets from a + specific external endpoint, it is necessary for the internal + endpoint to send packets first to that specific external + endpoint's IP address. + + Address and Port-Dependent Filtering: + + This is similar to the previous behavior, except that the + external port is also relevant. The NAT filters out packets + not destined for the internal address X:x. Additionally, the + NAT will filter out packets from Y:y destined for the internal + endpoint X:x if X:x has not sent packets to Y:y previously. In + other words, for receiving packets from a specific external + endpoint, it is necessary for the internal endpoint to send + packets first to that external endpoint's IP address and port. + + + +Audet & Jennings Best Current Practice [Page 15] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + REQ-8: If application transparency is most important, it is + RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + behavior. If a more stringent filtering behavior is most + important, it is RECOMMENDED that a NAT have an "Address-Dependent + Filtering" behavior. + + a) The filtering behavior MAY be an option configurable by the + administrator of the NAT. + + Justification: The recommendation to use Endpoint-Independent + Filtering is aimed at maximizing application transparency; in + particular, for applications that receive media simultaneously + from multiple locations (e.g., gaming), or applications that use + rendezvous techniques. However, it is also possible that, in some + circumstances, it may be preferable to have a more stringent + filtering behavior. Filtering independently of the external + endpoint is not as secure: An unauthorized packet could get + through a specific port while the port was kept open if it was + lucky enough to find the port open. In theory, filtering based on + both IP address and port is more secure than filtering based only + on the IP address (because the external endpoint could, in + reality, be two endpoints behind another NAT, where one of the two + endpoints is an attacker). However, such a policy could interfere + with applications that expect to receive UDP packets on more than + one UDP port. Using Endpoint-Independent Filtering or Address- + Dependent Filtering instead of Address and Port-Dependent + Filtering on a NAT (say, NAT-A) also has benefits when the other + endpoint is behind a non-BEHAVE compliant NAT (say, NAT-B) that + does not support REQ-1. When the endpoints use ICE, if NAT-A uses + Address and Port-Dependent Filtering, connectivity will require a + UDP relay. However, if NAT-A uses Endpoint-Independent Filtering + or Address-Dependent Filtering, ICE will ultimately find + connectivity without requiring a UDP relay. Having the filtering + behavior being an option configurable by the administrator of the + NAT ensures that a NAT can be used in the widest variety of + deployment scenarios. + +6. Hairpinning Behavior + + If two hosts (called X1 and X2) are behind the same NAT and + exchanging traffic, the NAT may allocate an address on the outside of + the NAT for X2, called X2':x2'. If X1 sends traffic to X2':x2', it + goes to the NAT, which must relay the traffic from X1 to X2. This is + referred to as hairpinning and is illustrated below. + + + + + + + +Audet & Jennings Best Current Practice [Page 16] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + NAT + +----+ from X1:x1 to X2':x2' +-----+ X1':x1' + | X1 |>>>>>>>>>>>>>>>>>>>>>>>>>>>>>--+--- + +----+ | v | + | v | + | v | + | v | + +----+ from X1':x1' to X2:x2 | v | X2':x2' + | X2 |<<<<<<<<<<<<<<<<<<<<<<<<<<<<<--+--- + +----+ +-----+ + + Hairpinning Behavior + + Hairpinning allows two endpoints on the internal side of the NAT to + communicate even if they only use each other's external IP addresses + and ports. + + More formally, a NAT that supports hairpinning forwards packets + originating from an internal address, X1:x1, destined for an external + address X2':x2' that has an active mapping to an internal address + X2:x2, back to that internal address, X2:x2. Note that typically X1' + is the same as X2'. + + Furthermore, the NAT may present the hairpinned packet with either an + internal (X1:x1) or an external (X1':x1') source IP address and port. + Therefore, the hairpinning NAT behavior can be either "External + source IP address and port" or "Internal source IP address and port". + "Internal source IP address and port" may cause problems by confusing + implementations that expect an external IP address and port. + + REQ-9: A NAT MUST support "Hairpinning". + + a) A NAT Hairpinning behavior MUST be "External source IP address + and port". + + Justification: This requirement is to allow communications between + two endpoints behind the same NAT when they are trying each + other's external IP addresses. + + a) Using the external source IP address is necessary for + applications with a restrictive policy of not accepting packets + from IP addresses that differ from what is expected. + +7. Application Level Gateways + + Certain NATs have implemented Application Level Gateways (ALGs) for + various protocols, including protocols for negotiating peer-to-peer + sessions, such as SIP. + + + +Audet & Jennings Best Current Practice [Page 17] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + Certain NATs have these ALGs turned on permanently, others have them + turned on by default but allow them to be turned off, and others have + them turned off by default but allow them be turned on. + + NAT ALGs may interfere with UNSAF methods or protocols that try to be + NAT-aware and therefore must be used with extreme caution. + + REQ-10: To eliminate interference with UNSAF NAT traversal + mechanisms and allow integrity protection of UDP communications, + NAT ALGs for UDP-based protocols SHOULD be turned off. Future + standards track specifications that define ALGs can update this to + recommend the defaults for the ALGs that they define. + + a) If a NAT includes ALGs, it is RECOMMENDED that the NAT allow + the NAT administrator to enable or disable each ALG separately. + + Justification: NAT ALGs may interfere with UNSAF methods. + + a) This requirement allows the user to enable those ALGs that are + necessary to aid in the operation of some applications without + enabling ALGs, which interfere with the operation of other + applications. + +8. Deterministic Properties + + The classification of NATs is further complicated by the fact that, + under some conditions, the same NAT will exhibit different behaviors. + This has been seen on NATs that preserve ports or have specific + algorithms for selecting a port other than a free one. If the + external port that the NAT wishes to use is already in use by another + session, the NAT must select a different port. This results in + different code paths for this conflict case, which results in + different behavior. + + For example, if three hosts X1, X2, and X3 all send from the same + port x, through a port preserving NAT with only one external IP + address, called X1', the first one to send (i.e., X1) will get an + external port of x, but the next two will get x2' and x3' (where + these are not equal to x). There are NATs where the External NAT + mapping characteristics and the External Filter characteristics + change between the X1:x and the X2:x mapping. To make matters worse, + there are NATs where the behavior may be the same on the X1:x and + X2:x mappings, but different on the third X3:x mapping. + + Another example is that some NATs have an "Endpoint-Independent + Mapping", combined with "Port Overloading", as long as two endpoints + are not establishing sessions to the same external direction, but + then switch their behavior to "Address and Port-Dependent Mapping" + + + +Audet & Jennings Best Current Practice [Page 18] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + without "Port Preservation" upon detection of these conflicting + sessions establishments. + + Any NAT that changes the NAT Mapping or the Filtering behavior + without configuration changes, at any point in time, under any + particular conditions, is referred to as a "non-deterministic" NAT. + NATs that don't are called "deterministic". + + Non-deterministic NATs generally change behavior when a conflict of + some sort happens, i.e., when the port that would normally be used is + already in use by another mapping. The NAT mapping and External + Filtering in the absence of conflict is referred to as the Primary + behavior. The behavior after the first conflict is referred to as + Secondary and after the second conflict is referred to as Tertiary. + No NATs have been observed that change on further conflicts, but it + is certainly possible that they exist. + + REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT + change the NAT translation (Section 4) or the Filtering + (Section 5) Behavior at any point in time, or under any particular + conditions. + + Justification: Non-deterministic NATs are very difficult to + troubleshoot because they require more intensive testing. This + non-deterministic behavior is the root cause of much of the + uncertainty that NATs introduce about whether or not applications + will work. + +9. ICMP Destination Unreachable Behavior + + When a NAT sends a packet toward a host on the other side of the NAT, + an ICMP message may be sent in response to that packet. That ICMP + message may be sent by the destination host or by any router along + the network path. The NAT's default configuration SHOULD NOT filter + ICMP messages based on their source IP address. Such ICMP messages + SHOULD be rewritten by the NAT (specifically, the IP headers and the + ICMP payload) and forwarded to the appropriate internal or external + host. The NAT needs to perform this function for as long as the UDP + mapping is active. Receipt of any sort of ICMP message MUST NOT + destroy the NAT mapping. A NAT that performs the functions described + in the paragraph above is referred to as "support ICMP Processing". + + There is no significant security advantage to blocking ICMP + Destination Unreachable packets. Additionally, blocking ICMP + Destination Unreachable packets can interfere with application + failover, UDP Path MTU Discovery (see [RFC1191] and [RFC1435]), and + traceroute. Blocking any ICMP message is discouraged, and blocking + ICMP Destination Unreachable is strongly discouraged. + + + +Audet & Jennings Best Current Practice [Page 19] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the + NAT mapping. + + a) The NAT's default configuration SHOULD NOT filter ICMP messages + based on their source IP address. + + b) It is RECOMMENDED that a NAT support ICMP Destination + Unreachable messages. + + Justification: This is easy to do and is used for many things + including MTU discovery and rapid detection of error conditions, + and has no negative consequences. + +10. Fragmentation of Outgoing Packets + + When the MTU of the adjacent link is too small, fragmentation of + packets going from the internal side to the external side of the NAT + may occur. This can occur if the NAT is doing Point-to-Point over + Ethernet (PPPoE), or if the NAT has been configured with a small MTU + to reduce serialization delay when sending large packets and small + higher-priority packets, or for other reasons. + + It is worth noting that many IP stacks do not use Path MTU Discovery + with UDP packets. + + The packet could have its Don't Fragment bit set to 1 (DF=1) or 0 + (DF=0). + + REQ-13: If the packet received on an internal IP address has DF=1, + the NAT MUST send back an ICMP message "Fragmentation needed and + DF set" to the host, as described in [RFC0792]. + + a) If the packet has DF=0, the NAT MUST fragment the packet and + SHOULD send the fragments in order. + + Justification: This is as per RFC 792. + + a) This is the same function a router performs in a similar + situation [RFC1812]. + +11. Receiving Fragmented Packets + + For a variety of reasons, a NAT may receive a fragmented packet. The + IP packet containing the header could arrive in any fragment, + depending on network conditions, packet ordering, and the + implementation of the IP stack that generated the fragments. + + + + + +Audet & Jennings Best Current Practice [Page 20] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + A NAT that is capable only of receiving fragments in order (that is, + with the header in the first packet) and forwarding each of the + fragments to the internal host is described as "Received Fragments + Ordered". + + A NAT that is capable of receiving fragments in or out of order and + forwarding the individual fragments (or a reassembled packet) to the + internal host is referred to as "Receive Fragments Out of Order". + See the Security Considerations section of this document for a + discussion of this behavior. + + A NAT that is neither of these is referred to as "Receive Fragments + None". + + REQ-14: A NAT MUST support receiving in-order and out-of-order + fragments, so it MUST have "Received Fragment Out of Order" + behavior. + + a) A NAT's out-of-order fragment processing mechanism MUST be + designed so that fragmentation-based DoS attacks do not + compromise the NAT's ability to process in-order and + unfragmented IP packets. + + Justification: See Security Considerations. + +12. Requirements + + The requirements in this section are aimed at minimizing the + complications caused by NATs to applications, such as realtime + communications and online gaming. The requirements listed earlier in + the document are consolidated here into a single section. + + It should be understood, however, that applications normally do not + know in advance if the NAT conforms to the recommendations defined in + this section. Peer-to-peer media applications still need to use + normal procedures, such as ICE [ICE]. + + A NAT that supports all the mandatory requirements of this + specification (i.e., the "MUST"), is "compliant with this + specification". A NAT that supports all the requirements of this + specification (i.e., including the "RECOMMENDED") is "fully compliant + with all the mandatory and recommended requirements of this + specification". + + + + + + + + +Audet & Jennings Best Current Practice [Page 21] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. + + REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" + behavior of "Paired". Note that this requirement is not + applicable to NATs that do not support IP address pooling. + + REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port + overloading". + + a) If the host's source port was in the range 0-1023, it is + RECOMMENDED the NAT's source port be in the same range. If the + host's source port was in the range 1024-65535, it is + RECOMMENDED that the NAT's source port be in that range. + + REQ-4: It is RECOMMENDED that a NAT have a "Port parity + preservation" behavior of "Yes". + + REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two + minutes, unless REQ-5a applies. + + a) For specific destination ports in the well-known port range + (ports 0-1023), a NAT MAY have shorter UDP mapping timers that + are specific to the IANA-registered application running over + that specific destination port. + + b) The value of the NAT UDP mapping timer MAY be configurable. + + c) A default value of five minutes or more for the NAT UDP mapping + timer is RECOMMENDED. + + REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + refresh behavior" of "True". + + a) The NAT mapping Refresh Direction MAY have a "NAT Inbound + refresh behavior" of "True". + + REQ-7 A NAT device whose external IP interface can be configured + dynamically MUST either (1) Automatically ensure that its internal + network uses IP addresses that do not conflict with its external + network, or (2) Be able to translate and forward traffic between + all internal nodes and all external nodes whose IP addresses + numerically conflict with the internal network. + + REQ-8: If application transparency is most important, it is + RECOMMENDED that a NAT have "Endpoint-Independent Filtering" + behavior. If a more stringent filtering behavior is most + important, it is RECOMMENDED that a NAT have "Address-Dependent + Filtering" behavior. + + + +Audet & Jennings Best Current Practice [Page 22] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + a) The filtering behavior MAY be an option configurable by the + administrator of the NAT. + + REQ-9: A NAT MUST support "Hairpinning". + + a) A NAT Hairpinning behavior MUST be "External source IP address + and port". + + REQ-10: To eliminate interference with UNSAF NAT traversal + mechanisms and allow integrity protection of UDP communications, + NAT ALGs for UDP-based protocols SHOULD be turned off. Future + standards track specifications that define an ALG can update this + to recommend the ALGs on which they define default. + + a) If a NAT includes ALGs, it is RECOMMENDED that the NAT allow + the NAT administrator to enable or disable each ALG separately. + + REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT + change the NAT translation (Section 4) or the Filtering + (Section 5) Behavior at any point in time, or under any particular + conditions. + + REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the + NAT mapping. + + a) The NAT's default configuration SHOULD NOT filter ICMP messages + based on their source IP address. + + b) It is RECOMMENDED that a NAT support ICMP Destination + Unreachable messages. + + REQ-13 If the packet received on an internal IP address has DF=1, + the NAT MUST send back an ICMP message "Fragmentation needed and + DF set" to the host, as described in [RFC0792]. + + a) If the packet has DF=0, the NAT MUST fragment the packet and + SHOULD send the fragments in order. + + REQ-14: A NAT MUST support receiving in-order and out-of-order + fragments, so it MUST have "Received Fragment Out of Order" + behavior. + + a) A NAT's out-of-order fragment processing mechanism MUST be + designed so that fragmentation-based DoS attacks do not + compromise the NAT's ability to process in-order and + unfragmented IP packets. + + + + + +Audet & Jennings Best Current Practice [Page 23] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + +13. Security Considerations + + NATs are often deployed to achieve security goals. Most of the + recommendations and requirements in this document do not affect the + security properties of these devices, but a few of them do have + security implications and are discussed in this section. + + This document recommends that the timers for mapping be refreshed on + outgoing packets (see REQ-6) and does not make recommendations about + whether or not inbound packets should update the timers. If inbound + packets update the timers, an external attacker can keep the mapping + alive forever and attack future devices that may end up with the same + internal address. A device that was also the DHCP server for the + private address space could mitigate this by cleaning any mappings + when a DHCP lease expired. For unicast UDP traffic (the scope of + this document), it may not seem relevant to support inbound timer + refresh; however, for multicast UDP, the question is harder. It is + expected that future documents discussing NAT behavior with multicast + traffic will refine the requirements around handling of the inbound + refresh timer. Some devices today do update the timers on inbound + packets. + + This document recommends that the NAT filters be specific to the + external IP address only (see REQ-8) and not to the external IP + address and UDP port. It can be argued that this is less secure than + using the IP and port. Devices that wish to filter on IP and port do + still comply with these requirements. + + Non-deterministic NATs are risky from a security point of view. They + are very difficult to test because they are, well, non-deterministic. + Testing by a person configuring one may result in the person thinking + it is behaving as desired, yet under different conditions, which an + attacker can create, the NAT may behave differently. These + requirements recommend that devices be deterministic. + + This document requires that NATs have an "external NAT mapping is + endpoint independent" behavior. This does not reduce the security of + devices. Which packets are allowed to flow across the device is + determined by the external filtering behavior, which is independent + of the mapping behavior. + + When a fragmented packet is received from the external side, and the + packets are out of order so that the initial fragment does not arrive + first, many systems simply discard the out-of-order packets. + Moreover, since some networks deliver small packets ahead of large + ones, there can be many out-of-order fragments. NATs that are + capable of delivering these out-of-order packets are possible, but + they need to store the out-of-order fragments, which can open up a + + + +Audet & Jennings Best Current Practice [Page 24] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + Denial-of-Service (DoS) opportunity, if done incorrectly. + Fragmentation has been a tool used in many attacks, some involving + passing fragmented packets through NATs, and others involving DoS + attacks based on the state needed to reassemble the fragments. NAT + implementers should be aware of [RFC3128] and [RFC1858]. + +14. IAB Considerations + + The IAB has studied the problem of "Unilateral Self Address Fixing", + which is the general process by which a client attempts to determine + its address in another realm on the other side of a NAT through a + collaborative protocol reflection mechanism [RFC3424]. + + This specification does not, in itself, constitute an UNSAF + application. It consists of a series of requirements for NATs aimed + at minimizing the negative impact that those devices have on peer-to- + peer media applications, especially when those applications are using + UNSAF methods. + + Section 3 of UNSAF lists several practical issues with solutions to + NAT problems. This document makes recommendations to reduce the + uncertainty and problems introduced by these practical issues with + NATs. In addition, UNSAF lists five architectural considerations. + Although this is not an UNSAF proposal, it is interesting to consider + the impact of this work on these architectural considerations. + + Arch-1: The scope of this is limited to UDP packets in NATs like the + ones widely deployed today. The "fix" helps constrain the + variability of NATs for true UNSAF solutions such as STUN. + + Arch-2: This will exit at the same rate that NATs exit. It does not + imply any protocol machinery that would continue to live + after NATs were gone, or make it more difficult to remove + them. + + Arch-3: This does not reduce the overall brittleness of NATs, but + will hopefully reduce some of the more outrageous NAT + behaviors and make it easer to discuss and predict NAT + behavior in given situations. + + Arch-4: This work and the results [RESULTS] of various NATs + represent the most comprehensive work at IETF on what the + real issues are with NATs for applications like VoIP. This + work and STUN have pointed out, more than anything else, the + brittleness NATs introduce and the difficulty of addressing + these issues. + + + + + +Audet & Jennings Best Current Practice [Page 25] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + Arch-5: This work and the test results [RESULTS] provide a reference + model for what any UNSAF proposal might encounter in + deployed NATs. + +15. Acknowledgments + + The editor would like to acknowledge Bryan Ford, Pyda Srisuresh, and + Dan Kegel for their multiple contributions on peer-to-peer + communications across a NAT. Dan Wing contributed substantial text + on IP fragmentation and ICMP behavior. Thanks to Rohan Mahy, + Jonathan Rosenberg, Mary Barnes, Melinda Shore, Lyndsay Campbell, + Geoff Huston, Jiri Kuthan, Harald Welte, Steve Casner, Robert + Sanders, Spencer Dawkins, Saikat Guha, Christian Huitema, Yutaka + Takeda, Paul Hoffman, Lisa Dusseault, Pekka Savola, Peter Koch, Jari + Arkko, and Alfred Hoenes for their contributions. + +16. References + +16.1. Normative References + + [RFC0768] Postel, J., "User Datagram Protocol", STD 6, RFC 768, + August 1980. + + [RFC0791] Postel, J., "Internet Protocol", STD 5, RFC 791, + September 1981. + + [RFC2119] Bradner, S., "Key words for use in RFCs to Indicate + Requirement Levels", BCP 14, RFC 2119, March 1997. + +16.2. Informative References + + [RFC0792] Postel, J., "Internet Control Message Protocol", STD 5, + RFC 792, September 1981. + + [RFC1191] Mogul, J. and S. Deering, "Path MTU discovery", + RFC 1191, November 1990. + + [RFC1435] Knowles, S., "IESG Advice from Experience with Path MTU + Discovery", RFC 1435, March 1993. + + [RFC1812] Baker, F., "Requirements for IP Version 4 Routers", + RFC 1812, June 1995. + + [RFC1858] Ziemba, G., Reed, D., and P. Traina, "Security + Considerations for IP Fragment Filtering", RFC 1858, + October 1995. + + + + + +Audet & Jennings Best Current Practice [Page 26] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + [RFC1918] Rekhter, Y., Moskowitz, R., Karrenberg, D., Groot, G., + and E. Lear, "Address Allocation for Private + Internets", BCP 5, RFC 1918, February 1996. + + [RFC2460] Deering, S. and R. Hinden, "Internet Protocol, Version + 6 (IPv6) Specification", RFC 2460, December 1998. + + [RFC2623] Eisler, M., "NFS Version 2 and Version 3 Security + Issues and the NFS Protocol's Use of RPCSEC_GSS and + Kerberos V5", RFC 2623, June 1999. + + [RFC2663] Srisuresh, P. and M. Holdrege, "IP Network Address + Translator (NAT) Terminology and Considerations", + RFC 2663, August 1999. + + [RFC3022] Srisuresh, P. and K. Egevang, "Traditional IP Network + Address Translator (Traditional NAT)", RFC 3022, + January 2001. + + [RFC3027] Holdrege, M. and P. Srisuresh, "Protocol Complications + with the IP Network Address Translator", RFC 3027, + January 2001. + + [RFC3128] Miller, I., "Protection Against a Variant of the Tiny + Fragment Attack (RFC 1858)", RFC 3128, June 2001. + + [RFC3261] Rosenberg, J., Schulzrinne, H., Camarillo, G., + Johnston, A., Peterson, J., Sparks, R., Handley, M., + and E. Schooler, "SIP: Session Initiation Protocol", + RFC 3261, June 2002. + + [RFC3424] Daigle, L. and IAB, "IAB Considerations for UNilateral + Self-Address Fixing (UNSAF) Across Network Address + Translation", RFC 3424, November 2002. + + [RFC3489] Rosenberg, J., Weinberger, J., Huitema, C., and R. + Mahy, "STUN - Simple Traversal of User Datagram + Protocol (UDP) Through Network Address Translators + (NATs)", RFC 3489, March 2003. + + [RFC3550] Schulzrinne, H., Casner, S., Frederick, R., and V. + Jacobson, "RTP: A Transport Protocol for Real-Time + Applications", STD 64, RFC 3550, July 2003. + + [RFC3605] Huitema, C., "Real Time Control Protocol (RTCP) + attribute in Session Description Protocol (SDP)", + RFC 3605, October 2003. + + + + +Audet & Jennings Best Current Practice [Page 27] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + + [RFC4380] Huitema, C., "Teredo: Tunneling IPv6 over UDP through + Network Address Translations (NATs)", RFC 4380, + February 2006. + + [RFC3489bis] Rosenberg, J., "Simple Traversal Underneath Network + Address Translators (NAT) (STUN)", Work in Progress, + October 2006. + + [ICE] Rosenberg, J., "Interactive Connectivity Establishment + (ICE): A Methodology for Network Address Translator + (NAT) Traversal for Offer/Answer Protocols", Work + in Progress, October 2006. + + [RESULTS] Jennings, C., "NAT Classification Test Results", Work + in Progress, October 2006. + + [TURN] Rosenberg, J., "Obtaining Relay Addresses from Simple + Traversal Underneath NAT (STUN)", Work in Progress, + October 2006. + + [ITU.H323] "Packet-based Multimedia Communications Systems", ITU- + T Recommendation H.323, July 2003. + +Authors' Addresses + + Francois Audet (editor) + Nortel Networks + 4655 Great America Parkway + Santa Clara, CA 95054 + US + + Phone: +1 408 495 2456 + EMail: audet@nortel.com + + + Cullen Jennings + Cisco Systems + 170 West Tasman Drive + MS: SJC-21/2 + San Jose, CA 95134 + US + + Phone: +1 408 902 3341 + EMail: fluffy@cisco.com + + + + + + + +Audet & Jennings Best Current Practice [Page 28] + +RFC 4787 NAT UDP Unicast Requirements January 2007 + + +Full Copyright Statement + + Copyright (C) The IETF Trust (2007). + + This document is subject to the rights, licenses and restrictions + contained in BCP 78, and except as set forth therein, the authors + retain all their rights. + + This document and the information contained herein are provided on an + "AS IS" basis and THE CONTRIBUTOR, THE ORGANIZATION HE/SHE REPRESENTS + OR IS SPONSORED BY (IF ANY), THE INTERNET SOCIETY, THE IETF TRUST AND + THE INTERNET ENGINEERING TASK FORCE DISCLAIM ALL WARRANTIES, EXPRESS + OR IMPLIED, INCLUDING BUT NOT LIMITED TO ANY WARRANTY THAT THE USE OF + THE INFORMATION HEREIN WILL NOT INFRINGE ANY RIGHTS OR ANY IMPLIED + WARRANTIES OF MERCHANTABILITY OR FITNESS FOR A PARTICULAR PURPOSE. + +Intellectual Property + + The IETF takes no position regarding the validity or scope of any + Intellectual Property Rights or other rights that might be claimed to + pertain to the implementation or use of the technology described in + this document or the extent to which any license under such rights + might or might not be available; nor does it represent that it has + made any independent effort to identify any such rights. Information + on the procedures with respect to rights in RFC documents can be + found in BCP 78 and BCP 79. + + Copies of IPR disclosures made to the IETF Secretariat and any + assurances of licenses to be made available, or the result of an + attempt made to obtain a general license or permission for the use of + such proprietary rights by implementers or users of this + specification can be obtained from the IETF on-line IPR repository at + http://www.ietf.org/ipr. + + The IETF invites any interested party to bring to its attention any + copyrights, patents or patent applications, or other proprietary + rights that may cover technology that may be required to implement + this standard. Please address the information to the IETF at + ietf-ipr@ietf.org. + +Acknowledgement + + Funding for the RFC Editor function is currently provided by the + Internet Society. + + + + + + + +Audet & Jennings Best Current Practice [Page 29] + diff --git a/nat/src/masquerade/apalloc/mod.rs b/nat/src/masquerade/apalloc/mod.rs index 3e3822421f..4d26ed4dd0 100644 --- a/nat/src/masquerade/apalloc/mod.rs +++ b/nat/src/masquerade/apalloc/mod.rs @@ -285,9 +285,15 @@ impl NatAllocator { //= type=todo //# REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior //# for TCP. + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.1 + //= type=todo + //# REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. //= https://www.rfc-editor.org/rfc/rfc5382#section-8 //# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port //# overloading" for TCP. + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1 + //# REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port + //# overloading". fn allocate_v4( &self, src_vpcd: VpcDiscriminant, diff --git a/nat/src/masquerade/apalloc/port_alloc.rs b/nat/src/masquerade/apalloc/port_alloc.rs index c951899dc1..76dbe75c7f 100644 --- a/nat/src/masquerade/apalloc/port_alloc.rs +++ b/nat/src/masquerade/apalloc/port_alloc.rs @@ -92,6 +92,10 @@ pub(crate) struct PortAllocator { exclude_wellknown_ports: bool, } +//= https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1 +//= type=exception +//# a) If the host's source port was in the range 0-1023, it is +//# RECOMMENDED the NAT's source port be in the same range. /// Ports 0..=1023 cover the IANA system/well-known range and should not be /// allocated by masquerade NAT for TCP or UDP. pub(super) const IANA_WELLKNOWN_PORT_LIMIT: u16 = 1024; diff --git a/nat/src/masquerade/fuzz.rs b/nat/src/masquerade/fuzz.rs index cf183b1007..5ea8bf8f4a 100644 --- a/nat/src/masquerade/fuzz.rs +++ b/nat/src/masquerade/fuzz.rs @@ -204,6 +204,10 @@ fn out_unchanged(out: &[Packet], before: (IpAddr, u16)) -> bool { //= type=test //# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port //# overloading" for TCP. +//= https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1 +//= type=test +//# REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port +//# overloading". #[test] fn distinct_flows_do_not_share_a_translation() { let tally = Tally::default(); diff --git a/nat/src/masquerade/mod.rs b/nat/src/masquerade/mod.rs index 1e708dd93d..07d3317ed2 100644 --- a/nat/src/masquerade/mod.rs +++ b/nat/src/masquerade/mod.rs @@ -17,6 +17,9 @@ mod state; mod state_machine; mod test; +//= https://www.rfc-editor.org/rfc/rfc4787#section-6 +//= type=todo +//# REQ-9: A NAT MUST support "Hairpinning". // re exports pub use allocator_writer::MasqueradeConfig; pub use allocator_writer::NatAllocatorWriter; diff --git a/nat/src/masquerade/nf.rs b/nat/src/masquerade/nf.rs index e3b0313cfa..7f5b7e4b2a 100644 --- a/nat/src/masquerade/nf.rs +++ b/nat/src/masquerade/nf.rs @@ -105,6 +105,14 @@ impl Masquerade { //# connection idle-timeout" MUST NOT be less than 2 hours 4 minutes. //# The value of the "transitory connection idle-timeout" MUST NOT be //# less than 4 minutes. + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 + //= type=todo + //# REQ-5: A NAT UDP mapping timer MUST NOT expire in less than two + //# minutes, unless REQ-5a applies. + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 + //= type=todo + //# c) A default value of five minutes or more for the NAT UDP mapping + //# timer is RECOMMENDED. pub const MASQUERADE_ONEWAY_TIMEOUT: Duration = Duration::from_secs(5 * Self::TIMEOUT_SCALE); pub const MASQUERADE_TWOWAY_TIMEOUT: Duration = Duration::from_secs(3 * Self::TIMEOUT_SCALE); pub const MASQUERADE_CLOSING_TIMEOUT: Duration = Duration::from_secs(2 * Self::TIMEOUT_SCALE); @@ -586,6 +594,11 @@ impl Masquerade { return; } + //= https://www.rfc-editor.org/rfc/rfc4787#section-11 + //= type=todo + //# REQ-14: A NAT MUST support receiving in-order and out-of-order + //# fragments, so it MUST have "Received Fragment Out of Order" + //# behavior. // TODO: Check whether the packet is fragmented if let Err(error) = self.masquerade_packet(packet) { packet.done((&error).into()); diff --git a/nat/src/masquerade/protocol.rs b/nat/src/masquerade/protocol.rs index fee71b3b5e..16829d8ede 100644 --- a/nat/src/masquerade/protocol.rs +++ b/nat/src/masquerade/protocol.rs @@ -14,6 +14,11 @@ use net::packet::Packet; use net::tcp::Tcp; impl NatFlowStatus { + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 + //# a) For specific destination ports in the well-known port range + //# (ports 0-1023), a NAT MAY have shorter UDP mapping timers that + //# are specific to the IANA-registered application running over + //# that specific destination port. fn udp_status_patch_dnat(self, packet: &Packet) -> NatFlowStatus { match packet.headers().pat().eth().net().udp().done() { Some((_, _, udp)) => match udp.source().as_u16() { @@ -53,6 +58,9 @@ fn next_flow_status_udp(action: NatAction, status: NatFlowStatus) -> NatFlowStat //= https://www.rfc-editor.org/rfc/rfc5382#section-8 //# REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the //# NAT mapping or TCP connection for which the ICMP was generated. +//= https://www.rfc-editor.org/rfc/rfc4787#section-9 +//# REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the +//# NAT mapping. #[allow(clippy::match_single_binding)] fn next_flow_status_icmp(action: NatAction, status: NatFlowStatus) -> NatFlowStatus { match action { From 92a5049d375de53070fd1e43fcb85df8a56fda87 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Tue, 18 Aug 2026 23:33:31 -0600 Subject: [PATCH 05/21] test(masquerade): State RFC 4787 endpoint independence as a property The goal is properties at the abstraction an RFC is written at: configure the NF, feed it generated packets, assert something implementation independent. RFC 4787 REQ-1 is a good test of whether that lines up, because it is a statement about the NF that no unit can make. It lines up, and it settles the question REQ-1 was left `todo` over. The open reading was that the mapping is endpoint-dependent only across destination VPCs, reasoning from `allocate_v4`'s signature, which takes no destination address and no destination port. That looked like grounds to call the deviation architectural and probably defensible. `an_internal_endpoint_keeps_one_public_address` holds the internal endpoint fixed and moves the destination. Written first as the full REQ-1 claim, it failed on the first input drawn: 10.0.0.0:1 gets public port 1024 talking to 3.3.3.1:1 and 1025 talking to 3.3.3.1:2. Same destination address, same VPC, only the destination *port* changed. That is "Address and Port-Dependent Mapping" in RFC 4787 section 4.1 -- the most restrictive of the three classes and the one REQ-1 forbids. The cost is UNSAF traversal, which is the entire justification the RFC gives for the requirement. The dependence was never in the allocator's arguments. It is in being called again for each new flow, which is invisible at the allocator and visible at the stage. That is the argument for this level of testing in one sentence. What the committed property asserts is the half that holds. The public *address* is stable across destinations even when the port is not, which is REQ-2, "IP address pooling behavior of Paired". So the same two lines in `Pool::allocate` satisfy REQ-2 and miss REQ-1, and they are now cited as both -- the partial conformance case in its clearest available form. REQ-1 stays `todo` rather than becoming an `exception`. An exception asserts somebody weighed the requirement and accepted the cost; that has not happened, and now that the cost is stated precisely it is worth asking for. Also carries the first use of `reason=` on a citation, which duvet permits on exception, implication, implementation and test but not on todo, and which must fit on one line -- a bare continuation is parsed as a second source. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/snapshot.txt | 4 +- nat/src/masquerade/apalloc/alloc.rs | 4 ++ nat/src/masquerade/fuzz.rs | 65 +++++++++++++++++++++++++++++ 3 files changed, 71 insertions(+), 2 deletions(-) diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index 6e05273b0f..c2a2510859 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -1,8 +1,8 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 SECTION: [Address and Port Mapping](#section-4.1) TEXT[!MUST,todo]: REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. - TEXT[!SHOULD]: REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" - TEXT[!SHOULD]: behavior of "Paired". + TEXT[!SHOULD,implementation,test]: REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" + TEXT[!SHOULD,implementation,test]: behavior of "Paired". SECTION: [Port Assignment Behavior](#section-4.2.1) TEXT[!MUST,implementation,test]: REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port diff --git a/nat/src/masquerade/apalloc/alloc.rs b/nat/src/masquerade/apalloc/alloc.rs index 50c2eb6412..6c8c94dbbb 100644 --- a/nat/src/masquerade/apalloc/alloc.rs +++ b/nat/src/masquerade/apalloc/alloc.rs @@ -117,6 +117,10 @@ impl IpAllocator { // FIXME: Should we clean up every time?? self.cleanup_used_ips(); + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.1 + //= reason=held: reuse before draw is what makes the pooling behaviour "Paired" + //# REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" + //# behavior of "Paired". // Draw a fresh address only when the addresses already in use are exhausted. Other errors // describe allocator failure and must be preserved. match self.reuse_allocated_ip(allow_null) { diff --git a/nat/src/masquerade/fuzz.rs b/nat/src/masquerade/fuzz.rs index 5ea8bf8f4a..08888ccf4f 100644 --- a/nat/src/masquerade/fuzz.rs +++ b/nat/src/masquerade/fuzz.rs @@ -200,6 +200,71 @@ fn out_unchanged(out: &[Packet], before: (IpAddr, u16)) -> bool { out[0].is_done() || source_of(&out[0]) == before } +//= https://www.rfc-editor.org/rfc/rfc4787#section-4.1 +//= type=test +//# REQ-2: It is RECOMMENDED that a NAT have an "IP address pooling" +//# behavior of "Paired". +#[test] +fn an_internal_endpoint_keeps_one_public_address() { + let tally = Tally::default(); + + with_runtime(|| { + bolero::check!() + .with_generator(Scenario { strays: false }) + .cloned() + .for_each(|(exposes, probes): (Vec, Vec)| { + tally.seen.fetch_add(1, Ordering::Relaxed); + let Some(fabric) = fabric(&exposes) else { + return; + }; + tally.built.fetch_add(1, Ordering::Relaxed); + let (mut lookup, mut masq) = fabric.stages(); + + for spec in &probes { + let probe = (*spec).resolve(&fabric); + let before = (probe.source, probe.sport); + let first = run(&mut lookup, &mut masq, vec![probe.packet()], probe.arrival.dst_vpcd); + if out_unchanged(&first, before) { + continue; + } + + let mut elsewhere = (*spec).resolve(&fabric); + elsewhere.dport = elsewhere.dport.wrapping_add(1).max(1); + if let Some(other) = fabric.peer.iter().find(|a| **a != probe.destination) { + elsewhere.destination = *other; + } + if (elsewhere.destination, elsewhere.dport) == (probe.destination, probe.dport) { + continue; + } + + let second = run( + &mut lookup, + &mut masq, + vec![elsewhere.packet()], + elsewhere.arrival.dst_vpcd, + ); + if out_unchanged(&second, before) { + continue; + } + + assert_eq!( + source_of(&second[0]).0, + source_of(&first[0]).0, + "{before:?} was given {:?} talking to {:?} and {:?} talking to {:?}, so the \ + public address it is given depends on who it is addressing", + source_of(&first[0]), + (probe.destination, probe.dport), + source_of(&second[0]), + (elsewhere.destination, elsewhere.dport) + ); + tally.reached.fetch_add(1, Ordering::Relaxed); + } + }); + }); + + tally.report("address pairing"); +} + //= https://www.rfc-editor.org/rfc/rfc5382#section-8 //= type=test //# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port From 98652bfe904a9d0dc41132efaa6dbefd7605aea2 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Tue, 18 Aug 2026 23:49:03 -0600 Subject: [PATCH 06/21] test(masquerade): Assert RFC 4787 outbound refresh, and where it is not REQ-6 is a MUST: an outbound packet must keep a mapping alive. The module already had `traffic_extends_a_flow_past_its_first_deadline`, which refreshes with replies -- that is *inbound* refresh, REQ-6a, and only a MAY. The permitted behaviour was covered and the required one was not. Writing the missing test turned up a defect. Measured on a paused clock, control against treatment, with a five-second `OneWay` lifetime: silent for eight seconds, the reply to the mapping is dropped, as expected; an outbound packet at four seconds, then the same probe at eight seconds, also dropped. The packet changed nothing. `refresh_masquerade_state` is where it comes from. Its `OneWay` arm yields `None` for the extension, so no outbound packet ever moves the deadline of a flow that has not yet had a reply. The comment there reasons about the reverse direction and treats `OneWay` as a corner case, which is what makes returning `None` look harmless. It is not a corner: it is the steady state of every outbound-only flow. Syslog, netflow, telemetry, a resolver query nobody answers -- each has its mapping torn down five seconds after its first packet however much it sends, and rebuilt on the next one. Once a reply arrives the flow reaches `Established` and outbound refresh does work, so REQ-6 is met for connections and missed for one-way traffic. That half is now asserted: three outbound packets a hundred seconds apart against a hundred-and-twenty second idle timeout, five minutes with nothing arriving from outside. Two details make the assertion mean what it says. The step is near the timeout, because a step comfortably inside the lifetime the previous packet already bought would pass with refresh deleted. And the probe comes after a delay longer than a `OneWay` lifetime, so a mapping that had been silently torn down and rebuilt by the last outbound packet is already dead when it is checked -- reissuing the identical tuple cannot fake a pass. The `OneWay` gap is recorded as `todo` rather than `exception`, and deliberately not written as a test. A test pinning the current behaviour would make the deviation permanent, which is the entrenchment failure this whole exercise exists to avoid. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/snapshot.txt | 4 +-- nat/src/masquerade/expiry.rs | 55 ++++++++++++++++++++++++++++++++++++ nat/src/masquerade/nf.rs | 4 +++ 3 files changed, 61 insertions(+), 2 deletions(-) diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index c2a2510859..fe89aaf19a 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -27,8 +27,8 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 TEXT[!MAY]: b) The value of the NAT UDP mapping timer MAY be configurable. TEXT[!SHOULD,todo]: c) A default value of five minutes or more for the NAT UDP mapping TEXT[!SHOULD,todo]: timer is RECOMMENDED. - TEXT[!MUST]: REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound - TEXT[!MUST]: refresh behavior" of "True". + TEXT[!MUST,test,todo]: REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + TEXT[!MUST,test,todo]: refresh behavior" of "True". TEXT[!MAY]: a) The NAT mapping Refresh Direction MAY have a "NAT Inbound TEXT[!MAY]: refresh behavior" of "True". diff --git a/nat/src/masquerade/expiry.rs b/nat/src/masquerade/expiry.rs index ccdd90ffdc..c6db91d541 100644 --- a/nat/src/masquerade/expiry.rs +++ b/nat/src/masquerade/expiry.rs @@ -19,6 +19,8 @@ const PAST_EXPIRY: Duration = Duration::from_secs(30); const WITHIN_LIFETIME: Duration = Duration::from_secs(1); +const NEARLY_ESTABLISHED: Duration = Duration::from_secs(100); + fn vni(raw: u32) -> Vni { Vni::new_checked(raw).unwrap_or_else(|_| unreachable!()) } @@ -159,6 +161,59 @@ fn traffic_extends_a_flow_past_its_first_deadline() { }); } +//= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 +//= type=test +//= reason=held: for established flows; see the OneWay gap recorded in nf.rs +//# REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound +//# refresh behavior" of "True". +#[test] +fn outbound_traffic_keeps_an_established_mapping_alive() { + with_paused_clock(|| async { + let (fabric, _) = fabric(); + let (mut lookup, mut masq) = fabric.stages(); + let peer = fabric.peer[0]; + let source: IpAddr = "10.0.0.7".parse().unwrap_or_else(|_| unreachable!()); + + let translated = open_flow(&mut lookup, &mut masq, source, peer, 1234) + .unwrap_or_else(|| unreachable!("a fixed private source is masqueraded")); + assert_eq!( + reply_to(&mut lookup, &mut masq, peer, translated), + Some(source), + "the reply that establishes the connection was not delivered" + ); + assert_eq!( + open_flow(&mut lookup, &mut masq, source, peer, 1234), + Some(translated), + "the packet that establishes the connection changed its translation" + ); + + for step in 1..=3 { + advance(NEARLY_ESTABLISHED).await; + assert_eq!( + open_flow(&mut lookup, &mut masq, source, peer, 1234), + Some(translated), + "at {}s an outbound packet no longer found the mapping", + step * NEARLY_ESTABLISHED.as_secs() + ); + } + + advance(PAST_EXPIRY).await; + assert_eq!( + reply_to(&mut lookup, &mut masq, peer, translated), + Some(source), + "the mapping did not survive five minutes of outbound traffic, so outbound packets \ + are not refreshing it" + ); + + advance(Duration::from_mins(5)).await; + assert_eq!( + reply_to(&mut lookup, &mut masq, peer, translated), + None, + "a mapping held open by outbound traffic never expired once that traffic stopped" + ); + }); +} + #[test] fn an_expired_flow_is_never_resurrected() { with_paused_clock(|| async { diff --git a/nat/src/masquerade/nf.rs b/nat/src/masquerade/nf.rs index 7f5b7e4b2a..0b33c64afe 100644 --- a/nat/src/masquerade/nf.rs +++ b/nat/src/masquerade/nf.rs @@ -202,6 +202,10 @@ impl Masquerade { | NatFlowStatus::CHalfClose | NatFlowStatus::SHalfClose | NatFlowStatus::LastAck => Some(Self::MASQUERADE_CLOSING_TIMEOUT), + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 + //= type=todo + //# REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + //# refresh behavior" of "True". NatFlowStatus::OneWay => { // this could happen if a burst of packets are sent before any state is there (snat), // or if we got a TCP segment back without expected flags. This should never happen for From 8e3eb249a65920995a721ea4457f4841fd5c7153 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Wed, 19 Aug 2026 00:16:42 -0600 Subject: [PATCH 07/21] test(masquerade): Classify the filtering, and check RFC 4787 determinism A deliberate move to a different species of requirement, to find where the method strains. Everything cited so far has been first order and unconditional -- do this, never do that, not less than this many seconds. RFC 4787 has two that are neither. REQ-11 is second order. It is not a requirement about a packet; it requires that the answers to the *other* requirements stay the same "at any point in time, or under any particular conditions". Citing it correctly needs two readings settled first. "Behavior" means the class from section 4, not the values -- section 4.2.1 explicitly permits random port assignment, so the allocator shuffling port blocks is not a violation, and a naive citation would have called it one. And the conflict the RFC is aimed at, in section 8, is port preservation with a fallback path, which does not exist here because nothing ever tries to preserve a source port. That leaves address pooling as the thing that could still change under pressure, and it does not. 254 hosts, 256 flows each, 65,024 flows against a public /24: the pool spills to a second public address and no internal host is ever given more than one. Pairing before the spill is pairing after it. Two structural facts fell out of measuring it. A single internal host can never break pairing, because its own source port space is exactly the size of one public address's port space -- 64,512 flows from one host stayed on one address with nothing denied. So the transition only exists under contention between hosts, which is why the probe needs 254 of them. It costs fourteen seconds against a suite that runs in four, so it is `#[ignore]`d as a characterization probe, following the precedent in `acl/src/dpdk/dyn_table.rs`. REQ-8 is conditional, and is the one that actually gives the method trouble. It does not state a behaviour; it states two and picks between them on a priority nobody has written down -- Endpoint-Independent Filtering "if application transparency is most important", Address-Dependent Filtering "if a more stringent filtering behavior is most important". Three probes against one flow classify what we do exactly: same address and port delivered, same address different port dropped, different address dropped. That is Address and Port-Dependent Filtering, the most restrictive of section 5's three classes, and it satisfies neither branch of REQ-8 -- we are stricter than the stringent option, which would let the second probe through. Stricter than a SHOULD asks is still a departure from it. Recorded as `todo` rather than `exception` because this reads as a consequence of keying the flow table on the whole five-tuple rather than a filtering policy anyone chose; what is missing is a recorded priority, not code. The filtering test is committed unignored regardless of how REQ-8 is resolved. An unsolicited packet reaching a tenant because it guessed a live public tuple is a security failure, and the second and third probes are what rule it out. It carries its own positive control: the first probe is delivered through the same path the other two are dropped by, so it cannot pass vacuously. Also confirms a FIXME in `apalloc/setup.rs` is unreachable rather than latent. It warns that a public range restricted to a port range is not modelled by the pools, which reads as a silent misconfiguration; in fact validation refuses such a config outright with "Port ranges are not supported with masquerade". Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/snapshot.txt | 20 ++++---- nat/src/masquerade/expiry.rs | 99 ++++++++++++++++++++++++++++++++++++ 2 files changed, 109 insertions(+), 10 deletions(-) diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index fe89aaf19a..0d495baebf 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -41,12 +41,12 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 TEXT[!MUST]: numerically conflict with the internal network. SECTION: [Filtering Behavior](#section-5) - TEXT[!SHOULD]: REQ-8: If application transparency is most important, it is - TEXT[!SHOULD]: RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" - TEXT[!SHOULD]: behavior. - TEXT[!SHOULD]: If a more stringent filtering behavior is most - TEXT[!SHOULD]: important, it is RECOMMENDED that a NAT have an "Address-Dependent - TEXT[!SHOULD]: Filtering" behavior. + TEXT[!SHOULD,todo]: REQ-8: If application transparency is most important, it is + TEXT[!SHOULD,todo]: RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" + TEXT[!SHOULD,todo]: behavior. + TEXT[!SHOULD,todo]: If a more stringent filtering behavior is most + TEXT[!SHOULD,todo]: important, it is RECOMMENDED that a NAT have an "Address-Dependent + TEXT[!SHOULD,todo]: Filtering" behavior. TEXT[!MAY]: a) The filtering behavior MAY be an option configurable by the TEXT[!MAY]: administrator of the NAT. @@ -63,10 +63,10 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 TEXT[!SHOULD]: the NAT administrator to enable or disable each ALG separately. SECTION: [Deterministic Properties](#section-8) - TEXT[!MUST]: REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT - TEXT[!MUST]: change the NAT translation (Section 4) or the Filtering - TEXT[!MUST]: (Section 5) Behavior at any point in time, or under any particular - TEXT[!MUST]: conditions. + TEXT[!MUST,test]: REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT + TEXT[!MUST,test]: change the NAT translation (Section 4) or the Filtering + TEXT[!MUST,test]: (Section 5) Behavior at any point in time, or under any particular + TEXT[!MUST,test]: conditions. SECTION: [ICMP Destination Unreachable Behavior](#section-9) TEXT[!SHOULD]: The NAT's default configuration SHOULD NOT filter diff --git a/nat/src/masquerade/expiry.rs b/nat/src/masquerade/expiry.rs index c6db91d541..6d8f6a814e 100644 --- a/nat/src/masquerade/expiry.rs +++ b/nat/src/masquerade/expiry.rs @@ -92,6 +92,105 @@ fn reply_to( .flatten() } +//= https://www.rfc-editor.org/rfc/rfc4787#section-8 +//= type=test +//= reason=held for the mapping dimension: pairing is unchanged by exhaustion, measured below +//# REQ-11: A NAT MUST have deterministic behavior, i.e., it MUST NOT +//# change the NAT translation (Section 4) or the Filtering +//# (Section 5) Behavior at any point in time, or under any particular +//# conditions. +#[test] +#[ignore = "characterization probe; run with --ignored --nocapture"] +fn pairing_is_unchanged_by_pool_exhaustion() { + use std::collections::{BTreeMap, BTreeSet}; + with_paused_clock(|| async { + let (fabric, _) = fabric(); + let (mut lookup, mut masq) = fabric.stages(); + let peer = fabric.peer[0]; + let mut given: BTreeMap> = BTreeMap::new(); + + for host in 1..=254u16 { + let source: IpAddr = format!("10.0.0.{host}") + .parse() + .unwrap_or_else(|_| unreachable!()); + for sport in 1024..1024 + 256u16 { + if let Some((public, _)) = open_flow(&mut lookup, &mut masq, source, peer, sport) { + given.entry(source).or_default().insert(public); + } + } + } + + let publics: BTreeSet<_> = given.values().flatten().copied().collect(); + println!( + "{} hosts, {} public addresses in use", + given.len(), + publics.len() + ); + assert!( + publics.len() > 1, + "the pool never spilled to a second address, so this measured nothing about conflict" + ); + let split: Vec<_> = given.iter().filter(|(_, a)| a.len() > 1).collect(); + assert!( + split.is_empty(), + "pooling changed under pressure: {} hosts were given more than one public address, \ + so the behaviour before the spill is not the behaviour after it", + split.len() + ); + }); +} + +fn inbound_from( + lookup: &mut FlowLookup, + masq: &mut Masquerade, + from: IpAddr, + sport: u16, + translated: (IpAddr, u16), +) -> bool { + let mut packet = build(from, translated.0, false, sport, translated.1); + Arrival::inbound().stamp(&mut packet); + let out: Vec> = run(lookup, masq, vec![packet], Some(vni(LOCAL_VNI))); + !out[0].is_done() +} + +//= https://www.rfc-editor.org/rfc/rfc4787#section-5 +//= type=todo +//# REQ-8: If application transparency is most important, it is +//# RECOMMENDED that a NAT have an "Endpoint-Independent Filtering" +//# behavior. If a more stringent filtering behavior is most +//# important, it is RECOMMENDED that a NAT have an "Address-Dependent +//# Filtering" behavior. +#[test] +fn only_the_endpoint_a_flow_addressed_can_reply() { + with_paused_clock(|| async { + let (fabric, _) = fabric(); + let (mut lookup, mut masq) = fabric.stages(); + let peer = fabric.peer[0]; + let elsewhere = *fabric + .peer + .iter() + .find(|a| **a != peer) + .unwrap_or_else(|| unreachable!("the fixture offers two peer addresses")); + let source: IpAddr = "10.0.0.7".parse().unwrap_or_else(|_| unreachable!()); + + let translated = open_flow(&mut lookup, &mut masq, source, peer, 1234) + .unwrap_or_else(|| unreachable!("a fixed private source is masqueraded")); + + assert!( + inbound_from(&mut lookup, &mut masq, peer, 80, translated), + "the endpoint the flow addressed could not answer it" + ); + assert!( + !inbound_from(&mut lookup, &mut masq, peer, 81, translated), + "a packet from the right address on the wrong port reached the tenant" + ); + assert!( + !inbound_from(&mut lookup, &mut masq, elsewhere, 80, translated), + "a packet from an address the flow never addressed reached the tenant" + ); + }); +} + #[test] fn a_flow_inside_its_lifetime_survives() { with_paused_clock(|| async { From 2cbb014ceabacc29f1ec992d53f6f91495cca5e9 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Wed, 19 Aug 2026 12:09:18 -0600 Subject: [PATCH 08/21] test(masquerade): State RFC 4787 REQ-12 as an executable contract A duvet citation records that somebody read a requirement. It cannot record that the code still does what the sentence says, and it cannot record that the test cited `type=test` checks the same thing as the code cited `type=implementation`. Those are two independent readings of one sentence, and a refactor can separate them without either comment changing. `contract::rfc4787::Req12` is the predicate written once. The implementation calls it from a `debug_assert!`; the exhaustive state machine test calls it directly. A mutant that sends an established flow to `Closed` on an ICMP packet now panics at the implementation site with the two states named, rather than at whichever assertion happened to notice. This closes a live gap rather than only demonstrating the pattern. RFC 4787 REQ-12 and RFC 5382 REQ-10 are the same sentence, kept by the same function and proved by the same test, but only RFC 5382 carried a `type=test` citation, so REQ-12 stood at `[!MUST,implementation]` -- implemented, untested -- while the test that establishes it sat six lines away. Manual bookkeeping across two specifications is exactly what a shared predicate removes. Only local predicates belong in `contract`. REQ-1 relates two mappings made at different times, REQ-6 relates a packet to a timer, REQ-11 is a statement about the answers to the other requirements; none is decidable at a point and all stay in `fuzz`. Where a requirement can be encoded more strongly it should be, per `development/code/avoid-global-reasoning.md`, and then it does not belong here at all. An `rfc5382::Req10` alias was written and deleted. Nothing called it, the compiler said so, and an uncalled contract is the decoration this replaces. The snapshot in `.duvet/` is not regenerated here: duvet is not on PATH outside the dev shell, and hand-editing the regression gate would defeat it. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- nat/src/masquerade/contract.rs | 86 +++++++++++++++++++++++++++++ nat/src/masquerade/mod.rs | 1 + nat/src/masquerade/protocol.rs | 12 +++- nat/src/masquerade/state_machine.rs | 26 ++++++--- 4 files changed, 114 insertions(+), 11 deletions(-) create mode 100644 nat/src/masquerade/contract.rs diff --git a/nat/src/masquerade/contract.rs b/nat/src/masquerade/contract.rs new file mode 100644 index 0000000000..6f0945583a --- /dev/null +++ b/nat/src/masquerade/contract.rs @@ -0,0 +1,86 @@ +// SPDX-License-Identifier: Apache-2.0 +// Copyright Open Network Fabric Authors + +use crate::common::NatFlowStatus; + +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub(crate) struct Violation { + pub(crate) requirement: &'static str, + pub(crate) detail: &'static str, +} + +impl std::fmt::Display for Violation { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + write!(f, "{}: {}", self.requirement, self.detail) + } +} + +pub(crate) mod rfc4787 { + use super::{NatFlowStatus, Violation}; + + #[derive(Debug, Clone, Copy)] + pub(crate) struct Req12 { + before: NatFlowStatus, + after: NatFlowStatus, + } + + impl Req12 { + pub(crate) const fn new(before: NatFlowStatus, after: NatFlowStatus) -> Self { + Self { before, after } + } + + pub(crate) const fn check(self) -> Result<(), Violation> { + if Self::terminal(self.after) && !Self::terminal(self.before) { + return Err(Violation { + requirement: "rfc4787#section-9 REQ-12 / rfc5382#section-8 REQ-10", + detail: "an ICMP packet moved a live flow into a terminal state", + }); + } + Ok(()) + } + + const fn terminal(status: NatFlowStatus) -> bool { + matches!(status, NatFlowStatus::Closed | NatFlowStatus::Reset) + } + } +} + +#[cfg(test)] +mod test { + use super::rfc4787::Req12; + use crate::common::NatFlowStatus; + + const STATUSES: [NatFlowStatus; 10] = [ + NatFlowStatus::OneWay, + NatFlowStatus::TwoWay, + NatFlowStatus::Established, + NatFlowStatus::CClosing, + NatFlowStatus::SClosing, + NatFlowStatus::CHalfClose, + NatFlowStatus::SHalfClose, + NatFlowStatus::LastAck, + NatFlowStatus::Reset, + NatFlowStatus::Closed, + ]; + + #[test] + fn the_contract_rejects_exactly_the_forbidden_transitions() { + let terminal = |s| matches!(s, NatFlowStatus::Closed | NatFlowStatus::Reset); + let mut rejected = 0; + for before in STATUSES { + for after in STATUSES { + let forbidden = terminal(after) && !terminal(before); + assert_eq!( + Req12::new(before, after).check().is_err(), + forbidden, + "{before:?} -> {after:?}" + ); + rejected += usize::from(forbidden); + } + } + assert_eq!( + rejected, 16, + "8 live statuses times 2 terminal ones must be the whole forbidden set" + ); + } +} diff --git a/nat/src/masquerade/mod.rs b/nat/src/masquerade/mod.rs index 07d3317ed2..b1ea45d9fb 100644 --- a/nat/src/masquerade/mod.rs +++ b/nat/src/masquerade/mod.rs @@ -4,6 +4,7 @@ pub(crate) mod allocation; mod allocator_writer; pub mod apalloc; +mod contract; mod expiry; pub(crate) mod flows; mod fuzz; diff --git a/nat/src/masquerade/protocol.rs b/nat/src/masquerade/protocol.rs index 16829d8ede..9c487849ac 100644 --- a/nat/src/masquerade/protocol.rs +++ b/nat/src/masquerade/protocol.rs @@ -6,6 +6,7 @@ //! for port conservation. use crate::common::{NatAction, NatFlowStatus}; +use crate::masquerade::contract::rfc4787::Req12; use net::buffer::PacketBufferMut; use net::headers::{TryHeaders, TryIp, TryTcp}; @@ -56,14 +57,16 @@ fn next_flow_status_udp(action: NatAction, status: NatFlowStatus) -> NatFlowStat } //= https://www.rfc-editor.org/rfc/rfc5382#section-8 +//= type=implementation //# REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the //# NAT mapping or TCP connection for which the ICMP was generated. //= https://www.rfc-editor.org/rfc/rfc4787#section-9 +//= type=implementation //# REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the //# NAT mapping. #[allow(clippy::match_single_binding)] fn next_flow_status_icmp(action: NatAction, status: NatFlowStatus) -> NatFlowStatus { - match action { + let next = match action { NatAction::SrcNat => match status { _ => status, }, @@ -71,7 +74,12 @@ fn next_flow_status_icmp(action: NatAction, status: NatFlowStatus) -> NatFlowSta NatFlowStatus::OneWay => NatFlowStatus::TwoWay, _ => status, }, - } + }; + debug_assert!( + Req12::new(status, next).check().is_ok(), + "{action} {status:?} -> {next:?}" + ); + next } fn next_flow_status_tcp(action: NatAction, status: NatFlowStatus, tcp: &Tcp) -> NatFlowStatus { diff --git a/nat/src/masquerade/state_machine.rs b/nat/src/masquerade/state_machine.rs index ca85d854c2..2b4d5f29af 100644 --- a/nat/src/masquerade/state_machine.rs +++ b/nat/src/masquerade/state_machine.rs @@ -4,6 +4,7 @@ #![cfg(test)] use crate::common::{NatAction, NatFlowStatus}; +use crate::masquerade::contract::rfc4787::Req12; use crate::masquerade::protocol::next_flow_status; use net::buffer::TestBuffer; use net::headers::TryTcpMut; @@ -199,6 +200,10 @@ fn ordinary_udp_opens_and_settles() { //= type=test //# REQ-10: Receipt of any sort of ICMP message MUST NOT terminate the //# NAT mapping or TCP connection for which the ICMP was generated. +//= https://www.rfc-editor.org/rfc/rfc4787#section-9 +//= type=test +//# REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the +//# NAT mapping. #[test] fn an_icmp_reply_makes_a_flow_two_way_and_nothing_more() { let packet = build_test_icmp4_echo( @@ -216,16 +221,19 @@ fn an_icmp_reply_makes_a_flow_two_way_and_nothing_more() { ); for status in STATUSES { - assert_eq!( - next_flow_status(&packet, NatAction::SrcNat, status), - status, - "an outbound icmp packet moved a flow in {status:?}" - ); - if status != NatFlowStatus::OneWay { + for action in [NatAction::SrcNat, NatAction::DstNat] { + let next = next_flow_status(&packet, action, status); assert_eq!( - next_flow_status(&packet, NatAction::DstNat, status), - status, - "an inbound icmp packet moved a flow in {status:?}" + Req12::new(status, next).check(), + Ok(()), + "{action} icmp packet terminated a flow in {status:?}" + ); + if action == NatAction::DstNat && status == NatFlowStatus::OneWay { + continue; + } + assert_eq!( + next, status, + "an {action} icmp packet moved a flow in {status:?}" ); } } From f1983560da118f5c14d4027549fa90ea53a5eea4 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Wed, 19 Aug 2026 13:02:41 -0600 Subject: [PATCH 09/21] test(masquerade): Make a stale citation a build failure `Requirement` replaces the inherent `check` method, so the convention cannot be got wrong by the next contract: a trait has one shape, an inherent method has as many as there are authors. `Violation` and its two `&'static str` fields are gone, replaced by a `thiserror` type per requirement carrying the values that broke it -- `development/code/error-handling.md` calls string errors "actively hostile", and nothing forced one here, since `before` and `after` were already in hand. The part worth having is `SPEC` and `ID` being `const`. duvet emits one TOML per specification section under `.duvet/requirements/`, so the tree already holds a machine-readable copy of every requirement tracked, and a `const` assertion over `include_str!` checks that the section a contract names really states the requirement it claims. Changing `REQ-12` to `REQ-42` now fails the build with E0080 rather than passing review. Dropping RFC 4787 from `.duvet/config.toml` would fail it too, and because `include_str!` is recorded in rustc's dependency information, re-extracting a specification rebuilds the check instead of leaving it stale. This is what the citation axis could not do on its own. duvet verifies that a quoted sentence matches the specification; it cannot verify that the code naming that sentence still exists, and nothing verified the reverse direction at all. `unreachable!` rather than `panic!` at the call site, per `development/code/error-handling.md`: reaching it is programmer error. It is guarded on `cfg!(debug_assertions)` first so the check does not run in release, and it names the specification URL, the requirement id and both states, so a failure is readable without opening the file. The trait method cannot be `const fn` on stable, which settles where the build-time tier lives: constraints over `const`s -- RFC 4787 REQ-5 bounds timers that are `const`s -- stay plain `const` assertions and are deliberately not `Requirement`s. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- nat/src/masquerade/contract.rs | 96 ++++++++++++++++++++++++----- nat/src/masquerade/protocol.rs | 14 +++-- nat/src/masquerade/state_machine.rs | 1 + 3 files changed, 91 insertions(+), 20 deletions(-) diff --git a/nat/src/masquerade/contract.rs b/nat/src/masquerade/contract.rs index 6f0945583a..d74d853ea0 100644 --- a/nat/src/masquerade/contract.rs +++ b/nat/src/masquerade/contract.rs @@ -3,20 +3,43 @@ use crate::common::NatFlowStatus; -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(crate) struct Violation { - pub(crate) requirement: &'static str, - pub(crate) detail: &'static str, +pub(crate) trait Requirement { + type Error: core::error::Error; + + const SPEC: &'static str; + + const ID: &'static str; + + fn check(&self) -> Result<(), Self::Error>; } -impl std::fmt::Display for Violation { - fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { - write!(f, "{}: {}", self.requirement, self.detail) +const fn contains(haystack: &str, needle: &str) -> bool { + let (h, n) = (haystack.as_bytes(), needle.as_bytes()); + if n.is_empty() { + return true; + } + if h.len() < n.len() { + return false; } + let mut i = 0; + while i <= h.len() - n.len() { + let mut j = 0; + while j < n.len() && h[i + j] == n[j] { + j += 1; + } + if j == n.len() { + return true; + } + i += 1; + } + false } pub(crate) mod rfc4787 { - use super::{NatFlowStatus, Violation}; + use super::{NatFlowStatus, Requirement, contains}; + + const SECTION_9: &str = + include_str!("../../../.duvet/requirements/www.rfc-editor.org/rfc/rfc4787/section-9.toml"); #[derive(Debug, Clone, Copy)] pub(crate) struct Req12 { @@ -24,30 +47,49 @@ pub(crate) mod rfc4787 { after: NatFlowStatus, } + #[derive(Debug, Clone, Copy, PartialEq, thiserror::Error)] + #[error("an ICMP packet moved a live flow from {before:?} to {after:?}")] + pub(crate) struct Req12Violated { + before: NatFlowStatus, + after: NatFlowStatus, + } + impl Req12 { pub(crate) const fn new(before: NatFlowStatus, after: NatFlowStatus) -> Self { Self { before, after } } - pub(crate) const fn check(self) -> Result<(), Violation> { + const fn terminal(status: NatFlowStatus) -> bool { + matches!(status, NatFlowStatus::Closed | NatFlowStatus::Reset) + } + } + + impl Requirement for Req12 { + type Error = Req12Violated; + const SPEC: &'static str = "https://www.rfc-editor.org/rfc/rfc4787#section-9"; + const ID: &'static str = "REQ-12"; + + fn check(&self) -> Result<(), Self::Error> { if Self::terminal(self.after) && !Self::terminal(self.before) { - return Err(Violation { - requirement: "rfc4787#section-9 REQ-12 / rfc5382#section-8 REQ-10", - detail: "an ICMP packet moved a live flow into a terminal state", + return Err(Req12Violated { + before: self.before, + after: self.after, }); } Ok(()) } - - const fn terminal(status: NatFlowStatus) -> bool { - matches!(status, NatFlowStatus::Closed | NatFlowStatus::Reset) - } } + + const _: () = assert!( + contains(SECTION_9, ::ID), + "rfc4787#section-9 does not state REQ-12" + ); } #[cfg(test)] mod test { use super::rfc4787::Req12; + use super::{Requirement, contains}; use crate::common::NatFlowStatus; const STATUSES: [NatFlowStatus; 10] = [ @@ -63,6 +105,17 @@ mod test { NatFlowStatus::Closed, ]; + #[test] + fn the_specification_search_can_fail() { + assert!(contains("REQ-12: Receipt of any", "REQ-12")); + assert!(!contains("REQ-12: Receipt of any", "REQ-42")); + assert!(!contains("REQ-1", "REQ-12"), "a prefix is not a match"); + assert!( + contains("anything", ""), + "the empty needle is always present" + ); + } + #[test] fn the_contract_rejects_exactly_the_forbidden_transitions() { let terminal = |s| matches!(s, NatFlowStatus::Closed | NatFlowStatus::Reset); @@ -83,4 +136,15 @@ mod test { "8 live statuses times 2 terminal ones must be the whole forbidden set" ); } + + #[test] + fn a_violation_reports_the_transition_that_caused_it() { + let err = Req12::new(NatFlowStatus::Established, NatFlowStatus::Closed) + .check() + .expect_err("an established flow moved to closed must violate REQ-12"); + assert_eq!( + err.to_string(), + "an ICMP packet moved a live flow from Established to Closed" + ); + } } diff --git a/nat/src/masquerade/protocol.rs b/nat/src/masquerade/protocol.rs index 9c487849ac..800159e832 100644 --- a/nat/src/masquerade/protocol.rs +++ b/nat/src/masquerade/protocol.rs @@ -6,6 +6,7 @@ //! for port conservation. use crate::common::{NatAction, NatFlowStatus}; +use crate::masquerade::contract::Requirement; use crate::masquerade::contract::rfc4787::Req12; use net::buffer::PacketBufferMut; use net::headers::{TryHeaders, TryIp, TryTcp}; @@ -75,10 +76,15 @@ fn next_flow_status_icmp(action: NatAction, status: NatFlowStatus) -> NatFlowSta _ => status, }, }; - debug_assert!( - Req12::new(status, next).check().is_ok(), - "{action} {status:?} -> {next:?}" - ); + if cfg!(debug_assertions) + && let Err(violation) = Req12::new(status, next).check() + { + unreachable!( + "{spec} {id}: {violation} ({action})", + spec = Req12::SPEC, + id = Req12::ID + ); + } next } diff --git a/nat/src/masquerade/state_machine.rs b/nat/src/masquerade/state_machine.rs index 2b4d5f29af..3585cb88d4 100644 --- a/nat/src/masquerade/state_machine.rs +++ b/nat/src/masquerade/state_machine.rs @@ -4,6 +4,7 @@ #![cfg(test)] use crate::common::{NatAction, NatFlowStatus}; +use crate::masquerade::contract::Requirement; use crate::masquerade::contract::rfc4787::Req12; use crate::masquerade::protocol::next_flow_status; use net::buffer::TestBuffer; From 6ea720302f2516fd05bbde41a4231ba4fb12f6d4 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Wed, 19 Aug 2026 20:05:23 -0600 Subject: [PATCH 10/21] docs(forwarding): Record RFC 4787 REQ-13, and why it is one decision not three Both clauses are uncited and unheld, and the reason is structural rather than a missing branch: there is no MTU anywhere on this datapath. net::interface::Mtu appears in ten files, all of them control plane -- config, the FRR renderer and interface-manager, which push it to the kernel over netlink. It reaches neither dataplane, pipeline nor nat. Cited at vxlan_encap because that is the one place the dataplane makes a packet larger, which is the condition RFC 4787 section 10 governs. Its only failure modes on size are mbuf headroom and the 2^16 ceiling of the IP length field, and neither is a link MTU. The finding worth taking to the team is the scope. This stage originates no ICMP error at all: TTL expiry drops on DoneReason::HopLimitExceeded where RFC 1812 asks a router for ICMP Time Exceeded, and nat::icmp_handler only translates errors that arrive. REQ-13, REQ-13a and the TTL case are one question -- does this gateway originate ICMP errors? -- whose answer needs an egress MTU first. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- dataplane/src/packet_processor/ipforward.rs | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/dataplane/src/packet_processor/ipforward.rs b/dataplane/src/packet_processor/ipforward.rs index 719d471f22..f41d2dcf2b 100644 --- a/dataplane/src/packet_processor/ipforward.rs +++ b/dataplane/src/packet_processor/ipforward.rs @@ -219,6 +219,15 @@ impl IpForwarder { } /// Encapsulate a packet in Vxlan with the provided [`VxlanEncapsulation`] params + //= https://www.rfc-editor.org/rfc/rfc4787#section-10 + //= type=todo + //# REQ-13: If the packet received on an internal IP address has DF=1, + //# the NAT MUST send back an ICMP message "Fragmentation needed and + //# DF set" to the host, as described in [RFC0792]. + //= https://www.rfc-editor.org/rfc/rfc4787#section-10 + //= type=todo + //# a) If the packet has DF=0, the NAT MUST fragment the packet and + //# SHOULD send the fragments in order. fn vxlan_encap( &self, packet: &mut Packet, From 5701c9ba82b9c154754f359c7aa558311184d288 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 20 Aug 2026 15:11:15 -0600 Subject: [PATCH 11/21] fix(duvet): Regenerate the snapshot the REQ-12 and REQ-13 commits left stale Nothing regenerated it, so the regression gate had drifted two commits after being introduced. `just duvet-check` now catches this. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- .duvet/snapshot.txt | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index 0d495baebf..f79ef0e675 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -77,19 +77,19 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 TEXT[!SHOULD]: host. TEXT[!MUST]: Receipt of any sort of ICMP message MUST NOT TEXT[!MUST]: destroy the NAT mapping. - TEXT[!MUST,implementation]: REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the - TEXT[!MUST,implementation]: NAT mapping. + TEXT[!MUST,implementation,test]: REQ-12: Receipt of any sort of ICMP message MUST NOT terminate the + TEXT[!MUST,implementation,test]: NAT mapping. TEXT[!SHOULD]: a) The NAT's default configuration SHOULD NOT filter ICMP messages TEXT[!SHOULD]: based on their source IP address. TEXT[!SHOULD]: b) It is RECOMMENDED that a NAT support ICMP Destination TEXT[!SHOULD]: Unreachable messages. SECTION: [Fragmentation of Outgoing Packets](#section-10) - TEXT[!MUST]: REQ-13: If the packet received on an internal IP address has DF=1, - TEXT[!MUST]: the NAT MUST send back an ICMP message "Fragmentation needed and - TEXT[!MUST]: DF set" to the host, as described in [RFC0792]. - TEXT[!MUST]: a) If the packet has DF=0, the NAT MUST fragment the packet and - TEXT[!MUST]: SHOULD send the fragments in order. + TEXT[!MUST,todo]: REQ-13: If the packet received on an internal IP address has DF=1, + TEXT[!MUST,todo]: the NAT MUST send back an ICMP message "Fragmentation needed and + TEXT[!MUST,todo]: DF set" to the host, as described in [RFC0792]. + TEXT[!MUST,todo]: a) If the packet has DF=0, the NAT MUST fragment the packet and + TEXT[!MUST,todo]: SHOULD send the fragments in order. SECTION: [Receiving Fragmented Packets](#section-11) TEXT[!MUST,todo]: REQ-14: A NAT MUST support receiving in-order and out-of-order From f2285c98e272ddf635f6a3d29f8bcac8be2c9d30 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Fri, 21 Aug 2026 11:40:55 -0600 Subject: [PATCH 12/21] test(net): Take the RFC 4884 minimum from both sides, in both families The citation claimed more than the test checked, twice over. Refusing 124 does not state "at least 128" -- a check written `<=` refuses a conforming 128-octet field and passes that test -- and the minimum is implemented once per address family, so the ICMPv6 copy had no test at all. Found by `just spec-interlock`, which flips this requirement from decorative to held: 4 surviving mutants to 6 caught. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- net/src/headers/embedded.rs | 58 +++++++++++++++++++++++++++++++++---- 1 file changed, 52 insertions(+), 6 deletions(-) diff --git a/net/src/headers/embedded.rs b/net/src/headers/embedded.rs index 5e3b1f2c99..dd7715123b 100644 --- a/net/src/headers/embedded.rs +++ b/net/src/headers/embedded.rs @@ -968,6 +968,28 @@ mod tests { buf } + fn create_full_ipv6_tcp_packet_with_payload() -> Vec { + let ipv6_header = Ipv6Header { + traffic_class: 0, + flow_label: 0.try_into().unwrap(), + payload_length: 80, + next_header: IpNumber::TCP, + hop_limit: 64, + source: [0x20, 0x01, 0x0d, 0xb8, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1], + destination: [0x20, 0x01, 0x0d, 0xb8, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 2], + }; + + let mut buf = Vec::new(); + ipv6_header.write(&mut buf).unwrap(); + + let tcp_header = etherparse::TcpHeader::new(80, 443, 1000, 0); + tcp_header.write(&mut buf).unwrap(); + + buf.extend_from_slice(&[1u8; 60]); + + buf + } + // Basic parsing, deparsing checks #[test] @@ -1323,6 +1345,17 @@ mod tests { (headers, consumed.get() as usize, buf) } + fn v6_with_field_of(field_len: usize, padding_byte: u8) -> (EmbeddedHeaders, usize, Vec) { + let mut buf = create_full_ipv6_tcp_packet_with_payload(); + assert_eq!(buf.len(), 120, "the embedded packet is 120 octets"); + buf.extend(std::iter::repeat_n(padding_byte, field_len - buf.len())); + assert_eq!(buf.len(), field_len); + buf.extend_from_slice(&[0x55u8; 32]); + let (headers, consumed) = + EmbeddedHeaders::parse_with(EmbeddedIpVersion::Ipv6, &buf).unwrap(); + (headers, consumed.get() as usize, buf) + } + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 //= type=test //# When the ICMP Extension Structure is appended to an ICMPv4 message @@ -1349,13 +1382,26 @@ mod tests { //# and that ICMP message contains an "original datagram" field, the //# "original datagram" field MUST contain at least 128 octets. #[test] - fn a_field_shorter_than_128_octets_is_refused() { - for field_len in [120usize, 124] { - let (mut headers, consumed, buf) = v4_with_field_of(field_len, 0); - headers.check_full_payload(&buf, buf.len(), consumed, field_len); + fn the_128_octet_minimum_is_exact() { + type Fixture = fn(usize, u8) -> (EmbeddedHeaders, usize, Vec); + for (family, build) in [ + ("ICMPv4", v4_with_field_of as Fixture), + ("ICMPv6", v6_with_field_of as Fixture), + ] { + for field_len in [120usize, 124] { + let (mut headers, consumed, buf) = build(field_len, 0); + headers.check_full_payload(&buf, buf.len(), consumed, field_len); + assert!( + !headers.is_full_payload(), + "{family}: a {field_len}-octet field is below the 128-octet minimum" + ); + } + + let (mut headers, consumed, buf) = build(128, 0); + headers.check_full_payload(&buf, buf.len(), consumed, 128); assert!( - !headers.is_full_payload(), - "a {field_len}-octet field is below the 128-octet minimum" + headers.is_full_payload(), + "{family}: 128 octets is the minimum, so a 128-octet field must be accepted" ); } } From 03f3125248e35c11b7db08a7ff07a50a8018bfef Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Fri, 21 Aug 2026 11:55:19 -0600 Subject: [PATCH 13/21] test(nat): Cite port overloading on the code that could commit it REQ-3 and REQ-7 sat on `allocate_v4`, which forwards to `allocate_from_tables` and decides nothing. Every mutant of it was unviable, so the interlock could not check the citation at all -- and a citation that nothing can break is not a claim. Moving it to the bitmap makes the claim checkable, and it immediately fails: of fourteen mutants the cited property catches four. The seven in the second-half path are unreached because no test exhausts 128 ports from one block, and one of those -- `|=` to `&=` -- is port overloading itself. The remaining three divert allocation to the second half but still yield unique ports, so they do not bear on this requirement. Recorded as a decorative citation rather than fixed here: closing it needs the property to drive a half-block dry, which is a change to what the generator produces, not to what it asserts. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- nat/src/masquerade/apalloc/mod.rs | 6 ------ nat/src/masquerade/apalloc/port_alloc.rs | 6 ++++++ 2 files changed, 6 insertions(+), 6 deletions(-) diff --git a/nat/src/masquerade/apalloc/mod.rs b/nat/src/masquerade/apalloc/mod.rs index 4d26ed4dd0..2670d2fcd9 100644 --- a/nat/src/masquerade/apalloc/mod.rs +++ b/nat/src/masquerade/apalloc/mod.rs @@ -288,12 +288,6 @@ impl NatAllocator { //= https://www.rfc-editor.org/rfc/rfc4787#section-4.1 //= type=todo //# REQ-1: A NAT MUST have an "Endpoint-Independent Mapping" behavior. - //= https://www.rfc-editor.org/rfc/rfc5382#section-8 - //# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port - //# overloading" for TCP. - //= https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1 - //# REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port - //# overloading". fn allocate_v4( &self, src_vpcd: VpcDiscriminant, diff --git a/nat/src/masquerade/apalloc/port_alloc.rs b/nat/src/masquerade/apalloc/port_alloc.rs index 76dbe75c7f..5d24873b68 100644 --- a/nat/src/masquerade/apalloc/port_alloc.rs +++ b/nat/src/masquerade/apalloc/port_alloc.rs @@ -834,6 +834,12 @@ impl Bitmap256 { // // In the last example above, we have three trailing ones in the first half, telling us that // port at 1 << 3 (port number 3) is free. + //= https://www.rfc-editor.org/rfc/rfc5382#section-8 + //# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port + //# overloading" for TCP. + //= https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1 + //# REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port + //# overloading". fn allocate_port_from_bitmap(&mut self) -> Result { #[allow(clippy::cast_possible_truncation)] // max value is 128 let ones = self.first_half.trailing_ones() as u16; From b636ceddaa1e30005ab2e824e5ae139cec2e916c Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Fri, 21 Aug 2026 13:16:23 -0600 Subject: [PATCH 14/21] test(nat): Cite the exhaustion walk alongside the stage property One requirement, two tests, because they reach different code. The stage property states port overloading where it is observable -- two flows, one reply path -- but it draws a handful of ports, so it never fills a 256-port block and never enters the second half of the bitmap. Walking a region dry does. Nine of the ten mutants the interlock reported now die, including the one that replaced the bit marking a port used. The test is unchanged: it already asserted this. Only the citation was incomplete, which is a failure mode duvet cannot see -- a requirement can be fully tested and still name the wrong test. Signed-off-by: Daniel Noland Co-Authored-By: Claude Opus 5 (1M context) --- nat/src/masquerade/apalloc/pool_fuzz.rs | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/nat/src/masquerade/apalloc/pool_fuzz.rs b/nat/src/masquerade/apalloc/pool_fuzz.rs index 9f7ee7cc08..274ef4e038 100644 --- a/nat/src/masquerade/apalloc/pool_fuzz.rs +++ b/nat/src/masquerade/apalloc/pool_fuzz.rs @@ -259,6 +259,14 @@ fn re_reservation_after_a_config_change_is_honoured() { }); } +//= https://www.rfc-editor.org/rfc/rfc5382#section-8 +//= type=test +//# REQ-7: A NAT MUST NOT have a "Port assignment" behavior of "Port +//# overloading" for TCP. +//= https://www.rfc-editor.org/rfc/rfc4787#section-4.2.1 +//= type=test +//# REQ-3: A NAT MUST NOT have a "Port assignment" behavior of "Port +//# overloading". #[test] #[cfg_attr(miri, ignore = "exhaustive allocator walk is too slow under miri")] fn a_region_can_be_allocated_dry() { From 68aba4c15c7de5600c4c74ad9d68e25e4b892843 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Wed, 26 Aug 2026 19:40:24 -0600 Subject: [PATCH 15/21] ci(dev): Gate on the compliance snapshot and print its tables duvet is deterministic and takes milliseconds, so it can gate where the interlock -- hours of mutation testing -- cannot. Both steps land in this chapter rather than with the recipes they call, because they need the specifications this one vendors: without `.duvet/` the gate refuses to run. Co-Authored-By: Claude Opus 5 (1M context) Signed-off-by: Daniel Noland --- .github/workflows/dev.yml | 17 +++++++++++++++++ justfile | 3 ++- 2 files changed, 19 insertions(+), 1 deletion(-) diff --git a/.github/workflows/dev.yml b/.github/workflows/dev.yml index da6e25d377..6d09e6e5d5 100644 --- a/.github/workflows/dev.yml +++ b/.github/workflows/dev.yml @@ -390,6 +390,21 @@ jobs: with: recipe: "license-headers" + - name: "duvet-check" + id: "duvet-check" + continue-on-error: true + uses: *just + with: + recipe: "duvet-check" + + - name: "duvet-summary" + id: "duvet-summary" + if: always() + continue-on-error: true + uses: *just + with: + recipe: "duvet-summary" + - name: "Flag any lint failures" if: always() env: @@ -406,6 +421,8 @@ jobs: check-push-filter=${{ steps.check-push-filter.outcome }} check-deps-reuse=${{ steps.check-deps-reuse.outcome }} license-headers=${{ steps.license-headers.outcome }} + duvet-check=${{ steps.duvet-check.outcome }} + duvet-summary=${{ steps.duvet-summary.outcome }} run: | set -euo pipefail status=0 diff --git a/justfile b/justfile index b96bf7b534..d099795855 100644 --- a/justfile +++ b/justfile @@ -729,7 +729,8 @@ lint: \ (nixfmt) \ (check-lint-wiring) \ (check-push-filter) \ - (license-headers) + (license-headers) \ + (duvet-check) {{ _just_debuggable_ }} # Cargo cannot archive doctests, so run them inside the Nix sandbox. From f776ec7739ba66665080144bc720a0d402d5a836 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 27 Aug 2026 01:52:36 -0600 Subject: [PATCH 16/21] fix(net): Read the RFC 4884 length attribute from the ICMP header `payload_length` indexes octet 5 (v4) or 4 (v6) of what it is given, which is where the length attribute sits in an ICMP error message. It was given `cursor.inner` -- the whole frame. `Reader::inner` is never advanced (`consume` only decrements `remaining`) and `Headers::parse` hands it the buffer at the Ethernet header, so the octet read is the last byte of the destination MAC. Every other argument at that call site slices to the unread part; this one did not. So `check_full_payload` has been deciding whether an embedded packet is whole against a MAC address, for exactly the types NAT has to translate -- `supports_extensions` is Destination Unreachable, Time Exceeded and (v4) Parameter Problem. The header is nameable only at the top of `parse_payload`: `Headers::parse` consumed it immediately before, so it is the `size()` octets ending where the unread buffer begins, and the embedded headers are consumed after. Hence taking the reading first. Nothing caught it because nothing exercises the call site. All seventeen uses of `is_full_payload` are in `embedded.rs`'s own tests and every one calls `check_full_payload` directly with a length of its own -- which is also why the RFC 4884 citations on that function pass. The read is bounds-checked rather than argued: a malformed packet must not panic the dataplane. Co-Authored-By: Claude Opus 5 (1M context) Signed-off-by: Daniel Noland --- net/src/icmp4/mod.rs | 13 +++++++++++-- net/src/icmp6/mod.rs | 12 ++++++++++-- 2 files changed, 21 insertions(+), 4 deletions(-) diff --git a/net/src/icmp4/mod.rs b/net/src/icmp4/mod.rs index 5595f80f45..828e4e6e62 100644 --- a/net/src/icmp4/mod.rs +++ b/net/src/icmp4/mod.rs @@ -619,7 +619,9 @@ impl Icmp4 { if !self.supports_extensions() { return 0; } - let payload_length = buf[5]; + let Some(&payload_length) = buf.get(5) else { + return 0; + }; payload_length as usize * 4 } @@ -627,6 +629,13 @@ impl Icmp4 { if !self.is_error_message() { return None; } + + let icmp_payload_length = { + let end = cursor.inner.len() - cursor.remaining as usize; + let start = end.checked_sub(self.size().get() as usize)?; + self.payload_length(&cursor.inner[start..end]) + }; + let (mut headers, consumed) = EmbeddedHeaders::parse_with( EmbeddedIpVersion::Ipv4, &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], @@ -639,7 +648,7 @@ impl Icmp4 { &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], cursor.remaining as usize, consumed.get() as usize, - self.payload_length(cursor.inner), + icmp_payload_length, ); Some(headers) diff --git a/net/src/icmp6/mod.rs b/net/src/icmp6/mod.rs index fc437db4a5..6a7867c913 100644 --- a/net/src/icmp6/mod.rs +++ b/net/src/icmp6/mod.rs @@ -635,7 +635,9 @@ impl Icmp6 { if !self.supports_extensions() { return 0; } - let payload_length = buf[4]; + let Some(&payload_length) = buf.get(4) else { + return 0; + }; payload_length as usize * 8 } @@ -643,6 +645,12 @@ impl Icmp6 { if !self.is_error_message() { return None; } + let icmp_payload_length = { + let end = cursor.inner.len() - cursor.remaining as usize; + let start = end.checked_sub(self.size().get() as usize)?; + self.payload_length(&cursor.inner[start..end]) + }; + let (mut headers, consumed) = EmbeddedHeaders::parse_with( EmbeddedIpVersion::Ipv6, &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], @@ -655,7 +663,7 @@ impl Icmp6 { &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], cursor.remaining as usize, consumed.get() as usize, - self.payload_length(cursor.inner), + icmp_payload_length, ); Some(headers) From 1b9ffbebe3f47d8831215209ff25d7f72cf9e331 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 27 Aug 2026 13:22:40 -0600 Subject: [PATCH 17/21] fix(net): Stop reading a length attribute out of an ICMPv6 pointer RFC 4884 section 3 lists two ICMPv6 messages that may carry an extension structure, not three: An ICMP Extension Structure MAY be appended to ICMPv6 Destination Unreachable, and Time Exceeded messages. The sentence is in .duvet/specifications/, vendored by this PR, and the code beside it said otherwise. ICMPv6 Parameter Problem uses bytes 4 through 7 for its 32-bit Pointer, so treating octet 4 as a length attribute read the top byte of that pointer and multiplied it by eight. The v4 list is right to include Parameter Problem -- section 4.3 gives it a one-octet pointer and a length attribute at octet 5. The two lists differing is what makes this easy to get wrong by symmetry, so the sentence that settles it is now quoted on the function, and duvet refused the first draft of that citation for naming the wrong section. Signed-off-by: Daniel Noland --- .duvet/snapshot.txt | 4 ++-- net/src/icmp6/mod.rs | 7 +++++-- 2 files changed, 7 insertions(+), 4 deletions(-) diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index f79ef0e675..e35c2b0bca 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -175,8 +175,8 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4884 SECTION: [Summary of Changes to ICMP](#section-3) TEXT[!MAY]: An ICMP Extension Structure MAY be appended to ICMPv4 Destination TEXT[!MAY]: Unreachable, Time Exceeded, and Parameter Problem messages. - TEXT[!MAY]: An ICMP Extension Structure MAY be appended to ICMPv6 Destination - TEXT[!MAY]: Unreachable, and Time Exceeded messages. + TEXT[!MAY,implementation]: An ICMP Extension Structure MAY be appended to ICMPv6 Destination + TEXT[!MAY,implementation]: Unreachable, and Time Exceeded messages. TEXT[!MUST,implementation,test]: When the ICMP Extension Structure is appended to an ICMP message TEXT[!MUST,implementation,test]: and that ICMP message contains an "original datagram" field, the TEXT[!MUST,implementation,test]: "original datagram" field MUST contain at least 128 octets. diff --git a/net/src/icmp6/mod.rs b/net/src/icmp6/mod.rs index 6a7867c913..4d2e835d29 100644 --- a/net/src/icmp6/mod.rs +++ b/net/src/icmp6/mod.rs @@ -621,12 +621,15 @@ impl Icmp6 { }) } + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 + //= type=implementation + //# An ICMP Extension Structure MAY be appended to ICMPv6 Destination + //# Unreachable, and Time Exceeded messages. #[must_use] pub(crate) fn supports_extensions(&self) -> bool { - // See RFC 4884. matches!( self.icmp_type(), - Icmp6Type::DestUnreachable(_) | Icmp6Type::TimeExceeded(_) | Icmp6Type::ParamProblem(_) + Icmp6Type::DestUnreachable(_) | Icmp6Type::TimeExceeded(_) ) } From ec2ac71ff7bd91d55dc214b87817593ee6e73034 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 27 Aug 2026 15:26:44 -0600 Subject: [PATCH 18/21] test(routing): Stop spending the reassembly test's budget on its setup `a_large_answer_arrives_whole` announced eight thousand routes before reaching its subject, and `announce_routes` is a send and a `recv` per route against `PATIENCE` as a per-`recv` socket timeout. On a loaded runner one of those reads missed its ten-second window and the test failed in `CpiPeer::recv`, having never exercised reassembly at all -- in `check/debug`, not only under coverage. Tuning the count against a measurement does not fix that: the next runner is slower or busier and the measurement is stale. Making the setup cheap enough that it is not the thing under time pressure does. 256 routes is fourteen chunks and thirty-two times less work, with the same assertions. Nothing is lost. The failures worth catching -- a reassembly that loses its place, a "more" flag set from the wrong end of the loop -- show in a handful of chunks, and the boundary that would be interesting, `cli_wake_on_writeable`, needs about a hundred and fifty thousand routes. Eight thousand did not come close either, which the doc comment already said. The bar is now stated in chunks rather than as `100 * 2048`, since chunks are what reassembly loops over. Signed-off-by: Daniel Noland --- routing/src/router/rio.rs | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/routing/src/router/rio.rs b/routing/src/router/rio.rs index a661ef7ff4..bd09af14b2 100644 --- a/routing/src/router/rio.rs +++ b/routing/src/router/rio.rs @@ -1475,7 +1475,9 @@ mod tests { #[test] #[cfg_attr(emulated, ignore = "binds Unix domain sockets")] fn a_large_answer_arrives_whole() { - const ROUTES: usize = 8192; + const ROUTES: usize = 256; + const CHUNK: usize = 2048; + const LEAST_CHUNKS: usize = 6; let rio = RunningRio::start(); let peer = CpiPeer::attach(&rio.dir); @@ -1499,9 +1501,10 @@ mod tests { .expect("the whole answer should arrive, across as many chunks as it takes"); let body = answer.result.expect("the listing should succeed"); + let chunks = body.len().div_ceil(CHUNK); assert!( - body.len() > 100 * 2048, - "the answer must span many chunks for this to test reassembly, got {} octets", + chunks >= LEAST_CHUNKS, + "the answer spans {chunks} chunks, too few to exercise reassembly ({} octets)", body.len() ); assert!(body.contains("10.0.0.0/24"), "the first route is missing"); From 135acfab2b8dc3185541063c4ae5158ab37ce7b9 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 27 Aug 2026 15:57:53 -0600 Subject: [PATCH 19/21] fix(nat): Let outbound traffic refresh a mapping nobody has answered `NatFlowStatus::OneWay` means two different things and got one rule. For UDP and ICMP it is the *steady* state of a flow that never gets a reply -- `next_flow_status_udp` leaves it only on an inbound packet -- so it is where syslog, netflow, telemetry and an unanswered resolver query live for their whole lives. Returning no extension tore such a flow down five seconds after its **first** packet however much it sent, and drew a fresh public port each time it was rebuilt. That is RFC 4787 REQ-6's outbound refresh behaviour being "False" in the one state where a UDP mapping actually lives, and the citation on that arm becomes an implementation rather than a todo. For TCP it is a half-open connection -- `next_flow_status_tcp` leaves it only on a SYN-ACK -- so an outbound packet is a retransmitted SYN. That is evidence nobody answered rather than evidence the connection exists, and refreshing on it would let a half-open connection hold a public tuple for as long as the sender retries. RFC 4787 is the UDP document and does not ask for it. The interval is unchanged at five seconds, and the REQ-5 deviation that makes it short against a two-minute floor stays a todo: the defect was measuring the leash from the first packet rather than the last, and lengthening it is a separate decision. The regression test needed two attempts, and the reason is in its doc comment: the allocator is deterministic under `set_randomize(false)`, so a flow that is torn down and rebuilt is handed the same tuple back and every outbound-side observable looks identical either way. What separates survival from rebuild is an inbound packet after a gap, past the deadline the first packet set and inside the one the last packet set. Signed-off-by: Daniel Noland --- .duvet/snapshot.txt | 4 +-- nat/src/masquerade/expiry.rs | 64 ++++++++++++++++++++++++++++++++++++ nat/src/masquerade/nf.rs | 16 ++++++--- 3 files changed, 77 insertions(+), 7 deletions(-) diff --git a/.duvet/snapshot.txt b/.duvet/snapshot.txt index e35c2b0bca..5c12884204 100644 --- a/.duvet/snapshot.txt +++ b/.duvet/snapshot.txt @@ -27,8 +27,8 @@ SPECIFICATION: https://www.rfc-editor.org/rfc/rfc4787 TEXT[!MAY]: b) The value of the NAT UDP mapping timer MAY be configurable. TEXT[!SHOULD,todo]: c) A default value of five minutes or more for the NAT UDP mapping TEXT[!SHOULD,todo]: timer is RECOMMENDED. - TEXT[!MUST,test,todo]: REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound - TEXT[!MUST,test,todo]: refresh behavior" of "True". + TEXT[!MUST,implementation,test]: REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound + TEXT[!MUST,implementation,test]: refresh behavior" of "True". TEXT[!MAY]: a) The NAT mapping Refresh Direction MAY have a "NAT Inbound TEXT[!MAY]: refresh behavior" of "True". diff --git a/nat/src/masquerade/expiry.rs b/nat/src/masquerade/expiry.rs index 6d8f6a814e..e1b20ac50a 100644 --- a/nat/src/masquerade/expiry.rs +++ b/nat/src/masquerade/expiry.rs @@ -347,6 +347,70 @@ fn an_expired_flow_is_never_resurrected() { }); } +//= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 +//= type=test +//# REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound +//# refresh behavior" of "True". +#[test] +fn outbound_traffic_keeps_an_unanswered_mapping_alive() { + const STEP: Duration = Duration::from_secs(2); + const REFRESHES: u32 = 2; + + let source: IpAddr = "10.0.0.21".parse().unwrap_or_else(|_| unreachable!()); + let elapsed = Duration::from_secs(u64::from(REFRESHES + 1) * STEP.as_secs()); + assert!( + elapsed > crate::Masquerade::MASQUERADE_ONEWAY_TIMEOUT, + "the probe must land past the deadline the first packet set, or neither half proves \ + anything" + ); + + with_paused_clock(|| async { + let (fabric, _) = fabric(); + let (mut lookup, mut masq) = fabric.stages(); + let peer = fabric.peer[0]; + + let translated = open_flow(&mut lookup, &mut masq, source, peer, 5300) + .unwrap_or_else(|| unreachable!("a fixed private source is masqueraded")); + for _ in 0..=REFRESHES { + advance(STEP).await; + } + assert_eq!( + reply_to(&mut lookup, &mut masq, peer, translated), + None, + "a mapping nobody refreshed survived {}s of silence against a {}s timeout, so the \ + treatment below proves nothing", + elapsed.as_secs(), + crate::Masquerade::MASQUERADE_ONEWAY_TIMEOUT.as_secs() + ); + }); + + with_paused_clock(|| async { + let (fabric, _) = fabric(); + let (mut lookup, mut masq) = fabric.stages(); + let peer = fabric.peer[0]; + + let translated = open_flow(&mut lookup, &mut masq, source, peer, 5300) + .unwrap_or_else(|| unreachable!("a fixed private source is masqueraded")); + for _ in 0..REFRESHES { + advance(STEP).await; + assert_eq!( + open_flow(&mut lookup, &mut masq, source, peer, 5300), + Some(translated), + "the sender was given a different public tuple mid-stream" + ); + } + advance(STEP).await; + assert_eq!( + reply_to(&mut lookup, &mut masq, peer, translated), + Some(source), + "outbound traffic did not keep an unanswered mapping alive: at t={}s the flow was \ + gone, so a one-way sender loses its public tuple every {}s however much it sends", + elapsed.as_secs(), + crate::Masquerade::MASQUERADE_ONEWAY_TIMEOUT.as_secs() + ); + }); +} + #[test] fn both_halves_of_a_pair_outlive_one_sided_traffic() { with_paused_clock(|| async { diff --git a/nat/src/masquerade/nf.rs b/nat/src/masquerade/nf.rs index 0b33c64afe..4fd6fbd801 100644 --- a/nat/src/masquerade/nf.rs +++ b/nat/src/masquerade/nf.rs @@ -176,6 +176,15 @@ impl Masquerade { packet.meta().dst_vpcd } + fn refreshes_while_unanswered(packet: &Packet) -> bool { + packet.try_ip().is_some_and(|ip| { + matches!( + ip.next_header(), + NextHeader::UDP | NextHeader::ICMP | NextHeader::ICMP6 + ) + }) + } + /// Update the `FlowStatus` of a masqueraded flow with a packet, depending on the direction of the /// communication and the protocol and extend the lifetime of the flow (or invalidate it) accordingly. fn refresh_masquerade_state( @@ -203,14 +212,11 @@ impl Masquerade { | NatFlowStatus::SHalfClose | NatFlowStatus::LastAck => Some(Self::MASQUERADE_CLOSING_TIMEOUT), //= https://www.rfc-editor.org/rfc/rfc4787#section-4.3 - //= type=todo + //= type=implementation //# REQ-6: The NAT mapping Refresh Direction MUST have a "NAT Outbound //# refresh behavior" of "True". NatFlowStatus::OneWay => { - // this could happen if a burst of packets are sent before any state is there (snat), - // or if we got a TCP segment back without expected flags. This should never happen for - // a UDP packet in the reverse direction, though. - None + Self::refreshes_while_unanswered(packet).then_some(Self::MASQUERADE_ONEWAY_TIMEOUT) } }; From 002c081e68fd60115218dd2d665155fdcf4ccebb Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 27 Aug 2026 16:36:35 -0600 Subject: [PATCH 20/21] fix(net): Measure the quoted datagram from where its lengths are measured `check_full_payload` compares two lengths -- `full_packet_length`, derived from the embedded IP header, and the RFC 4884 length attribute -- and both are offsets from the *start* of the original datagram. The buffer and remaining count it was given began `consumed` octets later, after `cursor.consume` had moved past the embedded headers. So the padding check indexed `buf[full_packet_length..icmp_length]` past the end of the field and into whatever followed it, and the no-extension case compared a whole-datagram length against a headers-excluded remainder. `parse_with` on the line above was already being handed the right slice; it is now shared. Two commits have corrected arguments at this call site now -- the length attribute read from the Ethernet header, and this -- and neither was catchable by the tests that existed. All seventeen uses of `check_full_payload` call it directly with a buffer and lengths of their own, so what the caller passes was never exercised. `an_icmp_error_from_the_wire_reports_a_full_payload` starts from a frame instead, which is the only place either defect is visible; it fails on the previous window and would have failed on the previous length. Still latent either way: `is_full_payload` has no caller in the workspace. Signed-off-by: Daniel Noland --- net/src/headers/embedded.rs | 33 +++++++++++++++++++++++++++++++++ net/src/icmp4/mod.rs | 15 ++++++++------- net/src/icmp6/mod.rs | 15 ++++++++------- 3 files changed, 49 insertions(+), 14 deletions(-) diff --git a/net/src/headers/embedded.rs b/net/src/headers/embedded.rs index dd7715123b..7e12f9f918 100644 --- a/net/src/headers/embedded.rs +++ b/net/src/headers/embedded.rs @@ -1356,6 +1356,39 @@ mod tests { (headers, consumed.get() as usize, buf) } + #[test] + fn an_icmp_error_from_the_wire_reports_a_full_payload() { + use crate::headers::TryEmbeddedHeaders; + use crate::ip::NextHeader; + use crate::packet::test_utils::build_test_icmp4_destination_unreachable_packet; + + let packet = build_test_icmp4_destination_unreachable_packet( + "10.0.0.1".parse().unwrap_or_else(|_| unreachable!()), + "10.0.0.2".parse().unwrap_or_else(|_| unreachable!()), + "192.168.0.1".parse().unwrap_or_else(|_| unreachable!()), + "192.168.0.2".parse().unwrap_or_else(|_| unreachable!()), + NextHeader::UDP, + 1234, + 80, + ) + .unwrap_or_else(|e| unreachable!("{e:?}")); + + let embedded = packet + .embedded_headers() + .unwrap_or_else(|| unreachable!("an icmp error carries embedded headers")); + assert!( + embedded.is_full_payload(), + "the quoted datagram is complete and nothing follows it, so the whole payload is \ + present -- reading this as truncated means the window handed to check_full_payload \ + does not start where the lengths it compares are measured from" + ); + assert_eq!( + embedded.payload_length(), + Some(0), + "the quoted UDP datagram carries no payload beyond its header" + ); + } + //= https://www.rfc-editor.org/rfc/rfc4884#section-3 //= type=test //# When the ICMP Extension Structure is appended to an ICMPv4 message diff --git a/net/src/icmp4/mod.rs b/net/src/icmp4/mod.rs index 828e4e6e62..8c94855edb 100644 --- a/net/src/icmp4/mod.rs +++ b/net/src/icmp4/mod.rs @@ -636,17 +636,18 @@ impl Icmp4 { self.payload_length(&cursor.inner[start..end]) }; - let (mut headers, consumed) = EmbeddedHeaders::parse_with( - EmbeddedIpVersion::Ipv4, - &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], - ) - .ok()?; + let embedded_start = cursor.inner.len() - cursor.remaining as usize; + let embedded_remaining = cursor.remaining as usize; + + let (mut headers, consumed) = + EmbeddedHeaders::parse_with(EmbeddedIpVersion::Ipv4, &cursor.inner[embedded_start..]) + .ok()?; cursor.consume(consumed).ok()?; // Mark whether the payload of the embedded IP packet is full headers.check_full_payload( - &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], - cursor.remaining as usize, + &cursor.inner[embedded_start..], + embedded_remaining, consumed.get() as usize, icmp_payload_length, ); diff --git a/net/src/icmp6/mod.rs b/net/src/icmp6/mod.rs index 4d2e835d29..eb7bf8ac2b 100644 --- a/net/src/icmp6/mod.rs +++ b/net/src/icmp6/mod.rs @@ -654,17 +654,18 @@ impl Icmp6 { self.payload_length(&cursor.inner[start..end]) }; - let (mut headers, consumed) = EmbeddedHeaders::parse_with( - EmbeddedIpVersion::Ipv6, - &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], - ) - .ok()?; + let embedded_start = cursor.inner.len() - cursor.remaining as usize; + let embedded_remaining = cursor.remaining as usize; + + let (mut headers, consumed) = + EmbeddedHeaders::parse_with(EmbeddedIpVersion::Ipv6, &cursor.inner[embedded_start..]) + .ok()?; cursor.consume(consumed).ok()?; // Mark whether the payload of the embedded IP packet is full headers.check_full_payload( - &cursor.inner[cursor.inner.len() - cursor.remaining as usize..], - cursor.remaining as usize, + &cursor.inner[embedded_start..], + embedded_remaining, consumed.get() as usize, icmp_payload_length, ); From 305015f4fef61288c1f5db5d5ec4411a1e5aa649 Mon Sep 17 00:00:00 2001 From: Daniel Noland Date: Thu, 27 Aug 2026 16:45:12 -0600 Subject: [PATCH 21/21] fix(nat): Refuse a port-forwarding ruleset the table cannot hold `validate_ruleset` returned `Ok(())` unconditionally, so `update_table` was infallible. The rules are installed later, inside `Absorb::absorb_first`, which returns nothing -- so a rule `add_entry` refused was logged and dropped while the caller was told the update succeeded, and the table held fewer rules than the configuration asked for. `mgmt`'s "whatever the validator accepts, the dataplane can enact" property asserts on that `Result`. Its port-forwarding leg has never been able to fail. `PortFwTable::dry_run` answers the question the writer needs to ask, and answers it where `add_entry` lives. A scratch table is equivalent to the real update because `update` removes every entry the incoming ruleset does not contain before adding, so whatever survives is a subset and re-adding a subset takes the "identical except for the timers" path. What is left to catch is a ruleset that disagrees with itself. Running the property suite against the live check found nothing, which is the answer worth having: the validator was not letting overlapping rulesets through, and now that is asserted rather than assumed. Two tests, because one is not enough and the reason is the same one that let this sit: `a_self_overlapping_ruleset_is_refused_up_front` covers `dry_run` and passes whether or not anything calls it, so `a_ruleset_the_table_cannot_hold_is_refused` goes through `PortFwTableWriter::update_table` instead. Only the second fails when `validate_ruleset` is put back the way it was. The absorb path keeps its log line, now naming the rule and the error it was given -- it was binding the error and never printing it -- against the case where the two checks ever disagree. Signed-off-by: Daniel Noland --- nat/src/portfw/portfwtable/access.rs | 40 ++++++++++++++++++++++---- nat/src/portfw/portfwtable/objects.rs | 41 ++++++++++++++++++++++++++- 2 files changed, 75 insertions(+), 6 deletions(-) diff --git a/nat/src/portfw/portfwtable/access.rs b/nat/src/portfw/portfwtable/access.rs index 400e28a701..71fc616534 100644 --- a/nat/src/portfw/portfwtable/access.rs +++ b/nat/src/portfw/portfwtable/access.rs @@ -29,11 +29,8 @@ impl Absorb for PortFwTable { pub struct PortFwTableWriter(WriteHandle); pub struct PortFwTableReader(ReadHandle); -#[allow(clippy::unnecessary_wraps)] -fn validate_ruleset(_ruleset: &[PortFwEntry]) -> Result<(), PortFwTableError> { - // deferring the implementation of this since it will change - // when we introduce port ranges - Ok(()) +fn validate_ruleset(ruleset: &[PortFwEntry]) -> Result<(), PortFwTableError> { + PortFwTable::dry_run(ruleset) } impl PortFwTableWriter { @@ -122,6 +119,39 @@ mod test { .unwrap() } + #[test] + fn a_ruleset_the_table_cannot_hold_is_refused() { + let rule = |ext: (u16, u16), int: (u16, u16)| { + PortFwEntry::new( + PortFwKey::new( + VpcDiscriminant::VNI(2000.try_into().unwrap()), + NextHeader::TCP, + ), + VpcDiscriminant::VNI(3000.try_into().unwrap()), + Prefix::from_str("70.71.72.73/32").unwrap(), + Prefix::from_str("192.168.1.1/32").unwrap(), + ext, + int, + None, + None, + ) + .unwrap() + }; + + let mut writer = PortFwTableWriter::new(); + writer + .update_table(&[rule((3000, 3009), (30, 39)), rule((3010, 3019), (40, 49))]) + .expect("rules that do not overlap are installable together"); + + let refused = + writer.update_table(&[rule((3000, 3009), (30, 39)), rule((3005, 3014), (50, 59))]); + assert!( + refused.is_err(), + "a ruleset whose rules claim overlapping external ports was accepted by the writer, \ + so the table now holds fewer rules than the caller believes it asked for" + ); + } + #[test] #[cfg_attr(not(emulated), traced_test)] fn test_port_forwarding_access_remove_rules_drops_refs() { diff --git a/nat/src/portfw/portfwtable/objects.rs b/nat/src/portfw/portfwtable/objects.rs index bbf13cd43c..b6e177d462 100644 --- a/nat/src/portfw/portfwtable/objects.rs +++ b/nat/src/portfw/portfwtable/objects.rs @@ -252,6 +252,14 @@ impl PortFwTable { Self::default() } + pub(crate) fn dry_run(ruleset: &[PortFwEntry]) -> Result<(), PortFwTableError> { + let mut scratch = Self::default(); + for rule in ruleset { + scratch.add_entry(Arc::new(rule.clone()))?; + } + Ok(()) + } + /// Add a `Arc` to this `PortFwTable`. fn add_entry(&mut self, entry: Arc) -> Result<(), PortFwTableError> { let key = &entry.key; @@ -294,7 +302,7 @@ impl PortFwTable { let mut ruleset = ruleset.to_vec(); while let Some(rule) = ruleset.pop().map(Arc::from) { if let Err(e) = self.add_entry(rule.clone()) { - error!("Failure adding port-forwarding rule (config validation failed)"); + error!("Dropping port-forwarding rule {rule}: {e}"); } } } @@ -494,6 +502,37 @@ mod test { assert_eq!(fwtable.0.len(), 1); } + #[test] + fn a_self_overlapping_ruleset_is_refused_up_front() { + let key = PortFwKey { + src_vpcd: VpcDiscriminant::VNI(2000.try_into().unwrap()), + proto: NextHeader::TCP, + }; + let rule = |ext_ports: (u16, u16), int_ports: (u16, u16)| { + PortFwEntry::new( + key, + VpcDiscriminant::VNI(3000.try_into().unwrap()), + Prefix::from_str("70.71.72.73/32").unwrap(), + Prefix::from_str("192.168.1.1/32").unwrap(), + ext_ports, + int_ports, + None, + None, + ) + .unwrap() + }; + + PortFwTable::dry_run(&[rule((3000, 3009), (30, 39)), rule((3010, 3019), (40, 49))]) + .expect("two rules that do not overlap are installable together"); + + let refused = + PortFwTable::dry_run(&[rule((3000, 3009), (30, 39)), rule((3005, 3014), (50, 59))]); + assert!( + matches!(refused, Err(PortFwTableError::OverlappingRange(_))), + "a ruleset whose rules claim overlapping external ports was accepted: {refused:?}" + ); + } + #[test] fn test_port_forwarding_entry_reject_distinct_ip_ver() { let key = PortFwKey {