From 8bbbb30f1b5668da8217972a69e7eba66d4c534d Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 1 Oct 2026 09:03:26 +0200 Subject: [PATCH 001/125] fix(daemon): harden socket attribution and local flow policy --- README.md | 33 ++-- crates/cfc-daemon/src/config.rs | 5 +- crates/cfc-daemon/src/nfqueue.rs | 181 +++++++++++++++-- crates/cfc-daemon/src/process_resolve.rs | 240 +++++++++++++++++------ crates/cfc-daemon/src/sock_diag.rs | 42 +++- docs/HARDENING.md | 48 +++-- systemd/nftables-snippet.conf | 4 +- 7 files changed, 434 insertions(+), 119 deletions(-) diff --git a/README.md b/README.md index 7aec5b6..14f247e 100644 --- a/README.md +++ b/README.md @@ -171,7 +171,7 @@ Enable the installed daemon and enforcement in First run below. ## First run A fresh install has **zero rules**: once enforcement is on, every new -outbound connection prompts (or falls back to the profile default). Do +remote outbound connection prompts (or falls back to the profile default). Do these three things, in order: **1. Enable enforcement persistently.** A companion unit loads the @@ -204,11 +204,10 @@ systemd-timesyncd and chronyd NTP (:123/udp), the DHCP clients (dhcpcd, NetworkManager and systemd-networkd, :67 and :547/udp), pacman and paru HTTPS mirrors (:443/tcp), and the SSH client (:22/tcp) - and is idempotent (already-present rules are skipped by name; `--dry-run` -previews). **Do not skip this step.** No profile allows anything on its -own, so on a machine with no rules and no UI connected nothing outbound -gets through - including the DHCP lease. Filtering starts before the -network is configured (see below), and these rules are what let the -machine come up at all. +previews). **Do not skip this step.** No profile allows unmatched remote flows +on its own. With no rules and no UI connected, unmatched queued remote +connections are denied. Filtering starts before the network is configured +(see below), and these rules keep DHCP, DNS and NTP usable. For everything else, there are bundles: @@ -240,8 +239,8 @@ On a headless machine, answer them from the terminal instead: cfc prompts ``` -With no subscriber at all the daemon applies `no_ui_action` to every -unmatched flow without asking anyone. **That is a denial under every +With no subscriber at all the daemon applies `no_ui_action` to unmatched +remote flows without asking anyone. **That is a denial under every profile.** "Nobody is connected" is a permanent condition on a headless box, not a passing one, and answering it with an allow would mean those hosts had no outbound firewall whatsoever. Stored rules are what such a @@ -267,11 +266,19 @@ initramfs, interfaces already configured before these units, other network managers, or a later external ruleset flush. Early unmatched flows use `no_ui_action`; bootstrap DHCP/DNS/NTP rules keep strict configurations usable. -**Scope.** Rules decide new tracked flows; established and related traffic -retains its connection-wide authorization. Passed or inherited sockets and -local DNS/proxy relays are not confined to their original executable. -Loopback is exempt, and packet-layer traffic from applications with -`CAP_NET_RAW` is outside these IP hooks. Use OS containment for those cases. +**Scope.** Normal mode decides new tracked IP flows from socket attribution; +established and related traffic retains its connection-wide authorization. +Passed or inherited sockets are not reauthorized for each sending executable. +A current descriptor holder does not prove which process sent a packet. +New direct loopback flows follow explicit rules; unmatched local IPC is allowed +without prompting. An allowed local resolver or proxy can still relay remote +traffic. CFC cannot establish the originating application's identity from +remote flows delegated through local brokers, including AF_UNIX and D-Bus. + +Applications with `CAP_NET_RAW` can use AF_PACKET outside the `inet OUTPUT` +hook. Raw IP packets can also coincide with another socket's tuple; socket +attribution does not prove their origin. Use explicit application confinement +or OS containment for those cases. Fast Allow is disabled even when `fast_allow = true` is configured; allowed flows use the normal NFQUEUE path. diff --git a/crates/cfc-daemon/src/config.rs b/crates/cfc-daemon/src/config.rs index 1cdda51..06a019d 100644 --- a/crates/cfc-daemon/src/config.rs +++ b/crates/cfc-daemon/src/config.rs @@ -156,8 +156,9 @@ impl Profile { /// /// The outbound table cannot lock an operator out of a remote machine: it /// hooks `output` on `ct state new` only, so an inbound SSH session's - /// replies are `ct state established` and are never queued, and loopback is - /// accepted outright. Rules can still be added with `cfc-cli` from that + /// replies are `ct state established` and are never queued. New loopback + /// flows follow explicit policy; unmatched local IPC is allowed without + /// prompting. Rules can still be added with `cfc-cli` from that /// session. What it *does* mean on a fresh headless install is that /// outbound traffic — package updates, NTP, backups — is denied until /// rules exist for it. diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index b7f2266..f119df1 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -240,6 +240,8 @@ trait PacketMessage { /// addresses - and guessing would be wrong on a multi-homed or routed /// host. This is what makes one queue able to serve both chains. fn hook(&self) -> u8; + /// Kernel output interface index; zero means it was not reported. + fn outdev(&self) -> u32; fn set_verdict(&mut self, verdict: NfqVerdict); } @@ -278,6 +280,10 @@ impl PacketMessage for Message { self.get_hook() } + fn outdev(&self) -> u32 { + self.get_outdev() + } + fn set_verdict(&mut self, verdict: NfqVerdict) { Message::set_verdict(self, verdict); } @@ -351,6 +357,15 @@ pub fn spawn( !cfg.fail_open, "NFQUEUE fail_open bypasses mandatory verdict auditing" ); + // Interface metadata, rather than destination addresses, includes every + // local host address routed over lo. A missing index cannot grant access. + // SAFETY: the C string is terminated and valid for this read-only query. + let loopback_ifindex = unsafe { libc::if_nametoindex(c"lo".as_ptr()) }; + anyhow::ensure!( + loopback_ifindex != 0, + "resolving loopback output interface: {}", + std::io::Error::last_os_error() + ); let queue_num = cfg.queue_num; info!(queue_num, "opening NFQUEUE"); @@ -408,6 +423,7 @@ pub fn spawn( let stop = Arc::new(AtomicBool::new(false)); let worker = Worker { queue, + loopback_ifindex, engine, rejecter, prompt_tx, @@ -533,6 +549,7 @@ impl Default for Tuning { /// for why the earlier blocking-when-idle mode had to go). struct Worker { queue: Q, + loopback_ifindex: u32, engine: Engine, /// Injects the TCP RST / ICMP port-unreachable that makes /// [`Action::Reject`] differ from [`Action::Deny`]. Inert (drop-only) @@ -717,6 +734,9 @@ impl Worker { uid: msg.uid(), gid: msg.gid(), direction: direction_for_hook(msg.hook()), + loopback: msg.hook() == NF_INET_LOCAL_OUT + && self.loopback_ifindex != 0 + && msg.outdev() == self.loopback_ifindex, }; let deps = PipelineDeps { engine: &self.engine, @@ -950,7 +970,7 @@ impl Worker { /// live /proc. trait ProcessResolver { #[allow(clippy::too_many_arguments)] // socket tuple plus kernel UID attribution - fn pid_for_socket( + fn socket_owner( &self, protocol: Protocol, direction: Direction, @@ -959,15 +979,15 @@ trait ProcessResolver { dst_ip: IpAddr, dst_port: u16, uid: Option, - ) -> Option; - fn resolve(&self, pid: u32) -> Process; + ) -> Option; + fn resolve(&self, owner: &process_resolve::SocketOwner) -> Process; } /// Production resolver backed by /proc. struct ProcfsResolver; impl ProcessResolver for ProcfsResolver { - fn pid_for_socket( + fn socket_owner( &self, protocol: Protocol, direction: Direction, @@ -976,14 +996,12 @@ impl ProcessResolver for ProcfsResolver { dst_ip: IpAddr, dst_port: u16, uid: Option, - ) -> Option { - process_resolve::pid_for_socket( - protocol, direction, src_ip, src_port, dst_ip, dst_port, uid, - ) + ) -> Option { + process_resolve::socket_owner(protocol, direction, src_ip, src_port, dst_ip, dst_port, uid) } - fn resolve(&self, pid: u32) -> Process { - process_resolve::resolve(pid) + fn resolve(&self, owner: &process_resolve::SocketOwner) -> Process { + owner.resolve() } } @@ -1021,6 +1039,8 @@ struct PacketMeta { gid: Option, /// Which way this packet is going, from the netfilter hook. direction: Direction, + /// Kernel OUTPUT interface is lo; includes local non-loopback addresses. + loopback: bool, } /// Environment for [`handle_packet`]. @@ -1081,7 +1101,7 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack // Inbound is not asked, rather than asked and told nothing. // - // `pid_for_socket` searches for a socket already holding this 4-tuple. + // `socket_owner` searches for a socket already holding this 4-tuple. // An inbound SYN has none - nothing has accepted it yet - so the search // could only ever miss, and missing meant reading /proc/net/tcp and // /proc/net/tcp6: 2.40 ms per packet for an answer of `None`. @@ -1090,10 +1110,10 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack // one is silent: the verdict is identical either way, so nothing would // have failed - the firewall would just be fourteen times slower on the // inbound side and say nothing about it. - let pid_hint = if conn.direction == Direction::Inbound { + let owner = if conn.direction == Direction::Inbound { None } else { - deps.resolver.pid_for_socket( + deps.resolver.socket_owner( conn.protocol, conn.direction, conn.src_ip, @@ -1103,6 +1123,7 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack meta.uid, ) }; + let pid_hint = owner.as_ref().map(process_resolve::SocketOwner::pid); // Always allow our own traffic. Otherwise the daemon's reverse DNS // resolver would itself be intercepted, deadlocking on a verdict @@ -1113,8 +1134,8 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack } } - let mut proc = match pid_hint { - Some(pid) => deps.resolver.resolve(pid), + let mut proc = match owner.as_ref() { + Some(owner) => deps.resolver.resolve(owner), None => Process::unknown(0), }; // The kernel-reported socket uid/gid come from the sk_buff itself and @@ -1174,6 +1195,15 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack verdict, }; } + // Preserve desktop IPC for unmatched local flows after explicit + // policy and incomplete-identity refusals have had their say. + if meta.loopback { + return PacketOutcome::Deliver { + connection: conn, + process: proc, + verdict: Verdict::default_allow(), + }; + } if deps.stats.is_paused() { // Paused means "stop prompting", not "stop filtering": // rules above still applied; only unmatched flows pass @@ -1291,7 +1321,7 @@ mod tests { } impl ProcessResolver for StubResolver { - fn pid_for_socket( + fn socket_owner( &self, _protocol: Protocol, _direction: Direction, @@ -1300,13 +1330,13 @@ mod tests { _dst_ip: IpAddr, _dst_port: u16, _uid: Option, - ) -> Option { + ) -> Option { self.socket_lookups .fetch_add(1, std::sync::atomic::Ordering::Relaxed); - self.pid + self.pid.map(process_resolve::SocketOwner::for_test) } - fn resolve(&self, _pid: u32) -> Process { + fn resolve(&self, _owner: &process_resolve::SocketOwner) -> Process { self.process.clone() } } @@ -1346,6 +1376,7 @@ mod tests { uid: None, gid: None, direction: Direction::Outbound, + loopback: false, }; /// The same, for a packet the kernel queued from the input hook. @@ -1353,6 +1384,7 @@ mod tests { uid: None, gid: None, direction: Direction::Inbound, + loopback: false, }; struct TestEnv { @@ -1960,7 +1992,7 @@ mod tests { let meta = PacketMeta { uid: Some(0), gid: Some(0), - direction: Direction::Outbound, + ..NO_META }; match env.handle(&tcp_packet(443), &meta) { PacketOutcome::Prompt { @@ -1984,7 +2016,7 @@ mod tests { let meta = PacketMeta { uid: Some(1000), gid: None, - direction: Direction::Outbound, + ..NO_META }; match env.handle(&tcp_packet(443), &meta) { PacketOutcome::Prompt { @@ -2189,6 +2221,7 @@ mod tests { uid: Option, gid: Option, hook: u8, + outdev: u32, verdict: Option, } @@ -2200,12 +2233,17 @@ mod tests { uid: None, gid: None, hook: NF_INET_LOCAL_OUT, + outdev: 0, verdict: None, } } } impl PacketMessage for FakeMsg { + fn outdev(&self) -> u32 { + self.outdev + } + fn hook(&self) -> u8 { self.hook } @@ -2324,6 +2362,7 @@ mod tests { script: script.into(), log: log.clone(), }, + loopback_ifindex: 1, engine: Engine::new(RuleSet { rules }, Arc::new(std::sync::RwLock::new(policy))), rejecter: Rejecter::open(), prompt_tx, @@ -2555,6 +2594,106 @@ mod tests { assert_eq!(h.stats.connections_denied(), 2); } + #[test] + fn loopback_unmatched_flows_keep_the_nonprompting_local_default() { + // The output interface, including local host addresses, defines this + // exception. IPv4 loopback, IPv6 loopback, and a local host address + // must all retain the same desktop IPC behavior. + let mut ipv4 = tcp_packet(53); + ipv4[12..16].copy_from_slice(&[127, 0, 0, 1]); + ipv4[16..20].copy_from_slice(&[127, 0, 0, 53]); + let mut ipv6 = vec![0u8; 44]; + ipv6[0] = 0x60; + ipv6[6] = 6; + ipv6[23] = 1; + ipv6[39] = 1; + ipv6[40..42].copy_from_slice(&5555u16.to_be_bytes()); + ipv6[42..44].copy_from_slice(&53u16.to_be_bytes()); + for payload in [ipv4, ipv6, tcp_packet(53)] { + let mut h = LoopHarness::new(vec![], vec![], dp_deny()); + let mut msg = FakeMsg::new(1, payload); + msg.outdev = 1; + h.worker().handle_message(msg).unwrap(); + assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Accept)]); + assert!(h.prompt_rx.try_recv().is_err(), "local IPC must not prompt"); + assert_eq!(h.stats.connections_allowed(), 1); + } + } + + #[test] + fn loopback_addresses_without_the_local_output_interface_still_prompt() { + for outdev in [0, 2] { + let mut h = LoopHarness::new(vec![], vec![], dp_deny()); + let mut payload = tcp_packet(53); + payload[16..20].copy_from_slice(&[127, 0, 0, 53]); + let mut msg = FakeMsg::new(1, payload); + msg.outdev = outdev; + h.worker().handle_message(msg).unwrap(); + assert!(h.verdicts().is_empty()); + assert!(h.prompt_rx.try_recv().is_ok()); + } + } + + #[test] + fn loopback_closed_rules_are_audited_before_release_even_when_paused() { + for action in [Action::Deny, Action::Reject] { + let mut scope = RuleScope::any(); + scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); + let rule = Rule::new("local application refusal", action, scope); + let store = RuleStore::open_in_memory().unwrap(); + let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()).with_store(store); + h.stats.set_paused(true); + let mut payload = tcp_packet(53); + // Policy needs only ports. No complete TCP header keeps refusal + // injection inert while testing the real verdict and audit gate. + payload.truncate(24); + let mut msg = FakeMsg::new(1, payload); + msg.outdev = 1; + h.worker().handle_message(msg).unwrap(); + assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); + assert_eq!(h.log.lock().unwrap().audited_at_verdict, vec![1]); + assert_eq!(h.observed_rx.try_recv().unwrap().verdict.action, action); + assert!(h.prompt_rx.try_recv().is_err()); + } + } + + #[test] + fn loopback_missing_identity_cannot_override_an_application_refusal() { + let mut scope = RuleScope::any(); + scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); + let rule = Rule::new("local application refusal", Action::Deny, scope); + let store = RuleStore::open_in_memory().unwrap(); + let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()).with_store(store); + h.worker().resolver = Box::new(StubResolver { + pid: None, + process: Process::unknown(0), + socket_lookups: std::sync::atomic::AtomicUsize::new(0), + }); + h.stats.set_paused(true); + let mut msg = FakeMsg::new(1, tcp_packet(53)); + msg.outdev = 1; + h.worker().handle_message(msg).unwrap(); + assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); + assert_eq!(h.log.lock().unwrap().audited_at_verdict, vec![1]); + assert!(h.prompt_rx.try_recv().is_err()); + } + + #[test] + fn loopback_keeps_the_root_daemon_dns_exception() { + let mut h = LoopHarness::new(vec![], vec![deny_port_rule(53)], dp_deny()); + h.worker().dns = Box::new(StubDns { + self_pid: Some(4242), + ..Default::default() + }); + let mut msg = FakeMsg::new(1, tcp_packet(53)); + msg.outdev = 1; + msg.uid = Some(0); + h.worker().handle_message(msg).unwrap(); + assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Accept)]); + assert!(h.prompt_rx.try_recv().is_err()); + assert_eq!(h.stats.connections_total(), 0); + } + #[test] fn allowed_delivery_enriches_and_counts_once_without_a_refusal_audit() { struct CountingDns(Arc); diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index ff9b5a3..2f10a78 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -9,7 +9,8 @@ //! unconnected-UDP, wildcard-bind, v4-mapped-in-v6). //! 3. inode -> pid via a verified TTL cache, else a /proc/*/fd walk. //! -//! TOCTOU note: the resolved pid may have exited by the time we describe it. +//! Socket ownership retains the process generation and descriptor across +//! image reads. A changed generation or closed descriptor leaves identity unknown. //! Process identity is read on every resolve: exec preserves pid and starttime. //! The inode cache re-verifies its answer with a single readlink before //! trusting it. @@ -98,6 +99,52 @@ pub fn resolve(pid: u32) -> Process { } } +/// Descriptor ownership carried across process-image resolution. This +/// establishes a current holder, not the process that sent a queued packet. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct SocketOwner { + pid: u32, + starttime: u64, + inode: u64, + fd: i32, +} + +impl SocketOwner { + pub fn pid(&self) -> u32 { + self.pid + } + + pub fn resolve(&self) -> Process { + if read_starttime(self.pid) != Some(self.starttime) { + return Process::unknown(self.pid); + } + let process = resolve_inner( + self.pid, + Some(self.starttime), + Instant::now(), + crate::ebpf::proc_table::global(), + ) + .unwrap_or_else(|_| Process::unknown(self.pid)); + if fd_points_at_socket(self.pid, self.fd, self.inode) + && read_starttime(self.pid) == Some(self.starttime) + { + process + } else { + Process::unknown(self.pid) + } + } + + #[cfg(test)] + pub(crate) fn for_test(pid: u32) -> Self { + Self { + pid, + starttime: 0, + inode: 0, + fd: -1, + } + } +} + /// The table is a parameter rather than a reach into /// `crate::ebpf::proc_table::global()` so the tests below can drive both /// branches without mutating process-wide state that every other test in the @@ -225,7 +272,7 @@ fn resolve_inner( }) } -/// Find the pid that owns a socket matching the given 5-tuple. +/// Find a current descriptor holder for a socket matching the given 5-tuple. /// /// TCP tries sock_diag first. UDP requires a unique inode across all relevant /// tables before a diagnostic cookie or an fd walk may identify the owner. @@ -255,7 +302,7 @@ fn resolve_inner( /// flow to the process *listening* on the port is a real and separate thing, /// and it would be one netlink round trip rather than two /proc scans; it is /// not done here because it would change which rules match, not just how fast. -pub fn pid_for_socket( +pub fn socket_owner( protocol: Protocol, direction: Direction, src_ip: IpAddr, @@ -263,7 +310,7 @@ pub fn pid_for_socket( dst_ip: IpAddr, dst_port: u16, uid: Option, -) -> Option { +) -> Option { if direction == Direction::Inbound { return None; } @@ -287,24 +334,16 @@ pub fn pid_for_socket( .filter(|info| uid.is_none_or(|uid| info.uid == uid)) .filter(|info| udp_inode.is_none_or(|inode| info.inode == inode)); - // Fastest path: the kernel recorded cookie -> tgid at connect() time - // (`SOCK_PIDS`, written by cfc_connect4|6 in the connecting process's own - // context). One map lookup replaces the /proc walk below, which measures - // 37-44 ms on a loaded desktop - per NEW connection, before rule - // evaluation, on the only worker thread. This one line is the difference - // between the firewall being invisible and being felt. - if let Some(cookie) = info.as_ref().and_then(|i| i.cookie) { - if let Some(pid) = crate::ebpf::cookie_pid(cookie) { - record_resolved_pid(pid); - return Some(pid); - } - } + let cookie_pid = info + .as_ref() + .and_then(|i| i.cookie) + .and_then(crate::ebpf::cookie_pid); let inode = udp_inode .or_else(|| info.map(|i| i.inode)) .or_else(|| proc_net_inode(protocol, src_ip, src_port, dst_ip, dst_port, uid, deadline))?; - pid_owning_inode(inode, deadline) + pid_owning_inode(inode, cookie_pid, deadline) } /// Pids that recently owned a resolved socket, most recent first. @@ -332,15 +371,30 @@ fn record_resolved_pid(pid: u32) { /// /// The unit of work the probe lists reuse: one process's fd table instead of /// every process's. -fn pid_has_socket_inode(pid: u32, inode: u64) -> Option { +fn pid_has_socket_inode(pid: u32, inode: u64, deadline: Instant) -> Option { + if inode == 0 || Instant::now() >= deadline { + return None; + } let p = ProcFsProcess::new(pid as i32).ok()?; + let starttime = read_starttime(pid)?; let fds = p.fd().ok()?; for fd in fds.flatten() { + if Instant::now() >= deadline { + return None; + } if matches!(fd.target, FDTarget::Socket(i) if i == inode) { + if read_starttime(pid) != Some(starttime) { + return None; + } INODE_PID_CACHE .lock() .insert(inode, (pid, fd.fd), Instant::now()); - return Some(pid); + return Some(SocketOwner { + pid, + starttime, + inode, + fd: fd.fd, + }); } } None @@ -430,21 +484,16 @@ fn udp_inode_from_tables( struct TableEntry { local: (IpAddr, u16), remote: (IpAddr, u16), + state: u8, inode: u64, uid: u32, } /// Match a socket table (the text of /proc/net/{tcp,udp}{,6}) against a /// flow. UDP requires one unique compatible inode across all match classes. -/// TCP uses decreasing precision and stops at the first hit. -/// -/// Pass 1 - exact local+remote: connected TCP/UDP sockets. -/// Pass 2 - UDP only, exact local, zero remote: unconnected UDP sockets -/// doing plain sendto() (mDNS, NTP, syslog, QUIC stacks) list their -/// remote as 0.0.0.0:0, so an exact-remote match can never hit them. -/// Pass 3 - wildcard local addr, matching port, remote exact-or-zero: -/// sockets bound to 0.0.0.0 / :: show the wildcard, not the address -/// the flow actually uses. +/// TCP requires an exact connected tuple and never selects a listener. +/// UDP also includes zero-remote and wildcard-local sockets for sendto() +/// users (mDNS, NTP, syslog, QUIC); every compatible inode must agree. /// /// All address comparisons canonicalize v4-mapped v6 (::ffff:a.b.c.d) to /// plain v4 first, which is how dual-stack sockets appear in the v6 tables. @@ -493,19 +542,11 @@ fn scan_table_entries( return inode; } - // Pass 1: exact 4-tuple. - for e in entries.clone() { - if endpoint_eq(e.local, local) && endpoint_eq(e.remote, remote) { - return Some(e.inode); - } - } - - // Pass 3: wildcard-bound local (port must match), remote exact or zero - // (zero covers listeners and wildcard-bound unconnected UDP). for e in entries { - if e.local.1 == local.1 - && e.local.0.to_canonical().is_unspecified() - && (endpoint_eq(e.remote, remote) || endpoint_is_zero(e.remote)) + if e.state != 0x0A // TCP_LISTEN + && !endpoint_is_zero(e.remote) + && endpoint_eq(e.local, local) + && endpoint_eq(e.remote, remote) { return Some(e.inode); } @@ -527,7 +568,7 @@ fn parse_table_line(line: &str) -> Option { let _sl = cols.next()?; let local = parse_hex_addr_port(cols.next()?)?; let remote = parse_hex_addr_port(cols.next()?)?; - let _state = cols.next()?; + let state = u8::from_str_radix(cols.next()?, 16).ok()?; let _txrx = cols.next()?; let _tr = cols.next()?; let _retr = cols.next()?; @@ -537,6 +578,7 @@ fn parse_table_line(line: &str) -> Option { Some(TableEntry { local, remote, + state, inode, uid, }) @@ -600,12 +642,33 @@ fn format_addr_port(ip: IpAddr, port: u16) -> String { /// A verified cache fronts the /proc/*/fd walk: on a hit we re-readlink /// the remembered fd and only trust the pid if it still points at /// `socket:[inode]`; otherwise the entry is dropped and we re-walk. -fn pid_owning_inode(inode: u64, deadline: Instant) -> Option { +fn pid_owning_inode(inode: u64, cookie_pid: Option, deadline: Instant) -> Option { + if inode == 0 || Instant::now() >= deadline { + return None; + } + // The connect-time cookie records a numeric PID, not its lifetime or + // current descriptor ownership. It is only a hint for the verified walk. + if let Some(pid) = cookie_pid { + if let Some(found) = pid_has_socket_inode(pid, inode, deadline) { + record_resolved_pid(found.pid); + return Some(found); + } + } let now = Instant::now(); let cached = INODE_PID_CACHE.lock().get(&inode, now); if let Some((pid, fd)) = cached { - if fd_points_at_socket(pid, fd, inode) { - return Some(pid); + let starttime = read_starttime(pid); + if fd_points_at_socket(pid, fd, inode) && Instant::now() < deadline { + if let Some(starttime) = + starttime.filter(|starttime| read_starttime(pid) == Some(*starttime)) + { + return Some(SocketOwner { + pid, + starttime, + inode, + fd, + }); + } } INODE_PID_CACHE.lock().remove(&inode); } @@ -625,15 +688,15 @@ fn pid_owning_inode(inode: u64, deadline: Instant) -> Option { // (16 entries against 24), which makes the miss cheaper as well. let recently_resolved: Vec = RESOLVED_PIDS.lock().iter().copied().collect(); for pid in recently_resolved { - if let Some(found) = pid_has_socket_inode(pid, inode) { - record_resolved_pid(found); + if let Some(found) = pid_has_socket_inode(pid, inode, deadline) { + record_resolved_pid(found.pid); return Some(found); } } let recent_execs = crate::ebpf::proc_table::global().recent_pids(24, Instant::now()); for pid in recent_execs { - if let Some(found) = pid_has_socket_inode(pid, inode) { - record_resolved_pid(found); + if let Some(found) = pid_has_socket_inode(pid, inode, deadline) { + record_resolved_pid(found.pid); return Some(found); } } @@ -652,8 +715,8 @@ fn pid_owning_inode(inode: u64, deadline: Instant) -> Option { if Instant::now() > deadline { return None; } - if let Some(found) = pid_has_socket_inode(pid, inode) { - record_resolved_pid(found); + if let Some(found) = pid_has_socket_inode(pid, inode, deadline) { + record_resolved_pid(found.pid); return Some(found); } } @@ -952,6 +1015,52 @@ mod tests { // -- table scanning --------------------------------------------------- + #[test] + fn cookie_pid_hint_requires_live_socket_ownership() { + use std::os::fd::AsRawFd; + use std::os::unix::net::UnixDatagram; + + let pid = std::process::id(); + let socket = UnixDatagram::unbound().unwrap(); + let link = fs::read_link(format!("/proc/self/fd/{}", socket.as_raw_fd())).unwrap(); + let inode = link + .to_str() + .unwrap() + .strip_prefix("socket:[") + .unwrap() + .strip_suffix(']') + .unwrap() + .parse() + .unwrap(); + let budget = || Instant::now() + Duration::from_secs(2); + let owner = pid_owning_inode(inode, Some(pid), budget()).unwrap(); + assert_eq!(owner.pid(), pid); + assert_ne!(owner.resolve().exe.to_str(), Some(cfc_core::UNKNOWN_EXE)); + assert_eq!(pid_owning_inode(u64::MAX, Some(pid), budget()), None); + // A dead hint must not mask the descriptor's current holder. + assert_eq!( + pid_owning_inode(inode, Some(u32::MAX), budget()).map(|o| o.pid()), + Some(pid) + ); + assert_eq!(pid_owning_inode(0, Some(pid), budget()), None); + assert_eq!( + pid_owning_inode(inode, Some(pid), Instant::now() - Duration::from_secs(1)), + None + ); + // Model a different process generation at the validation-to-use edge. + let changed_generation = SocketOwner { + starttime: owner.starttime + 1, + ..owner + }; + assert_eq!( + changed_generation.resolve().exe.to_str(), + Some(cfc_core::UNKNOWN_EXE) + ); + // A closed descriptor cannot authorize a subsequently read image. + drop(socket); + assert_eq!(owner.resolve().exe.to_str(), Some(cfc_core::UNKNOWN_EXE)); + } + const HEADER: &str = " sl local_address rem_address st tx_queue rx_queue tr tm->when retrnsmt uid timeout inode\n"; @@ -994,9 +1103,7 @@ mod tests { #[test] fn unconnected_fallback_is_udp_only() { - // The zero-remote pass must not apply to TCP: a TCP row with a - // zero remote is a listener, matched (if at all) by the wildcard - // pass, not by pretending it is connected to our destination. + // The zero-remote UDP match must not attribute TCP to a listener. let local = v4(10, 0, 2, 15, 5353); let table = format!("{HEADER}{}", line(local, v4(0, 0, 0, 0, 0), "0A", 4242)); assert_eq!( @@ -1026,8 +1133,8 @@ mod tests { } #[test] - fn wildcard_v6_matches_v4_flow() { - // Dual-stack socket bound to [::]:8080 must attribute v4 traffic. + fn outbound_tcp_cannot_borrow_a_dual_stack_listener() { + // An outbound flow must not inherit a listening application's policy. let local = (IpAddr::V6(Ipv6Addr::UNSPECIFIED), 8080); let remote = (IpAddr::V6(Ipv6Addr::UNSPECIFIED), 0); let table = format!("{HEADER}{}", line(local, remote, "0A", 909)); @@ -1039,7 +1146,28 @@ mod tests { v4(1, 2, 3, 4, 55000), None ), - Some(909) + None + ); + } + + #[test] + fn outbound_tcp_requires_a_connected_exact_tuple() { + let local = v4(10, 0, 0, 7, 8080); + let remote = v4(1, 2, 3, 4, 55000); + for listener_local in [local, v4(0, 0, 0, 0, 8080)] { + let table = format!( + "{HEADER}{}", + line(listener_local, v4(0, 0, 0, 0, 0), "0A", 909) + ); + assert_eq!( + scan_table_content(&table, Protocol::Tcp, local, remote, Some(1000)), + None + ); + } + let table = format!("{HEADER}{}", line(local, remote, "0A", 909)); + assert_eq!( + scan_table_content(&table, Protocol::Tcp, local, remote, Some(1000)), + None ); } diff --git a/crates/cfc-daemon/src/sock_diag.rs b/crates/cfc-daemon/src/sock_diag.rs index 7e54a36..4032c55 100644 --- a/crates/cfc-daemon/src/sock_diag.rs +++ b/crates/cfc-daemon/src/sock_diag.rs @@ -210,13 +210,13 @@ fn reply_seq(buf: &[u8]) -> Option { } /// answer with a single SOCK_DIAG_BY_FAMILY message or an NLMSG_ERROR. -fn parse_response(buf: &[u8]) -> Option { +fn parse_response(buf: &[u8], protocol: u8) -> Option { if buf.len() < NLMSG_HDR_LEN { return None; } let msg_len = u32::from_ne_bytes(buf[0..4].try_into().ok()?) as usize; let msg_type = u16::from_ne_bytes(buf[4..6].try_into().ok()?); - if msg_type != SOCK_DIAG_BY_FAMILY || msg_len > buf.len() { + if msg_type != SOCK_DIAG_BY_FAMILY || msg_len < NLMSG_HDR_LEN || msg_len > buf.len() { // NLMSG_ERROR (no such socket, EPERM, ...) or truncated reply. return None; } @@ -224,6 +224,11 @@ fn parse_response(buf: &[u8]) -> Option { if payload.len() < INET_DIAG_MSG_LEN { return None; } + // A listener is not the connected socket that emitted an outbound flow. + // UDP's unconnected state remains valid and is checked by the caller. + if protocol == libc::IPPROTO_TCP as u8 && payload[1] == 0x0A { + return None; + } // struct inet_diag_msg: id.idiag_cookie sits at payload offset 44 // (family/state/timer/retrans = 4, sport+dport = 4, src = 16, dst = 16, // if = 4), then expires, rqueue, wqueue, uid at 64, inode at 68. The @@ -323,7 +328,7 @@ impl DiagSocket { trace!("sock_diag answered a different request; discarding the socket"); return Reply::Desync; } - match parse_response(buf) { + match parse_response(buf, req[17]) { Some(info) => Reply::Found(info), None => Reply::NotFound, } @@ -442,7 +447,7 @@ mod tests { buf[NLMSG_HDR_LEN + 64..NLMSG_HDR_LEN + 68].copy_from_slice(&1000u32.to_ne_bytes()); buf[NLMSG_HDR_LEN + 68..NLMSG_HDR_LEN + 72].copy_from_slice(&31337u32.to_ne_bytes()); assert_eq!( - parse_response(&buf), + parse_response(&buf, libc::IPPROTO_TCP as u8), Some(SockInfo { inode: 31337, cookie: None, @@ -451,6 +456,27 @@ mod tests { ); } + #[test] + fn outbound_tcp_diag_rejects_listeners_without_rejecting_udp() { + let mut buf = vec![0u8; NLMSG_HDR_LEN + INET_DIAG_MSG_LEN]; + let len = buf.len() as u32; + buf[0..4].copy_from_slice(&len.to_ne_bytes()); + buf[4..6].copy_from_slice(&SOCK_DIAG_BY_FAMILY.to_ne_bytes()); + buf[NLMSG_HDR_LEN + 68..NLMSG_HDR_LEN + 72].copy_from_slice(&31337u32.to_ne_bytes()); + for state in [0x01, 0x02, 0x0A] { + buf[NLMSG_HDR_LEN + 1] = state; + assert_eq!( + parse_response(&buf, libc::IPPROTO_TCP as u8).map(|i| i.inode), + (state != 0x0A).then_some(31337) + ); + } + buf[NLMSG_HDR_LEN + 1] = 0x07; + assert_eq!( + parse_response(&buf, libc::IPPROTO_UDP as u8).map(|i| i.inode), + Some(31337) + ); + } + #[test] fn error_reply_is_none() { // NLMSG_ERROR (type 2) reply, as the kernel sends for a miss. @@ -459,17 +485,19 @@ mod tests { buf[0..4].copy_from_slice(&len.to_ne_bytes()); buf[4..6].copy_from_slice(&2u16.to_ne_bytes()); buf[NLMSG_HDR_LEN..].copy_from_slice(&(-2i32).to_ne_bytes()); // -ENOENT - assert_eq!(parse_response(&buf), None); + assert_eq!(parse_response(&buf, libc::IPPROTO_TCP as u8), None); } #[test] fn truncated_reply_is_none() { - assert_eq!(parse_response(&[0u8; 8]), None); + assert_eq!(parse_response(&[0u8; 8], libc::IPPROTO_TCP as u8), None); let mut buf = vec![0u8; NLMSG_HDR_LEN + 8]; let len = buf.len() as u32; buf[0..4].copy_from_slice(&len.to_ne_bytes()); buf[4..6].copy_from_slice(&SOCK_DIAG_BY_FAMILY.to_ne_bytes()); - assert_eq!(parse_response(&buf), None); + assert_eq!(parse_response(&buf, libc::IPPROTO_TCP as u8), None); + buf[0..4].copy_from_slice(&8u32.to_ne_bytes()); + assert_eq!(parse_response(&buf, libc::IPPROTO_TCP as u8), None); } #[test] diff --git a/docs/HARDENING.md b/docs/HARDENING.md index ed53008..0d804f4 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -11,7 +11,7 @@ desktop, not what's theoretically pure. 2. Click through prompts for a week. Save persistent rules as you go. 3. Run `cfc rules bootstrap-defaults` to install common system rules. 4. Once the prompt rate drops to maybe 1-2 a day, switch to - `profile = "strict"` for fail-closed behavior. + `profile = "strict"` if you prefer a shorter prompt timeout. 5. Audit `cfc rules list` monthly. Remove rules for apps you no longer use, and check `cfc log --since 30d` for destinations you did not expect. @@ -23,14 +23,14 @@ running" throughout - it subscribes the same way the GUI does. | Profile | No UI | Timeout | Window | Use when | |----------|----------|----------|--------|------------------------------------------------| -| relaxed | Allow | Deny | 60s | Headless servers / can't always be at the UI | -| balanced | Allow | Deny | 30s | Daily-driver workstations (default) | -| strict | Deny | Deny | 15s | Lockdown posture, UI always present | +| relaxed | Deny | Deny | 60s | Longer time to answer prompts | +| balanced | Deny | Deny | 30s | Daily-driver workstations (default) | +| strict | Deny | Deny | 15s | Shorter time to answer prompts | -**No profile ever permits a connection by itself.** Not on timeout, not +**No profile ever permits a remote connection by itself.** Not on timeout, not when nothing is subscribed. The presets differ only in how long a prompt -waits for an answer. Only a stored rule, or a person answering, allows -traffic. +waits for an answer. Under these presets, a stored rule or a prompt answer +permits remote traffic. Unmatched local IPC is allowed without prompting. A timeout means the question *was* put to you and went unanswered; if that granted access, the cheapest attack would be to connect while @@ -58,8 +58,8 @@ before the daemon and `network-pre.target`. Enabled enforcement is required by NetworkManager and systemd-networkd, so an nft load failure blocks their startup. Initial daemon failure leaves the table loaded and drops new flows. This does not cover initramfs networking, already configured interfaces, or -other network managers. Once loaded, strict -filtering denies unmatched flows, so DHCP, DNS and NTP need standing rules or +other network managers. Once loaded, strict filtering denies unmatched remote +flows, so DHCP, DNS and NTP need standing rules or the machine cannot even get a lease. Network managers retrying DNS will look like total network failure. **Only flip to strict after you have rules for every always-on system service**. @@ -174,20 +174,32 @@ real path under `/usr/lib/...` or pin by SHA-256 (`scope.exe_sha256`). ## What this firewall does *not* protect against +Normal mode follows the desktop application firewall model of OpenSnitch and +Windows Firewall Control. It filters new tracked IP flows using socket +attribution. [Explicit application confinement](../README.md#explicit-application-confinement) +is a separate launch mode. + - **Anything from root**: `/usr/bin/colony-firewalld` itself is trusted, and so is any other root process. Use this firewall alongside, not instead of, traditional access controls. - **eBPF / unprivileged user namespaces**: a sufficiently privileged user can bypass NFQUEUE entirely with `unshare -rn` and a custom net namespace. -- **Local relays and DNS**: loopback is exempt. A denied application can use - an allowed local resolver or proxy; outbound traffic is attributed to that - service. Hostname rules and observed answers do not isolate DNS queries. +- **Local relays and DNS**: explicit rules apply to new direct loopback flows. + Unmatched local IPC is allowed without prompting. An authorized local + resolver or proxy can relay remote traffic, which is attributed to that + service. CFC cannot establish the originating application's identity from + remote flows delegated through AF_UNIX or D-Bus brokers. Existing local + connections retain their authorization. Hostname rules and observed answers + do not isolate DNS queries. - **Inherited or passed sockets**: established/related traffic keeps its connection-wide authorization. An inherited or passed descriptor is not - reauthorized for each sending executable. -- **Packet-layer privileges**: applications with `CAP_NET_RAW` can use packet - sockets outside the shipped IP OUTPUT hooks. These rules do not provide - layer-2 containment. + reauthorized for each sending executable. Current descriptor ownership + and validated eBPF hints reduce false attribution; neither proves which + process sent a packet. +- **Raw and packet sockets**: applications with `CAP_NET_RAW` can use AF_PACKET + outside the shipped `inet OUTPUT` hook. Raw IP packets can coincide with + another socket's tuple even when TCP matching is strict. Tuple and inode + checks do not prove raw packet provenance or provide layer-2 containment. - **DNS-over-HTTPS embedded in browsers**: the firewall sees the outer HTTPS flow. Domain isolation requires an application-aware proxy or separate containment. - **Container traffic**: Docker / Podman / LXC route through their own @@ -361,8 +373,8 @@ to shrink what a code-execution bug could reach: **`ProtectProc=invisible` is deliberately absent.** It would hide other processes' `/proc` entries from the daemon, and that is precisely how process attribution works: `/proc/net/{tcp,udp}` gives a socket inode, -and the owning pid is found by walking `/proc/*/fd` for a matching -`socket:[inode]` link. Turning it on makes every connection resolve to an +and a current descriptor holder is found by walking `/proc/*/fd` for a +matching `socket:[inode]` link. Turning it on makes every connection resolve to an unknown process, which defeats the entire tool. Same reason `CAP_SYS_PTRACE` is in the bounding set. If you are hand-editing the unit, do not "harden" either of these. diff --git a/systemd/nftables-snippet.conf b/systemd/nftables-snippet.conf index 48a0579..2eb8d1e 100644 --- a/systemd/nftables-snippet.conf +++ b/systemd/nftables-snippet.conf @@ -1,6 +1,7 @@ # New outbound flows require a verdict from NFQUEUE 0, without bypass. # Established flows retain their connection-wide authorization, including SSH -# replies. Loopback is outside application policy; see docs/HARDENING.md. +# replies. New loopback flows follow explicit application policy; unmatched +# local IPC is allowed without prompting. See docs/HARDENING.md. # Enable colony-firewall-nft.service for persistence. Its rules remain loaded # across daemon restarts and stops. Stop that unit explicitly to lift filtering. @@ -18,7 +19,6 @@ table inet colony_firewall { chain output { type filter hook output priority 0; policy drop; - oifname "lo" accept ct state established,related accept # The daemon's dedicated Reject sockets stamp this value. The root UID From 69bda5bd45502bf351be99657c669587223e41c1 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:07:06 +0200 Subject: [PATCH 002/125] refactor(daemon)!: remove the Fast Allow userspace path Fast Allow was disabled during the security remediation because a socket mark cannot prove which process sends a packet. Remove what stayed compiled in: the grant writers, the eligibility ladder, mark drawing, the heartbeat, the ALLOW_EVENTS consumer, the sendmsg attach and the status reporting. What remains is legacy cleanup for hosts upgrading from 0.4-0.6: one nft flush of the fast_allow set at start (outside --dry-run), a disarm of the pinned kernel maps on every load, and removal of the old sendmsg pins and cookie-variants marker. The [ebpf] fast_allow keys still parse and only log a warning. StatusResponse field 16 is reserved, and cfc --json status no longer has a fast_allow key. --- crates/cfc-cli/src/main.rs | 49 +- crates/cfc-cli/tests/cli_e2e.rs | 6 +- crates/cfc-daemon/src/config.rs | 68 +- crates/cfc-daemon/src/decision.rs | 405 +-------- crates/cfc-daemon/src/ebpf.rs | 285 +------ crates/cfc-daemon/src/ebpf/enforce.rs | 986 +++------------------- crates/cfc-daemon/src/ebpf/loader.rs | 1092 +++---------------------- crates/cfc-daemon/src/ebpf/nft_set.rs | 505 +----------- crates/cfc-daemon/src/ipc.rs | 3 - crates/cfc-daemon/src/main.rs | 24 +- crates/cfc-proto/proto/cfc.proto | 15 +- 11 files changed, 355 insertions(+), 3083 deletions(-) diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index d29a9d9..7eaf85d 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -397,7 +397,6 @@ struct StatusJson { uptime_seconds: u64, enforcing: bool, enforcement: String, - fast_allow: String, paused: bool, resume_at_unix_ms: i64, resume_at: Option, @@ -420,7 +419,6 @@ fn status_json(s: &proto::StatusResponse, now_unix_ms: i64) -> StatusJson { uptime_seconds: s.uptime_seconds, enforcing: s.enforcing, enforcement: s.enforcement.clone(), - fast_allow: s.fast_allow.clone(), paused: s.paused, resume_at_unix_ms: s.resume_at_unix_ms, resume_at: output::rfc3339(s.resume_at_unix_ms), @@ -481,19 +479,6 @@ fn enforcement_cell(level: &str) -> String { } } -/// The fast-allow cell: the daemon's own sentence, or why there is none. -/// -/// The daemon already spells this one out (`live`, or `off: ` and the reason -/// the path is inert), so the CLI only has to name the value the daemon cannot -/// send: the proto3 default, which is what a daemon without the field answers -/// and also what a daemon whose startup has not decided yet answers. -fn fast_allow_cell(level: &str) -> String { - match level { - "" => "unknown (still starting, or this daemon is too old to say)".to_string(), - other => other.to_string(), - } -} - fn paused_cell(s: &proto::StatusResponse, now_unix_ms: i64) -> String { if !s.paused { return "no".to_string(); @@ -527,7 +512,6 @@ async fn cmd_status(client: &mut Client, format: OutputFormat) -> CliResult { if s.enforcing { "yes" } else { "no" } ); println!(" in-kernel {}", enforcement_cell(&s.enforcement)); - println!(" fast-allow {}", fast_allow_cell(&s.fast_allow)); println!("paused {}", paused_cell(&s, now)); println!("rules {}", s.rules_count); println!("prompts pending {}", s.prompts_pending); @@ -614,7 +598,6 @@ mod tests { skipped_rules: 0, enforcing: true, enforcement: "pinned".to_string(), - fast_allow: "live".to_string(), } } @@ -772,34 +755,12 @@ mod tests { assert_eq!(v["warnings"].as_array().unwrap().len(), 2); } - /// The daemon's sentence passes through untouched in both output modes; - /// only its absence gets words, and those must not read as an answer. + /// The proto3 default is what an older daemon sends, and what a new one + /// sends before startup has answered; the text mode must say what the + /// blank means rather than print nothing. #[test] - fn fast_allow_passes_through_and_names_its_own_absence() { - let now = 1_700_000_000_000; - let mut s = status(false, 0); - let v = serde_json::to_value(status_json(&s, now)).unwrap(); - assert_eq!(v["fast_allow"], "live"); - - s.fast_allow = "off: [ebpf] fast_allow is not set".to_string(); - let v = serde_json::to_value(status_json(&s, now)).unwrap(); - assert_eq!(v["fast_allow"], "off: [ebpf] fast_allow is not set"); - assert_eq!(fast_allow_cell(&s.fast_allow), s.fast_allow); - assert_eq!(fast_allow_cell("live"), "live"); - - // The proto3 default is what an older daemon sends, and what a new - // one sends before startup has answered. JSON keeps it verbatim so a - // script can tell "" from a real value; the text mode says what the - // blank means, as the enforcement cell does for its own. - s.fast_allow = String::new(); - let v = serde_json::to_value(status_json(&s, now)).unwrap(); - assert_eq!(v["fast_allow"], ""); - let cell = fast_allow_cell(""); - assert!(cell.starts_with("unknown ("), "{cell}"); - assert!( - enforcement_cell("").starts_with("unknown ("), - "the two cells must agree on how an absent answer reads" - ); + fn an_absent_enforcement_level_reads_as_unknown() { + assert!(enforcement_cell("").starts_with("unknown (")); } #[test] diff --git a/crates/cfc-cli/tests/cli_e2e.rs b/crates/cfc-cli/tests/cli_e2e.rs index 16cc220..60b027f 100644 --- a/crates/cfc-cli/tests/cli_e2e.rs +++ b/crates/cfc-cli/tests/cli_e2e.rs @@ -146,7 +146,6 @@ impl Firewall for FakeDaemon { skipped_rules: 2, enforcing: false, enforcement: "pinned".to_string(), - fast_allow: "off: [ebpf] fast_allow is not set".to_string(), })) } @@ -570,9 +569,8 @@ async fn status_json_round_trips_over_a_real_socket() { assert_eq!(v["enforcing"], false); assert_eq!(v["skipped_rules"], 2); assert_eq!(v["timeout_action"], "deny"); - // The daemon's own sentence, verbatim: a script must be able to read - // the reason, not only that there is one. - assert_eq!(v["fast_allow"], "off: [ebpf] fast_allow is not set"); + // The removed Fast Allow field must not come back under its old name. + assert!(v.get("fast_allow").is_none(), "{v}"); // Both warnings must be machine-readable too, not just printed. let warnings = v["warnings"].as_array().expect("warnings array"); assert_eq!(warnings.len(), 2, "{warnings:?}"); diff --git a/crates/cfc-daemon/src/config.rs b/crates/cfc-daemon/src/config.rs index 06a019d..cfb0afd 100644 --- a/crates/cfc-daemon/src/config.rs +++ b/crates/cfc-daemon/src/config.rs @@ -318,22 +318,13 @@ pub struct EbpfConfig { /// Where the BPF object built by `cargo xtask build-ebpf` was installed. /// `None` means `crate::ebpf::DEFAULT_OBJECT_PATH`. pub object_path: Option, - /// Compatibility setting, currently ignored: Fast Allow is disabled for - /// every configuration because socket marks cannot attest the sender. - /// Ordinary traffic uses NFQUEUE; the startup report explains the refusal. + /// Legacy key from the removed Fast Allow path. Ignored; a warning is + /// logged. Still parsed rather than rejected: a parse error stops the + /// daemon while the fail-closed nftables table stays loaded, so an + /// upgrade would take the host offline. pub fast_allow: bool, - /// The `SO_MARK` value the fast path uses, when the machine needs a - /// specific one. - /// - /// `None` - the default - draws one at random at each start, which is what - /// keeps it from being a forgeable token. Set it only to resolve a - /// collision: the mark space is shared with the whole machine, and a - /// consumer that selects on a *mask* will match a random value with a - /// probability its mask decides. See `ebpf::loader::pick_mark` for the - /// selectors CFC already avoids, and `docs/TROUBLESHOOTING.md` for how to - /// find the one it does not know about. - /// - /// Zero is refused: it is the mark of every socket nothing has marked. + /// Legacy key from the removed Fast Allow path. Ignored; a warning is + /// logged. pub fast_allow_mark: Option, } @@ -671,42 +662,15 @@ enabled = " Auto ""# assert_eq!(cfg.ebpf.enabled, EbpfMode::Auto); } - /// `daemon.toml.sample` documents the mark in hex, so hex has to parse. - /// A sample that shows a spelling the parser rejects is worse than no - /// sample: the operator only finds out when the daemon refuses to start. + /// The removed Fast Allow keys must keep parsing: rejecting them would + /// stop the daemon on upgrade with the fail-closed table still loaded. #[test] - fn the_fast_allow_mark_parses_in_the_spelling_the_sample_documents() { - let mark = |toml: &str| Config::from_toml_str(toml).unwrap().ebpf.fast_allow_mark; - assert_eq!(mark(""), None, "absent means draw one"); - assert_eq!( - mark("[ebpf]\nfast_allow_mark = 0x00033331\n"), - Some(0x0003_3331) - ); - assert_eq!(mark("[ebpf]\nfast_allow_mark = 209713\n"), Some(209_713)); - // The whole word must fit: the mark is a u32, and the top bit is as - // legitimate a mark bit as any other. - assert_eq!( - mark("[ebpf]\nfast_allow_mark = 0xffffffff\n"), - Some(u32::MAX) - ); - } - - /// The fast path is opt-in: nothing short of `fast_allow = true` turns it - /// on, and the layer's own eligibility checks still get the last word. - #[test] - fn ebpf_fast_allow_is_off_unless_asked_for() { - let fast_allow = |toml: &str| Config::from_toml_str(toml).unwrap().ebpf.fast_allow; - - assert!(!fast_allow(""), "absent means off"); - assert!( - !fast_allow("[ebpf]\n"), - "an empty section keeps the default" - ); - assert!(!fast_allow("[ebpf]\nfast_allow = false\n")); - assert!(fast_allow("[ebpf]\nfast_allow = true\n")); - // Parsed independently of `enabled`: the switch says what was asked - // for, and the layer decides whether it can honour it. - assert!(fast_allow("[ebpf]\nenabled = false\nfast_allow = true\n")); + fn the_legacy_fast_allow_keys_still_parse() { + let cfg = + Config::from_toml_str("[ebpf]\nfast_allow = true\nfast_allow_mark = 0x00033331\n") + .expect("legacy keys must not abort startup"); + assert!(cfg.ebpf.fast_allow); + assert_eq!(cfg.ebpf.fast_allow_mark, Some(0x0003_3331)); } #[test] @@ -903,8 +867,8 @@ enabled = " Auto ""# config resolves to the automatic default" ); assert!( - !cfg.ebpf.fast_allow, - "a shipped config must not take allows off the packet path" + !cfg.ebpf.fast_allow && cfg.ebpf.fast_allow_mark.is_none(), + "a shipped config must not set the legacy Fast Allow keys" ); assert_eq!( cfg.storage.path, diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index 5f0cae6..d6670f7 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -44,33 +44,6 @@ struct EngineInner { on_change: RwLock>>, } -/// What [`Engine::process_wide_verdict`] found: the action that holds for a -/// process wherever it connects, and the rule that says so. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub struct ProcessWideVerdict { - pub action: cfc_core::Action, - pub rule_id: uuid::Uuid, - pub duration: cfc_core::Duration, -} - -impl ProcessWideVerdict { - /// Whether this verdict may be handed to the in-kernel fast path. - /// - /// An allow, from a rule that lasts. `Once` never reaches here (it is - /// not stored), and a timed rule is excluded on purpose: the fast path - /// re-checks grants only on flow starts and rule changes, so a timed - /// allow would keep marking sockets until the next flush noticed its - /// deadline - up to thirty seconds past the moment the user chose. - /// Denies never qualify either way; they have their own map. - pub fn fast_allow_eligible(&self) -> bool { - self.action == cfc_core::Action::Allow - && matches!( - self.duration, - cfc_core::Duration::Always | cfc_core::Duration::UntilRestart - ) - } -} - pub enum Decision { /// A persistent rule matched. Return the verdict immediately. Resolved(Verdict), @@ -137,14 +110,6 @@ impl Engine { self.notify_changed(); } - /// Credits a hit to `rule_id` for a flow the packet path never saw - a - /// fast-allowed connection reported by the kernel. The same counter - /// `evaluate` bumps, so the busiest allow rule stops reading as dead the - /// day its traffic skips the queue. - pub fn record_hit(&self, rule_id: uuid::Uuid) { - *self.inner.hits.lock().entry(rule_id).or_insert(0) += 1; - } - fn notify_changed(&self) { if let Some(f) = self.inner.on_change.read().as_ref() { f(); @@ -209,15 +174,6 @@ impl Engine { /// `None` means "ask the packet path", which is always a safe answer: it /// is what happened before this existed. pub fn process_wide_action(&self, proc: &Process) -> Option { - self.process_wide_verdict(proc).map(|v| v.action) - } - - /// [`process_wide_action`](Self::process_wide_action) with the rule that - /// answered: its id, for crediting a hit the packet path will never see, - /// and its duration, because the fast path is offered only to rules that - /// last. A timed allow (`--for 1h`) keeps the packet path, so its expiry - /// is exact rather than "within the next flush tick". - pub fn process_wide_verdict(&self, proc: &Process) -> Option { let now_unix_ms = chrono::Utc::now().timestamp_millis(); let rules = self.inner.rules.read(); for rule in rules @@ -247,43 +203,11 @@ impl Engine { if !rule.scope.matches_process(proc) { continue; } - return (!rule.scope.constrains_destination()).then_some(ProcessWideVerdict { - action: rule.action, - rule_id: rule.id, - duration: rule.duration, - }); + return (!rule.scope.constrains_destination()).then_some(rule.action); } None } - /// Whether any enabled rule could grant the fast path to *some* process. - /// - /// The same predicate `process_wide_verdict` applies per process, asked of - /// the rule set as a whole: outbound, not flow-scoped, and eligible - an - /// `Allow` that lasts. Reuses [`ProcessWideVerdict::fast_allow_eligible`] - /// rather than restating it, so there is one definition of "could grant". - /// - /// For `sweep_fast_allow`, which otherwise walks /proc on every rule - /// change to reach a conclusion this answers in a few comparisons. - pub fn any_fast_allow_rule(&self) -> bool { - let now_unix_ms = chrono::Utc::now().timestamp_millis(); - let rules = self.inner.rules.read(); - rules - .rules - .iter() - .filter(|r| r.enabled && !r.is_expired(now_unix_ms)) - .filter(|r| r.scope.direction != Some(cfc_core::Direction::Inbound)) - .filter(|r| !r.scope.constrains_destination()) - .any(|r| { - ProcessWideVerdict { - action: r.action, - rule_id: r.id, - duration: r.duration, - } - .fast_allow_eligible() - }) - } - /// Whether some resolution of the rules this caller cannot decide would /// still deny this process outright - the question that separates the two /// meanings of `process_wide_action`'s `None`. @@ -414,42 +338,6 @@ impl Engine { ) } - /// Whether any live rule scoped to a uid could apply to `exe`. - /// - /// The mirror of [`Self::compilable_exe_paths`], for the grant map rather - /// than the deny map, and it exists for the same reason turned around. - /// - /// The two deciders do not read the same uid. The packet path takes the - /// uid from the kernel's exec record when it has one - the uid at - /// `execve` - and does not read `/proc//status` at all. The grant - /// path resolves through `/proc` and gets the uid the process holds - /// *now*. For a program that drops privileges after exec - `named`, - /// `postfix`, a browser entering its sandbox - those are different, so - /// `deny --exe X --uid 0` above `allow --exe X` can be answered "deny" by - /// the packet path and "allow" by the grant path. A grant is process-wide - /// and destination-blind, so that disagreement is not a slower answer, it - /// is the deny never being applied at all. - /// - /// Rather than decide which uid is the right one - a semantic change to - /// what every existing uid-scoped rule means - the fast path simply - /// abstains wherever a uid could matter. Those flows take the queue, - /// where the uid question has one answer and it is the packet path's. - /// Hosts with no uid-scoped rule, which is nearly all of them, pay - /// nothing: the walk stops at the first predicate. - pub fn uid_scoped_may_apply(&self, exe: &std::path::Path) -> bool { - let now_unix_ms = chrono::Utc::now().timestamp_millis(); - self.inner - .rules - .read() - .rules - .iter() - .filter(|r| r.enabled && !r.is_expired(now_unix_ms)) - .filter(|r| r.scope.uid.is_some()) - // A uid-scoped rule that names no executable can apply to any - // program, exactly as in `compilable_exe_paths`. - .any(|r| r.scope.exe_path.as_deref().is_none_or(|p| p == exe)) - } - /// What an inbound flow gets when no rule matches. /// /// Separate from `no_ui_action` because it answers a different question. @@ -665,83 +553,6 @@ mod tests { Engine::new(RuleSet { rules }, shared(dp_deny())) } - // --- the uid guard on the grant path -------------------------------- - - #[test] - fn a_uid_scoped_rule_takes_its_program_off_the_fast_path() { - let exe = std::path::Path::new("/usr/sbin/named"); - let other = std::path::Path::new("/usr/bin/curl"); - - // The shape that made the fast path grant what the packet path denies: - // the program execs as root and drops to its own uid, so the grant - // path reads 53 and the packet path reads 0, and only one of them - // sees the deny. - let mut denied_as_root = RuleScope::any(); - denied_as_root.exe_path = Some(PathBuf::from(exe)); - denied_as_root.uid = Some(0); - let mut allowed = RuleScope::any(); - allowed.exe_path = Some(PathBuf::from(exe)); - let engine = engine_with(vec![ - Rule::new("deny-as-root".to_string(), Action::Deny, denied_as_root), - Rule::new("allow".to_string(), Action::Allow, allowed), - ]); - assert!( - engine.uid_scoped_may_apply(exe), - "a uid-scoped rule names this program, so the fast path must stand aside" - ); - assert!( - !engine.uid_scoped_may_apply(other), - "a program no uid-scoped rule names is unaffected" - ); - - // A rule set with no uid predicate at all costs nothing: this is the - // common case and it must not be taken off the fast path. - let mut plain = RuleScope::any(); - plain.exe_path = Some(PathBuf::from(exe)); - let engine = engine_with(vec![Rule::new("a".to_string(), Action::Allow, plain)]); - assert!(!engine.uid_scoped_may_apply(exe)); - } - - #[test] - fn a_uid_rule_naming_no_program_takes_everything_off_the_fast_path() { - // It could apply to anything, and nothing here can tell whether it - // would - the same reasoning `compilable_exe_paths` uses to return - // `None` rather than a list. - let mut any_program = RuleScope::any(); - any_program.uid = Some(1000); - let engine = engine_with(vec![Rule::new( - "per-user".to_string(), - Action::Deny, - any_program, - )]); - for exe in ["/usr/bin/curl", "/usr/sbin/named", "/opt/whatever"] { - assert!( - engine.uid_scoped_may_apply(std::path::Path::new(exe)), - "{exe} must not be granted while a uid rule can reach any program" - ); - } - } - - #[test] - fn a_disabled_or_expired_uid_rule_does_not_hold_the_fast_path_back() { - let exe = std::path::Path::new("/usr/sbin/named"); - let mut scope = RuleScope::any(); - scope.exe_path = Some(PathBuf::from(exe)); - scope.uid = Some(0); - - let mut disabled = Rule::new("off".to_string(), Action::Deny, scope.clone()); - disabled.enabled = false; - assert!(!engine_with(vec![disabled]).uid_scoped_may_apply(exe)); - - let mut expired = Rule::new("gone".to_string(), Action::Deny, scope); - expired.duration = cfc_core::Duration::Seconds(1); - expired.created_at = chrono::Utc::now() - chrono::Duration::hours(1); - assert!( - !engine_with(vec![expired]).uid_scoped_may_apply(exe), - "an expired rule constrains nothing" - ); - } - // --- process_wide_action ------------------------------------------- // // This is what the `cgroup/connect4|6` programs are steered by, and it @@ -869,186 +680,29 @@ mod tests { assert_eq!(engine.process_wide_action(&without), None); } - /// The eligibility predicate, reached the way the daemon reaches it: through - /// `process_wide_verdict` on a real engine with a real rule, one case per - /// shape of answer. + /// A rule that says anything about the flow must never become a + /// process-wide answer. /// - /// The exhaustive version further down tests the predicate on its own over - /// the whole Action x Duration space; this one is the through-the-engine - /// complement, and the pair is deliberate. (The doc comment that used to - /// sit here described a rule-expiry test that lives elsewhere - a leftover - /// from a move, and a reader looking for the expiry test would have been - /// sent to the wrong function.) - #[test] - fn only_lasting_allows_are_fast_allow_eligible() { - // The fast path re-checks grants on flow starts and rule changes, not - // on a clock, so a timed allow would outlive its deadline by up to a - // flush tick. Denies have their own map and never qualify. - let mut scope = RuleScope::any(); - scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); - let eligible = |action: Action, duration: cfc_core::Duration| { - let mut rule = Rule::new("r".to_string(), action, scope.clone()); - rule.duration = duration; - let engine = engine_with(vec![rule]); - let proc = Process { - exe: PathBuf::from("/usr/bin/curl"), - ..Process::unknown(1) - }; - engine - .process_wide_verdict(&proc) - .map(|v| v.fast_allow_eligible()) - }; - assert_eq!( - eligible(Action::Allow, cfc_core::Duration::Always), - Some(true) - ); - assert_eq!( - eligible(Action::Allow, cfc_core::Duration::UntilRestart), - Some(true) - ); - assert_eq!( - eligible(Action::Allow, cfc_core::Duration::Seconds(3600)), - Some(false), - "a timed allow keeps the packet path so its expiry is exact" - ); - assert_eq!( - eligible(Action::Deny, cfc_core::Duration::Always), - Some(false) - ); - assert_eq!( - eligible(Action::Reject, cfc_core::Duration::Always), - Some(false) - ); - } - - /// The rule-set-wide question `sweep_fast_allow` asks before walking /proc: - /// one case per way a rule set can fail to grant anyone, and one that can. + /// `deny --dst-port 443` means "this program may not reach 443", and an + /// in-kernel verdict means "every `connect()` this program makes is + /// refused". Conflating them would refuse everything the rule never + /// named, so each predicate that makes a rule flow-scoped is checked on + /// its own - a missing one would only show up as a rule type that quietly + /// denies everything. #[test] - fn a_rule_set_that_cannot_grant_anyone_says_so() { - use cfc_core::Duration; - let exe = || Some(PathBuf::from("/usr/bin/curl")); - let rule = |action: Action, duration: Duration, f: fn(&mut RuleScope)| { - let mut scope = RuleScope::any(); - scope.exe_path = exe(); - f(&mut scope); - let mut r = Rule::new("r".to_string(), action, scope); - r.duration = duration; - r - }; - let none = |_: &mut RuleScope| {}; - - assert!(!engine_with(vec![]).any_fast_allow_rule(), "no rules"); - assert!( - !engine_with(vec![rule(Action::Deny, Duration::Always, none)]).any_fast_allow_rule(), - "only denies" - ); - assert!( - !engine_with(vec![rule(Action::Allow, Duration::Seconds(3600), none)]) - .any_fast_allow_rule(), - "only a timed allow" - ); - assert!( - !engine_with(vec![rule(Action::Allow, Duration::Always, |s| s - .dst_port = - Some(443))]) - .any_fast_allow_rule(), - "only a flow-scoped allow" - ); - let mut disabled = rule(Action::Allow, Duration::Always, none); - disabled.enabled = false; - assert!( - !engine_with(vec![disabled]).any_fast_allow_rule(), - "only a disabled allow" - ); - - assert!( - engine_with(vec![ - rule(Action::Deny, Duration::Always, none), - rule(Action::Allow, Duration::Always, none), - ]) - .any_fast_allow_rule(), - "one lasting outright allow among denies is enough" - ); - } - - /// The one predicate the fast path's safety rests on, over its whole - /// input space rather than a sample: three actions and four durations is - /// all of it. - /// - /// The expected answers are a table, not the implementation's own - /// expression rewritten - a test that recomputes what it is checking - /// passes for a wrong implementation too. Adding a variant to either enum - /// breaks this test, which is the point: a new action or a new duration is - /// a decision about whether it may mark a socket, and it should not be - /// possible to make it by accident. - #[test] - fn only_a_lasting_allow_may_ever_mark_a_socket() { - use cfc_core::{Action, Duration}; - - // Everything that may. Everything else may not. - let may_mark = [ - (Action::Allow, Duration::Always), - (Action::Allow, Duration::UntilRestart), - ]; - - let every_action = [Action::Allow, Action::Deny, Action::Reject]; - let every_duration = [ - Duration::Once, - Duration::UntilRestart, - Duration::Always, - // Both ends of the timed range: a timed allow must never mark, - // because the mark outlives the second it expires on - nothing - // re-passes a hook just because a clock ticked. - Duration::Seconds(0), - Duration::Seconds(u32::MAX), - ]; - - for action in every_action { - for duration in every_duration { - let verdict = ProcessWideVerdict { - action, - rule_id: uuid::Uuid::nil(), - duration, - }; - let expected = may_mark.contains(&(action, duration)); - assert_eq!( - verdict.fast_allow_eligible(), - expected, - "{action:?} + {duration:?} must {} be fast-allow eligible", - if expected { "" } else { "not" } - ); - } - } - } - - /// A rule that says anything about the flow must never become a blanket - /// mark on a process's sockets. - /// - /// `allow --dst-port 443` means "this program may reach 443", and a mark - /// means "every packet this program sends skips the queue". Conflating - /// them would be the widest possible failure of this feature, so each - /// predicate that makes a rule flow-scoped is checked on its own - a - /// missing one would only show up as a rule type that quietly grants - /// everything. - #[test] - fn a_flow_scoped_allow_never_grants_the_fast_path() { + fn a_flow_scoped_rule_never_yields_a_process_wide_action() { let exe = PathBuf::from("/usr/bin/curl"); let proc = Process { exe: exe.clone(), ..Process::unknown(1) }; - // The unconstrained rule does grant - otherwise the cases below would + // The unconstrained rule does answer - otherwise the cases below would // pass for the wrong reason. let mut open = RuleScope::any(); open.exe_path = Some(exe.clone()); - let engine = engine_with(vec![Rule::new("open".to_string(), Action::Allow, open)]); - assert!( - engine - .process_wide_verdict(&proc) - .is_some_and(|v| v.fast_allow_eligible()), - "an exe-only allow is the case this feature exists for" - ); + let engine = engine_with(vec![Rule::new("open".to_string(), Action::Deny, open)]); + assert_eq!(engine.process_wide_action(&proc), Some(Action::Deny)); // Named, because clippy is right that the bare tuple is a mouthful - // and because the name says what the table is: one way each of making @@ -1074,36 +728,15 @@ mod tests { let mut scope = RuleScope::any(); scope.exe_path = Some(exe.clone()); constrain(&mut scope); - let engine = engine_with(vec![Rule::new(what.to_string(), Action::Allow, scope)]); - let granted = engine - .process_wide_verdict(&proc) - .is_some_and(|v| v.fast_allow_eligible()); - assert!( - !granted, - "an allow scoped by {what} must not mark every socket this process opens" + let engine = engine_with(vec![Rule::new(what.to_string(), Action::Deny, scope)]); + assert_eq!( + engine.process_wide_action(&proc), + None, + "a deny scoped by {what} must not refuse every connect() this process makes" ); } } - #[test] - fn process_wide_verdict_names_the_rule_that_answered() { - // The allow consumer credits hits by this id; the wrong id would - // credit the wrong rule, which is worse than crediting none. - let mut scope = RuleScope::any(); - scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); - let rule = Rule::new("mine".to_string(), Action::Allow, scope); - let id = rule.id; - let engine = engine_with(vec![rule]); - let proc = Process { - exe: PathBuf::from("/usr/bin/curl"), - ..Process::unknown(1) - }; - let v = engine.process_wide_verdict(&proc).expect("a verdict"); - assert_eq!(v.rule_id, id); - assert_eq!(v.action, Action::Allow); - assert_eq!(engine.process_wide_action(&proc), Some(Action::Allow)); - } - /// An expired rule must drop out of what gets compiled into the kernel. /// /// The kernel table is rebuilt on rule edits, and expiry is not one. The @@ -1522,7 +1155,7 @@ mod tests { let original = allow_port_rule(80); let id = original.id; let engine = engine_with(vec![original]); - engine.record_hit(id); + engine.evaluate(&conn(80), &proc("/usr/bin/curl")); let mut batch = engine.snapshot().rules; batch.push(allow_port_rule(443)); engine.replace_rules(batch); diff --git a/crates/cfc-daemon/src/ebpf.rs b/crates/cfc-daemon/src/ebpf.rs index 4a40a0e..e7a50e9 100644 --- a/crates/cfc-daemon/src/ebpf.rs +++ b/crates/cfc-daemon/src/ebpf.rs @@ -374,131 +374,6 @@ pub fn enforcement_level() -> Option { } } -/// Whether the fast-allow path - a process-wide allow marking its sockets so -/// nftables accepts them ahead of the queue - is actually doing anything. -/// -/// Two-way, with the reason attached. `Off` carries *why*, because the path -/// has many ways to be silently inert - the config switch, a kernel whose -/// verifier lacks `bpf_setsockopt` on sock_addr, exit not tracked at all, ring -/// consumers that did not start, an nftables set the snippet does not declare, -/// a table that is not loaded yet, a ruleset reload that emptied the set - and -/// a feature that is off for a reason nobody can read is -/// a feature nobody can rely on. That lesson was learned once already with the -/// enforcement level above. -/// -/// Not "an inherited attach from a build that predates it", which this list -/// used to open with. The pin directory carries the ABI version and the fast -/// path arrived with a version bump, so a daemon never inherits pins from a -/// build that lacks it - it would find no pins at that path at all and attach -/// fresh. -#[derive(Debug, Clone, PartialEq, Eq)] -pub enum FastAllow { - /// Marks are being set and the nftables set holds this daemon's value. - /// - /// `deadline_secs` is how long a grant outlives a daemon that stops - /// refreshing it: the full [`fast_allow::DEADLINE_SECS`] when every - /// guarantee holds, the shorter [`fast_allow::DEADLINE_SECS_REDUCED`] when - /// one does not. `reduced` says which - the exec/exit links could not be - /// pinned, or exit is detected by leader only and grants are swept every - /// beat - or is `None` for the full guarantee. - /// - /// Carried rather than assumed, because a status line that says `live` - /// without saying which guarantee is a status line that misleads on - /// exactly the kernels where the guarantee is weaker. Two different - /// weaknesses give the same six seconds, so the number alone would not do. - Live { - deadline_secs: u64, - reduced: Option, - }, - /// Not doing anything, and this is the one sentence that says why. - Off(String), -} - -impl FastAllow { - /// One token plus the reason, for `cfc status`: `live` or `off: `. - pub fn describe(&self) -> String { - match self { - Self::Live { - reduced: None, - deadline_secs, - } if *deadline_secs == cfc_ebpf_common::fast_allow::DEADLINE_SECS => "live".to_string(), - Self::Live { - deadline_secs, - reduced, - } => match reduced { - Some(why) => format!("live, grants lapse within {deadline_secs}s ({why})"), - None => format!("live, grants lapse within {deadline_secs}s"), - }, - Self::Off(why) => format!("off: {why}"), - } - } -} - -static FAST_ALLOW_LEVEL: std::sync::RwLock> = std::sync::RwLock::new(None); - -/// Records the fast path's current state for `cfc status`. -/// -/// Not "called once": the loader publishes the startup decision before the -/// heartbeat exists, the heartbeat publishes every arm and every loss of the -/// nftables element, the late withdrawal publishes its refusal, and `Drop` -/// publishes the stop. Last writer wins, which is why the loader must publish -/// *before* spawning the heartbeat rather than after it returns. -pub fn set_fast_allow_level(level: FastAllow) { - *FAST_ALLOW_LEVEL - .write() - .unwrap_or_else(std::sync::PoisonError::into_inner) = Some(level); -} - -/// The fast path's state, `None` until startup has answered - the same -/// "ask again" sentinel `enforcement_level` uses, for the same reason. -pub fn fast_allow_level() -> Option { - FAST_ALLOW_LEVEL - .read() - .unwrap_or_else(std::sync::PoisonError::into_inner) - .clone() -} - -/// What this kernel's verifier let the fast path have. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub enum FastPathCapability { - /// Cookie connect variants and both sendmsg programs verified. - Ready, - /// The connect hooks fell back to the `_basic` twins, which carry no - /// `mark_decision` at all: nothing would ever mark a socket, so the fast - /// path cannot run. In the same era of kernels, no `bpf_setsockopt` on - /// sock_addr either. - BasicConnect, - /// The connect hooks took with the mark decision in them, and a sendmsg - /// hook did not load, attach or pin. The path runs; this is a caveat. - /// - /// The likely cause is the verifier: this kernel allows `bpf_getsockopt` / - /// `bpf_setsockopt` on connect programs and not yet on UDP sendmsg ones - - /// 5.10 answers `unknown func bpf_getsockopt#57`, 6.12 accepts. It is not - /// the only cause, which is why neither this comment nor `caveat` states it - /// as fact: `attach_one` also fails at the attach, at taking the link, and - /// at pinning. The real error is in the log line beside it. - /// - /// Why this stopped refusing the path: the sendmsg hooks used to re-decide - /// a UDP socket's mark per datagram, and were load-bearing. No UDP socket - /// is marked any more, so all they can do is strip a mark somebody - /// *forged* onto an unconnected UDP socket - a process that was granted, - /// learned the value with `getsockopt`, was revoked, and set it back. - /// Defence in depth against a narrow attacker, worth having where the - /// kernel allows it and not worth the whole feature where it does not. - SendmsgUnavailable, -} - -impl FastPathCapability { - /// One grep-able word for the startup log line and the matrix summary. - pub fn as_str(self) -> &'static str { - match self { - Self::Ready => "ready", - Self::BasicConnect => "basic-connect", - Self::SendmsgUnavailable => "sendmsg-unavailable", - } - } -} - /// What actually came up. Reported once at startup and otherwise inert. #[derive(Debug, Clone, Default, PartialEq, Eq)] pub struct Report { @@ -536,45 +411,20 @@ pub struct Report { /// more expensive" into something visible here, rather than into "it /// stopped loading on someone else's kernel". pub verified_insns: Vec<(String, u32)>, - /// The fast path's state after startup, `None` when the layer never got - /// far enough to have an opinion (eBPF off, object not loaded). - pub fast_allow: Option, /// Whether process exit is detected exactly (`group_dead` read from the - /// tracepoint record) rather than approximated by leader exit. A deny - /// evicted late is an inconvenience; a fast-allow grant evicted late is a - /// mark on a recycled pid - so when this is false the fast path runs with - /// the reduced deadline and sweeps its grants on every heartbeat, dropping - /// any pid whose start time no longer matches the one recorded at grant. - /// It used to refuse the path outright, which withheld it from every - /// kernel without `group_dead` - 5.10 and 6.12 in the matrix. + /// tracepoint record) rather than approximated by thread-group leader + /// exit. A kernel fact the matrix asserts; where it is false the exit + /// consumer confirms a leader exit against /proc before evicting. pub exit_precise: bool, /// Whether *both* lifecycle tracepoint links were pinned to bpffs, rather /// than merely attached. /// - /// The two are not the same and the difference is the fast path's whole - /// safety argument. `attach_tracepoint` re-attaches unpinned when a link - /// cannot be pinned - no `BPF_LINK_TYPE_PERF_EVENT` before 5.15, or a - /// read-only bpffs - which keeps eviction working for as long as this - /// daemon runs and stops the moment it does not. The connect programs' own - /// links are pinned separately and go on marking sockets either way, so - /// the difference is whether a grant can outlive this daemon at all - and - /// this flag is what tells the two cases apart. - /// - /// It is one of the two facts that select the reduced deadline rather than - /// gating the path - the other is `exit_precise` - and `cfc status` names - /// which of the two applies. For a day it was a rung on the eligibility - /// ladder instead. + /// `attach_tracepoint` re-attaches unpinned when a link cannot be pinned - + /// no `BPF_LINK_TYPE_PERF_EVENT` before 5.15, or a read-only bpffs - which + /// keeps eviction working for as long as this daemon runs and stops the + /// moment it does not, while the connect programs' own links stay pinned + /// and go on refusing. This flag is what tells the two cases apart. pub lifecycle_pinned: bool, - /// What this kernel's verifier let the fast path's kernel side have: - /// the cookie connect variants with both sendmsg hooks, the variants - /// without them, or only the `_basic` twins that mark nothing. The one - /// fact of the eligibility ladder that nothing else in this report - /// carries, and on a kernel older than 5.16 - which reports no verified - /// instruction counts - the only trace of which programs attached. `None` - /// where no connect hook attached at all: a layer that is off or could not - /// attach has no capability to report, and must not read as having the - /// `_basic` twins. - pub fast_path_capability: Option, } impl Report { @@ -705,19 +555,6 @@ impl Report { exit_tracking = self.exit_tracking, dns_capture = self.dns_capture, ppid_from_btf = self.ppid_offsets, - fast_path = self - .fast_path_capability - .map_or("none", FastPathCapability::as_str), - // The fast path has five reasons to be off and they used to reach - // `cfc status` only. An operator who set `fast_allow = true`, - // restarted, and never ran the CLI had no way to learn from the - // journal that the kernel had refused a hook - which is the whole - // point of degrading loudly. - fast_allow = self - .fast_allow - .as_ref() - .map(FastAllow::describe) - .unwrap_or_else(|| "not decided".to_string()), "attribution sources: sock_diag + /proc{}; hostnames: PTR + FCrDNS{}", if self.exec_tracking { " + eBPF exec events" @@ -754,18 +591,17 @@ pub fn nft_table_loaded() -> anyhow::Result { nft_set::table_loaded() } -/// Flushes a previous daemon's fast-allow mark out of the nftables set, for -/// the starts where [`start`] never reaches the loader's own flush: the layer -/// switched off in the config, or a build without it. The set outlives -/// daemons and the accept rule reads it whether or not anything still marks, -/// so a daemon that crashed while armed and came back with the layer off -/// would otherwise leave a standing bypass token behind it. A table that is -/// not loaded yet is not an error here - the nft unit is ordered after the -/// daemon - and `--dry-run` must not call this at all: it touches nothing, and -/// `main` is the one that knows it is running. -pub fn flush_stale_fast_allow() { - if let Err(e) = nft_set::disarm_for_start() { - tracing::error!("could not disable previous Fast Allow state: {e:#}; old marks may still bypass filtering; run systemctl reload colony-firewall-nft and inspect the journal before relying on filtering"); +/// Flushes the legacy `fast_allow` nftables set, once, at daemon start. +/// +/// Fast Allow shipped opt-in in 0.4.0 and is gone, but a 0.4-0.6 daemon that +/// crashed while armed can have left its mark in a set that a ruleset not yet +/// reloaded still accepts. Called from `main` in every build, whatever the +/// layer's mode, and never under `--dry-run`, which touches nothing. A table +/// or set that is not loaded is nothing to flush: at boot the nft unit is +/// ordered after the daemon. +pub fn flush_legacy_fast_allow_set() { + if let Err(e) = nft_set::flush() { + tracing::error!("could not flush the legacy fast_allow nftables set: {e:#}; a mark left by an older daemon may still bypass filtering; run systemctl reload colony-firewall-nft and inspect the journal before relying on filtering"); } } @@ -787,21 +623,10 @@ pub fn start( #[cfg_attr(not(feature = "ebpf"), allow(unused_variables))] engine: Option< crate::decision::Engine, >, - // The fast path's reporting: flows the kernel waved past the queue are - // fed back into the same observed stream and the same counters NFQUEUE - // feeds, so the live feed and the enforcing heuristic keep telling the - // truth about traffic the packet path never sees. - #[cfg_attr(not(feature = "ebpf"), allow(unused_variables))] - observed: tokio::sync::broadcast::Sender, - #[cfg_attr(not(feature = "ebpf"), allow(unused_variables))] stats: crate::stats::Stats, ) -> Runtime { - if cfg.fast_allow { - tracing::warn!("Fast Allow is disabled: socket marks cannot verify the current sender; use normal NFQUEUE filtering and remove fast_allow = true from daemon.toml"); + if cfg.fast_allow || cfg.fast_allow_mark.is_some() { + tracing::warn!("[ebpf] fast_allow and fast_allow_mark are ignored: Fast Allow was removed because a socket mark cannot prove which process sent a packet; delete them from daemon.toml"); } - // The answer until the layer says otherwise. Every early return below - // leaves it standing, which is the truthful default: no layer, no fast - // path. - set_fast_allow_level(FastAllow::Off("the in-kernel layer is not up".to_string())); if !cfg.enabled.wants_load() { return Runtime { report: Report::inert( @@ -825,9 +650,6 @@ pub fn start( // `dns` and `table` are the loader's inputs; without it they are // simply never wired to anything. let _ = (dns, table); - // And the loader's flush of a predecessor's mark is never reached in - // this build, so it happens here. - flush_stale_fast_allow(); Runtime { report: Report::inert_because( cfg.enabled, @@ -858,19 +680,7 @@ pub fn start( loader::Trust::Refuse, ), }; - match loader::load_and_attach( - &path, - dns, - table.clone(), - engine, - trust, - observed, - stats, - loader::FastAllowCfg { - on: cfg.fast_allow, - mark: cfg.fast_allow_mark, - }, - ) { + match loader::load_and_attach(&path, dns, table.clone(), engine, trust) { Ok((attached, mut report)) => { // The loader builds its report before it knows how it was // asked for; only `start` does. Without this an `auto` host @@ -878,33 +688,20 @@ pub fn start( // error policy. report.mode = cfg.enabled; table.set_live(report.exec_tracking); - // The fast-allow level is NOT published here. The loader - // publishes it before it spawns the heartbeat, because that - // task publishes too - and this call site runs after both, so - // it could only ever overwrite a fresher answer with a staler - // one. It did: on a restart the heartbeat arms immediately and - // says `Live`, and this line put "waiting for the nftables - // table" back on top of it, permanently. Runtime { report, _attached: Some(attached), } } - Err(e) => { - // The loader never got far enough to publish one. - set_fast_allow_level(FastAllow::Off( - "the in-kernel layer did not come up".to_string(), - )); - Runtime { - report: Report::inert_because( - cfg.enabled, - true, - e.degrade, - format!("load failed, continuing without it: {:#}", e.source), - ), - _attached: None, - } - } + Err(e) => Runtime { + report: Report::inert_because( + cfg.enabled, + true, + e.degrade, + format!("load failed, continuing without it: {:#}", e.source), + ), + _attached: None, + }, } }; @@ -928,8 +725,6 @@ mod tests { DnsCache::new(), proc_table::KernelProcTable::new(), None, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), ); assert!(!rt.report.any_active()); assert_eq!(rt.report.mode, EbpfMode::Off); @@ -962,14 +757,7 @@ mod tests { // what *this* load did, and against the process-wide table it would be // an assertion about every other test in the binary as well. let table = proc_table::KernelProcTable::new(); - let rt = start( - &cfg, - DnsCache::new(), - table.clone(), - None, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - ); + let rt = start(&cfg, DnsCache::new(), table.clone(), None); assert_eq!(rt.report.mode, EbpfMode::On); assert!(!rt.report.any_active(), "nothing can have attached"); assert_eq!(rt.report.ring0(), Ring0::Unavailable); @@ -1023,14 +811,7 @@ mod tests { ); let table = proc_table::KernelProcTable::new(); - let rt = start( - &cfg, - DnsCache::new(), - table.clone(), - None, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - ); + let rt = start(&cfg, DnsCache::new(), table.clone(), None); for note in &rt.report.notes { println!("note: {note}"); } diff --git a/crates/cfc-daemon/src/ebpf/enforce.rs b/crates/cfc-daemon/src/ebpf/enforce.rs index f19e456..364e6df 100644 --- a/crates/cfc-daemon/src/ebpf/enforce.rs +++ b/crates/cfc-daemon/src/ebpf/enforce.rs @@ -104,24 +104,21 @@ pub(super) const MAP_DENY_EVENTS: &str = "DENY_EVENTS"; pub(super) const MAP_EXE_RULES: &str = "EXE_RULES"; pub(super) const MAP_EXE_RULES_ON: &str = "EXE_RULES_ON"; -/// The fast path's maps. All four pinned, for the reason every enforcement -/// map is: a restarting daemon must steer the maps the still-attached -/// programs read, and an unpinned one would be a fresh map nobody reads. -/// Pinning is also what makes the deadline necessary - the programs keep -/// these alive after the daemon dies, so nothing empties `FAST_ALLOW` by -/// itself; `FAST_ALLOW_UNTIL` running out is what stops the marks. +/// The legacy Fast Allow maps. The daemon no longer grants, but the kernel +/// object still carries them (ABI v4) and the connect programs still read +/// them, so they stay pinned: [`disarm_legacy_fast_allow`] has to reach the +/// copies the pinned programs see, not a fresh map nobody reads. pub(super) const MAP_FAST_ALLOW: &str = "FAST_ALLOW"; pub(super) const MAP_FAST_ALLOW_UNTIL: &str = "FAST_ALLOW_UNTIL"; pub(super) const MAP_FAST_ALLOW_MARK: &str = "FAST_ALLOW_MARK"; pub(super) const MAP_ALLOW_EVENTS: &str = "ALLOW_EVENTS"; -/// The mark decision for UDP that never calls `connect()`. Fast-path only: -/// they refuse nothing, so a failure to attach them costs the fast path and -/// not enforcement, and they have no `_basic` twins. -pub(super) const PROG_SENDMSG4: &str = "cfc_sendmsg4"; -pub(super) const PROG_SENDMSG6: &str = "cfc_sendmsg6"; +/// Pin names a 0.4-0.6 daemon gave its Fast Allow sendmsg links, and the +/// directory it left beside the connect pins. Nothing attaches or reads these +/// any more; [`prepare`] removes them. pub(super) const LINK_SENDMSG4: &str = "sendmsg4"; pub(super) const LINK_SENDMSG6: &str = "sendmsg6"; +const LEGACY_COOKIE_MARKER: &str = "cookie-variants"; /// Pin name for the `sched_process_exit` link. /// @@ -182,58 +179,6 @@ pub(super) struct VerdictSink { /// What was last written to the kernel, so an unchanged recompute costs no /// syscalls. `None` until the first compile. last_compiled: Arc>>>, - /// The fast path's maps, `None` when this daemon must not grant: the - /// object predates them, or the loader judged the path ineligible (see - /// `FastAllow::Off`). Granting is gated here rather than at each call - /// site so an ineligible daemon cannot grant by accident from one path - /// and not another. - fast: Option, -} - -/// The kernel side of the fast path, from the daemon's chair. -/// -/// One rule for every writer here: **grants are re-earned, never inherited.** -/// The kernel clears `FAST_ALLOW` on exec and exit by itself; this side only -/// adds entries, and only for a process whose process-wide verdict is an -/// allow from a rule that lasts. Anything else - a deny, an abstention, a -/// destination-scoped rule, a timed allow, a process the engine cannot decide: -/// each of these *removes* the entry. There is no "keep" arm as there is for -/// denies, because a deny kept in doubt fails closed and an allow kept in -/// doubt is a bypass. -#[derive(Clone)] -pub(super) struct FastAllowMaps { - map: Arc>>, - until: Arc>>, - mark: Arc>>, - /// pid -> the rule that granted and the start time it was judged at. - /// - /// The rule is so an `ALLOW_EVENTS` record can credit the hit the packet - /// path will never see. The start time is what `sweep_stale_grants` compares - /// against, on a kernel whose exit detection is leader-only: a pid whose - /// start time no longer matches is not the process that was granted. - /// Userspace-only; a pid missing here when its event arrives is a grant - /// from a previous daemon, credited to nobody rather than to the wrong - /// rule. - granted_by: Arc>>, -} - -/// What the daemon remembers about one grant it wrote. -#[derive(Debug, Clone, Copy)] -struct Granted { - rule: uuid::Uuid, - /// `/proc//stat` field 22 when the grant was judged. `None` cannot - /// reach the map - `grant_if_still` refuses to grant without one - but the - /// type says so rather than the comment. - starttime: Option, -} - -/// One grant decision, for the three writers to share. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -enum Grant { - /// Write the entry; the rule that justifies it. - Yes(uuid::Uuid), - /// Remove the entry, whatever it held. - No, } impl VerdictSink { @@ -266,26 +211,6 @@ impl VerdictSink { ); } - // All three or none: a fast path with a grant map but no deadline map - // would be one the kernel honours forever, which is the exact state - // the deadline exists to make impossible. - let fast = match ( - bpf.take_map(MAP_FAST_ALLOW) - .and_then(|m| BpfHashMap::<_, u32, u32>::try_from(m).ok()), - bpf.take_map(MAP_FAST_ALLOW_UNTIL) - .and_then(|m| aya::maps::Array::<_, u64>::try_from(m).ok()), - bpf.take_map(MAP_FAST_ALLOW_MARK) - .and_then(|m| aya::maps::Array::<_, u32>::try_from(m).ok()), - ) { - (Some(map), Some(until), Some(mark)) => Some(FastAllowMaps { - map: Arc::new(Mutex::new(map)), - until: Arc::new(Mutex::new(until)), - mark: Arc::new(Mutex::new(mark)), - granted_by: Arc::new(Mutex::new(std::collections::HashMap::new())), - }), - _ => None, - }; - Ok(Self { map: Arc::new(Mutex::new(map)), engine, @@ -293,353 +218,9 @@ impl VerdictSink { exe_rules, exe_rules_on, last_compiled: Arc::new(Mutex::new(None)), - fast, }) } - /// The engine this sink decides with, for the allow consumer to credit - /// hits against. - pub(super) fn engine(&self) -> &Engine { - &self.engine - } - - /// Whether this sink can grant at all. The loader consults it to decide - /// the reported `FastAllow` state; it is true iff the maps exist. - pub(super) fn has_fast_path(&self) -> bool { - self.fast.is_some() - } - - /// Withdraws the ability to grant, for a daemon that loaded the maps but - /// then judged the path ineligible (config off, basic connect variants, - /// exit not tracked, could not arm) or lost it after arming (the ring - /// consumers did not start). Also unarms the kernel side and empties the - /// map, so every hook takes its cheapest exit and nothing is left to - /// honour whatever the deadline says. - pub(super) fn withdraw_fast_path(&mut self) { - if let Some(fast) = self.fast.take() { - // Unarm the kernel side too, not only empty the map. The maps are - // pinned, so a daemon that crashed while armed leaves - // `FAST_ALLOW_MARK` set; a successor started with the fast path - // *off* used to leave it that way, and every TCP `connect()` on the - // machine then paid a `getsockopt` and two map reads to strip a - // mark nobody would ever set - for as long as that daemon ran. With - // `UNARMED` written, `mark_decision` returns at its first array - // read, and the exec/exit programs skip their grant delete as well. - if let Err(e) = fast.until.lock().set(0, 0u64, 0) { - warn!("could not zero FAST_ALLOW_UNTIL while withdrawing the fast path: {e}"); - } - if let Err(e) = fast - .mark - .lock() - .set(0, cfc_ebpf_common::fast_allow::UNARMED, 0) - { - warn!("could not unarm FAST_ALLOW_MARK while withdrawing the fast path: {e}"); - } - let mut map = fast.map.lock(); - let pids: Vec = map.keys().flatten().collect(); - for pid in pids { - let _ = map.remove(&pid); - } - } - } - - /// The grant decision for one process: the shared rule for every writer. - fn grant_for(&self, proc: &Process) -> Grant { - // Abstain wherever a uid-scoped rule could apply. This decider and the - // packet path read different uids for a process that dropped - // privileges, and a grant is process-wide - see - // `Engine::uid_scoped_may_apply`, which explains why the answer is to - // step aside rather than to pick a uid. - if self.engine.uid_scoped_may_apply(&proc.exe) { - return Grant::No; - } - match self.engine.process_wide_verdict(proc) { - Some(v) if v.fast_allow_eligible() => Grant::Yes(v.rule_id), - _ => Grant::No, - } - } - - /// Applies a grant decision to the kernel map, under the caller's lock. - /// `judged_at` is the start time a `Yes` was decided against; a `No` does - /// not need one. - fn apply_grant( - fast: &FastAllowMaps, - map: &mut BpfHashMap, - pid: u32, - grant: Grant, - judged_at: Option, - ) { - match grant { - Grant::Yes(rule) => { - if let Err(e) = map.insert(pid, cfc_ebpf_common::fast_allow::GRANTED, 0) { - warn!(pid, "could not write a fast-allow grant: {e}"); - return; - } - fast.granted_by.lock().insert( - pid, - Granted { - rule, - starttime: judged_at, - }, - ); - } - Grant::No => { - // Absent is the common case and not an error: see `clear`. - let _ = clear(map, pid); - fast.granted_by.lock().remove(&pid); - } - } - } - - /// Recomputes the grant for one process, from any writer. - /// - /// `judged_at` is the start time the caller read `proc` at - see - /// [`grant_if_still`](Self::grant_if_still) for why a grant needs it and a - /// withdrawal does not. - fn regrant(&self, pid: u32, judged_at: Option, proc: &Process) { - self.grant_if_still(pid, proc, judged_at, self.grant_for(proc)); - } - - /// Withdraws any grant for `pid`, unconditionally. - /// - /// The counterpart to `grant_if_still`, and deliberately unguarded: - /// removing a grant from a pid whose owner changed is harmless, because - /// the new owner has not earned one yet and its own exec flow will grant - /// it if a rule says so. Withdrawing is always the safe direction. - fn drop_grant(&self, pid: u32) { - if let Some(fast) = self.fast.as_ref() { - Self::apply_grant(fast, &mut fast.map.lock(), pid, Grant::No, None); - } - } - - /// Applies a grant decision, but writes a *grant* only if `pid` still holds - /// the process the caller judged. - /// - /// Between the /proc read that produced the decision and this write there - /// is real work - the read itself, the engine call, and this lock - while - /// `on_exec` and the pinned exec program keep running. The pid can be - /// recycled in that window, and a grant landing on its new owner is a - /// marked socket that owner never earned: the fail-open direction, on a - /// process that may match no rule at all. - /// - /// The deny side has had this guard since the orphan sweep was written - - /// `doomed` carries the start time each pid was judged at - and the grant - /// side, which needs it more, did not have it. - fn grant_if_still(&self, pid: u32, judged: &Process, judged_at: Option, grant: Grant) { - let Some(fast) = self.fast.as_ref() else { - return; - }; - if !matches!(grant, Grant::Yes(_)) { - Self::apply_grant(fast, &mut fast.map.lock(), pid, grant, None); - return; - } - - // `None` is not a start time, it is the absence of a process, and - // comparing it directly let a grant through on exactly the case that - // must refuse. `proc_view` reads the exe, then the uid, then the start - // time; a process that exits in between yields a full view with - // `judged_at = None`, and `None != None` is false, so the guard fell - // through and wrote a grant for a pid with no process. Nothing would - // have cleared it either: the kernel's exec and exit clears both - // belong to a process that has already gone, and a pid recycled by a - // fork that never execs would inherit the mark. - let Some(judged_at) = judged_at else { - debug!(pid, "not granting: no start time, so no process to grant"); - return; - }; - if proc_starttime(pid) != Some(judged_at) { - debug!( - pid, - "not granting: the pid changed hands while it was judged" - ); - return; - } - - // Before the write, so an execve that has already happened is not paid - // for with a real marked flow. - if proc_exe(pid).as_deref() != Some(judged.exe.as_path()) { - debug!( - pid, - "not granting: the program changed while the grant was decided" - ); - return; - } - - Self::apply_grant(fast, &mut fast.map.lock(), pid, grant, Some(judged_at)); - - // And once more, on the program rather than the pid - after the write, - // deliberately. - // - // `execve` keeps the start time (field 22 of /proc//stat is when - // the *process* began, not when it last exec'd), so the guard above - // cannot see one. That matters because the kernel's exec program - // removes this pid's grant on every execve: a process judged as an - // allowed binary, exec'ing into a denied one while this function was - // deciding, would have its grant correctly cleared by the kernel and - // then reinstated here, for a program nothing granted. - // - // Checked before the write as well as after - and neither makes this - // race-free, which the comment here used to claim. - // - // What remains is the interval between the pre-check and the insert. - // An execve landing there has its grant cleared by the kernel and then - // re-added by this write, and the post-write check removes it again - // only after a `read_link`. A `connect()` inside *that* window finds - // the entry present and marks the socket, and removing the map entry - // afterwards does not unmark it. So the exposure is one flow rather - // than none. It is bounded, and no standing refusal is skipped - - // `VERDICTS` is consulted before `mark_decision` - but "narrower" is - // the honest word and "race-free" was not. - if proc_exe(pid).as_deref() != Some(judged.exe.as_path()) { - debug!( - pid, - "withdrawing: the program changed while the grant was decided" - ); - Self::apply_grant(fast, &mut fast.map.lock(), pid, Grant::No, None); - } - } - - /// The rule that granted `pid`, for crediting an `ALLOW_EVENTS` record. - pub(super) fn granted_by(&self, pid: u32) -> Option { - self.fast - .as_ref() - .and_then(|f| f.granted_by.lock().get(&pid).map(|g| g.rule)) - } - - /// Drops every grant whose pid no longer holds the process it was judged - /// for, and returns how many. - /// - /// For a kernel whose `sched_process_exit` record has no readable - /// `group_dead`. There the exit program evicts on thread-group *leader* - /// exit, so a process whose leader exits first and dies later is never - /// evicted - while this daemon is alive and refreshing the deadline, which - /// therefore bounds nothing. This is the bound instead: called on every - /// heartbeat, it compares each granted pid's current start time with the - /// one recorded when it was granted; a mismatch, or no process at all, is - /// a grant for whoever owns the pid next. The exposure that remains is a - /// pid recycled *and* connecting within one heartbeat, without an exec in - /// between (an exec clears the grant in the kernel). - /// - /// Snapshot first, then read /proc with no lock held: `on_exec` and - /// `on_exit` take `granted_by` from the ring consumers. - pub(super) fn sweep_stale_grants(&self) -> usize { - let Some(fast) = self.fast.as_ref() else { - return 0; - }; - let snapshot: Vec<(u32, Option)> = fast - .granted_by - .lock() - .iter() - .map(|(pid, g)| (*pid, g.starttime)) - .collect(); - let mut dropped = 0usize; - for (pid, recorded) in snapshot { - if !grant_is_stale(recorded, proc_starttime(pid)) { - continue; - } - // Re-read what is recorded *now*, not what the snapshot said. In - // the window since the snapshot this pid may have died, been - // recycled, exec'd, and been granted afresh by `on_exec` with its - // own start time - and dropping that grant on the strength of the - // old one would take a legitimate grant from a legitimate process - // until the next rule change. If the record moved, it is someone - // else's decision and stands. - let recorded_now = fast.granted_by.lock().get(&pid).map(|g| g.starttime); - if recorded_now != Some(recorded) { - continue; - } - self.drop_grant(pid); - dropped += 1; - } - dropped - } - - /// Empties `FAST_ALLOW`. At every start, before anything is granted: the - /// map is pinned, so it holds the previous daemon's grants, made under the - /// previous daemon's rules. Returns how many were dropped, for the log. - pub(super) fn flush_fast_allow(&self) -> usize { - let Some(fast) = self.fast.as_ref() else { - return 0; - }; - let mut map = fast.map.lock(); - let pids: Vec = map.keys().flatten().collect(); - let n = pids.len(); - for pid in pids { - let _ = map.remove(&pid); - } - fast.granted_by.lock().clear(); - n - } - - /// Writes the mark value the kernel side will set, arming the path. - /// - /// The deadline is zeroed first, and by this function rather than by - /// assumption. Callers used to say "the deadline is still zero until the - /// first `beat`, so nothing is honoured before the heartbeat runs", which - /// is not a property the code had: `FAST_ALLOW_UNTIL` is a *pinned* map, - /// so after an unclean death it holds whatever deadline the previous - /// daemon last wrote - up to a minute into the future. Nothing was - /// actually honoured on the strength of it, because the grant map is - /// flushed at start and the nft set holds no mark yet, but the sentence - /// was load-bearing in two comments and true in neither. One `set` makes - /// it true. - /// - /// Order matters: zero the deadline, then write the mark. Between the two - /// the kernel reads an armed mark against a lapsed deadline, counts a - /// `STALE`, and marks nothing. - pub(super) fn arm(&self, mark: u32) -> anyhow::Result<()> { - let fast = self - .fast - .as_ref() - .ok_or_else(|| anyhow!("no fast path to arm"))?; - fast.until - .lock() - .set(0, 0u64, 0) - .context("zeroing FAST_ALLOW_UNTIL")?; - fast.mark - .lock() - .set(0, mark, 0) - .context("writing FAST_ALLOW_MARK")?; - Ok(()) - } - - /// Pushes the deadline out to now + `deadline_secs` on `CLOCK_BOOTTIME`, - /// the clock `bpf_ktime_get_boot_ns` reads. Called every `HEARTBEAT_SECS` - /// by the runtime; if it ever stops being called, the kernel side stops - /// honouring grants within one deadline - by design, not by accident. - pub(super) fn beat(&self, deadline_secs: u64) -> anyhow::Result<()> { - let Some(fast) = self.fast.as_ref() else { - return Ok(()); - }; - let until = boottime_ns()? + deadline_secs * 1_000_000_000; - fast.until - .lock() - .set(0, until, 0) - .context("writing FAST_ALLOW_UNTIL")?; - Ok(()) - } - - /// Disarms immediately: zero deadline, unarmed mark, empty map. For a - /// clean shutdown, so the marks stop now rather than when the deadline - /// this daemon last wrote runs out. Best effort in every step - a daemon on its way out has - /// nowhere to report a failure but the log. - pub(super) fn disarm(&self) { - let Some(fast) = self.fast.as_ref() else { - return; - }; - if let Err(e) = fast.until.lock().set(0, 0u64, 0) { - warn!("could not zero FAST_ALLOW_UNTIL on shutdown: {e}"); - } - if let Err(e) = fast - .mark - .lock() - .set(0, cfc_ebpf_common::fast_allow::UNARMED, 0) - { - warn!("could not unarm FAST_ALLOW_MARK on shutdown: {e}"); - } - self.flush_fast_allow(); - } - /// Recomputes every live process's verdict. /// /// Called when the rule set changes, which is the only event that can @@ -655,12 +236,11 @@ impl VerdictSink { /// (`source="rule"`) and the kernel counters never moved. /// /// Cost, since it is no longer only map operations: a handful of small - /// /proc reads per process - three for the view, one to re-date it before - /// the deny write, and up to three more inside `grant_if_still` when the - /// answer is a grant - for every recently-exec'd process and every orphan, - /// and one walk of /proc for `sweep_fast_allow` at the end. All of it on whichever - /// thread changed the rules - an IPC handler or startup, never the packet - /// path - and only when a human or the CLI actually changed something. + /// /proc reads per process - three for the view and one to re-date it + /// before the deny write - for every recently-exec'd process and every + /// orphan. All of it on whichever thread changed the rules - an IPC + /// handler or startup, never the packet path - and only when a human or + /// the CLI actually changed something. /// /// None of those reads happen while a map lock is held. That is a /// constraint, not an accident: `on_exec` and `on_exit` block on the @@ -698,43 +278,18 @@ impl VerdictSink { .iter() .filter_map(|proc| { // From /proc, like every other decider - not from the exec - // record. + // record. The record carries the execve *string*, which is + // neither absolute for `./foo` nor resolved through a + // symlink, and the uid at exec, which a process that dropped + // privileges no longer holds; the orphan sweep below reads + // /proc, and two deciders that disagree about one process is + // the defect. // - // Both loops in this function ask one question about one - // process, and for a long time they asked it of different - // inputs: this one of the `ExecEvent` (the execve *string*, - // and the uid the process had when it exec'd), the orphan - // sweep below of /proc. Two deciders that disagree about the - // same process is the defect, and all three ways it showed up - // were fail-open: - // - // * `execve("./foo")` records no absolute path, so - // `absolute_exe` answered None and this loop skipped the pid - // whole. Deleting the rule that granted such a process, or - // replacing it with a Block, left the grant standing in the - // kernel: a marked socket past the queue for a program no - // rule allowed any more. - // * the execve string is what the caller typed, not what ran. - // A rule naming a symlink - or `/bin/curl` on a merged-usr - // system - granted here what `on_exec` and the packet path, - // both of which resolve, refuse. - // * the recorded uid is the uid at exec. A process that - // dropped privileges kept a uid-scoped grant it had stopped - // qualifying for until its next execve, and absent one, - // forever. - match proc_view(proc.pid) { - Some((view, judged_at)) => Some((proc.pid, view, judged_at)), - None => { - // Gone, or /proc unreadable. Do not fall back to the - // exec record - that is the guess this comment exists - // to refuse. Withdraw the grant (a grant kept in doubt - // is a marked socket) and leave the deny to the exit - // tracepoint, which owns eviction and can tell - // "exited" from "unreadable". - self.drop_grant(proc.pid); - None - } - } + // Gone, or /proc unreadable: no fallback to the exec record, + // which is the guess this comment exists to refuse. The deny + // is left to the exit tracepoint, which owns eviction and can + // tell "exited" from "unreadable". + proc_view(proc.pid).map(|(view, judged_at)| (proc.pid, view, judged_at)) }) .collect(); @@ -800,18 +355,6 @@ impl VerdictSink { } drop(map); - // The grant side, with the verdict lock released: `grant_if_still` - // takes the grant map's own mutex, and holding both at once would put - // an ordering constraint on two locks that otherwise never nest. - // - // No tri-state here: the same engine answer either says "allow, - // lasting" or the entry goes. In particular an abstention - which - // keeps a deny above - removes a grant, because a grant kept in doubt - // is a marked socket past the queue. - for (pid, as_process, judged_at) in &views { - self.grant_if_still(*pid, as_process, *judged_at, self.grant_for(as_process)); - } - // And the entries the live list does not cover. // // `live_processes` only returns pids the proc table has seen exec @@ -829,31 +372,12 @@ impl VerdictSink { // dropped record, which is a process with no in-kernel verdict. let known: std::collections::HashSet = live.iter().map(|p| p.pid).collect(); let map = self.map.lock(); - let mut orphans: Vec = map + let orphans: Vec = map .keys() .flatten() .filter(|pid| !known.contains(pid)) .collect(); drop(map); - // Grants have orphans too - a long-running allowed process drops off - // the live list on the same TTL - and a grant whose rule is gone must - // go with it. Walk the grant map's own keys; the loop below re-decides - // each pid from /proc and applies the grant answer alongside the deny - // answer, so the two maps never disagree about one process. - if let Some(fast) = self.fast.as_ref() { - let granted: Vec = fast - .map - .lock() - .keys() - .flatten() - .filter(|pid| !known.contains(pid)) - .collect(); - for pid in granted { - if !orphans.contains(&pid) { - orphans.push(pid); - } - } - } // Each doomed pid carries the start time it was judged at, so the // final pass can tell "still the process I judged" from "the kernel @@ -870,10 +394,8 @@ impl VerdictSink { // a uid-scoped allow - so guessing would clear a denial the allow // was never meant to lift, which is the fail-open direction. let Some((proc, judged_at)) = proc_view(pid) else { - // Gone. Clear, or a recycled pid inherits its answer - and a - // grant even more so. + // Gone. Clear, or a recycled pid inherits its answer. doomed.push((pid, None)); - self.drop_grant(pid); continue; }; // `None` is two opposite answers and they must not be conflated. @@ -895,12 +417,6 @@ impl VerdictSink { } } } - // The grant answer for the same process, from the same /proc - // read - with the real uid, so a process that dropped privileges - // loses a uid-scoped grant here rather than keeping what it - // earned as root - and against the same start time, so it cannot - // land on a pid the kernel recycled while this loop was working. - self.grant_if_still(pid, &proc, judged_at, self.grant_for(&proc)); } if !doomed.is_empty() { @@ -935,80 +451,6 @@ impl VerdictSink { processes = live.len(), denied, "resynced in-kernel verdicts after a rule change" ); - - // And the processes neither loop above can reach. - self.sweep_fast_allow(); - } - - /// Grants every process on the machine that a lasting rule allows. - /// - /// Every other writer of the grant map needs an *event*: `on_exec` needs an - /// execve, and the two loops above walk the proc table's recent execs and - /// the maps' own keys. None of them reaches a process that was already - /// running - which is exactly the population this feature exists for. It - /// showed up two ways, and in both the path reported `live` while doing - /// nothing at all: - /// - /// * after `systemctl restart colony-firewalld`. The pinned map is flushed - /// at start (those grants were made under the previous daemon's rules) - /// and the proc table starts empty, so the browser, the mail client - - /// everything long-lived - was never granted again for the rest of that - /// daemon's life. The restart is the common case: an upgrade, a crash, a - /// config reload. - /// * `allow --exe .../firefox always` on a browser started three hours ago. - /// The proc table's entries expire on a one-hour TTL, so the live loop - /// never saw it either. The feature only ever worked for a process that - /// exec'd *after* the daemon and less than an hour before its rule. - /// - /// So this walks /proc. O(processes), three small reads each and up to - /// three more for the ones a rule grants, on whichever thread changed the - /// rules - an IPC handler or startup, never - /// the packet path - and rule changes are paced by a human or the CLI. - /// - /// It only ever *adds*. Withdrawal is already covered and must stay where - /// it is: every pid holding a grant is re-decided by the live loop or the - /// orphan sweep above, and those two also handle pids that have left /proc - /// entirely, which this walk by construction cannot see. - pub(super) fn sweep_fast_allow(&self) { - if self.fast.is_none() { - return; - } - // A rule set that cannot grant anyone - only denies, only timed or - // flow-scoped allows - makes the walk below a few hundred /proc reads - // and engine calls for an answer already known. That is the common - // shape of a rule set with the fast path switched on, and this runs on - // every rule change. - if !self.engine.any_fast_allow_rule() { - debug!("no rule could grant the fast path; not walking /proc"); - return; - } - let entries = match std::fs::read_dir("/proc") { - Ok(e) => e, - Err(e) => { - warn!("could not read /proc to seed fast-allow grants: {e}"); - return; - } - }; - let (mut seen, mut granted) = (0usize, 0usize); - for entry in entries.flatten() { - let Some(pid) = entry - .file_name() - .to_str() - .and_then(|s| s.parse::().ok()) - else { - continue; - }; - seen += 1; - let Some((proc, judged_at)) = proc_view(pid) else { - continue; - }; - let grant = self.grant_for(&proc); - if matches!(grant, Grant::Yes(_)) { - granted += 1; - self.grant_if_still(pid, &proc, judged_at, grant); - } - } - debug!(seen, granted, "swept /proc for fast-allow grants"); } /// Decides whether this newly-exec'd process gets an in-kernel answer. @@ -1017,14 +459,10 @@ impl VerdictSink { /// process depends on a destination. Two things follow from that, both /// deliberate: /// - /// * **an allow is never written *here*.** `VERDICTS` holds denials only. - /// Allows that buy something - a lasting, process-wide one - go to the - /// fast-allow map through [`regrant`](Self::regrant) at the end of this - /// function, under rules of their own: cleared by the kernel on exec and - /// exit, honoured only while the daemon's heartbeat keeps the deadline - /// ahead of now, and re-earned per execve. A stale allow after pid reuse - /// is a security problem rather than an inconvenience, which is why the - /// two maps do not share a sweep. + /// * **an allow is never written.** `VERDICTS` holds denials only; an + /// allowed process keeps the packet path, which re-checks every flow. A + /// stale allow after pid reuse would be a security problem rather than an + /// inconvenience, which is why there is none to go stale. /// * **a stale entry is always cleared**, even when the answer is "no /// answer". A pid that re-execs into a different binary must not inherit /// the verdict written for the one before it. @@ -1053,8 +491,6 @@ impl VerdictSink { None => exe, } }); - // Dates the read above, for the grant at the end of this function. - let judged_at = resolved.is_some().then(|| proc_starttime(pid)).flatten(); // The uid here stays the event's, not a fresh read: at the moment of an // execve that *is* the process's uid, and a drop of privileges between // the kernel's tracepoint and this consumer is both vanishingly narrow @@ -1062,10 +498,9 @@ impl VerdictSink { // both. Mixing a live path with an event-time uid is worth naming // rather than leaving for a reader to find. // When /proc is unreadable, retain the unknown executable supplied by - // the event consumer. The execve argument cannot attest a mapped image. - // Missing identity keeps grants absent and leaves packet policy to - // NFQUEUE; it must not satisfy an executable-scoped rule. - let readable = resolved.is_some(); + // the event consumer. The execve argument cannot attest a mapped image, + // so missing identity leaves packet policy to NFQUEUE; it must not + // satisfy an executable-scoped rule. let corrected = match resolved { Some(exe) if exe != proc.exe => Some(Process { exe, @@ -1097,22 +532,6 @@ impl VerdictSink { debug!(pid, exe = %as_process.exe.display(), "in-kernel deny installed"); } drop(map); - - // The grant, re-earned for this exec. The kernel already removed the - // predecessor's entry on the exec path, so this is the daemon's only - // role in the fast path: say yes for the new binary, or say nothing. - // The engine is asked once more rather than reusing `deny` because - // the answer that matters here is "allow, from a rule that lasts", - // which the deny decision above did not compute. - if readable { - self.regrant(pid, judged_at, as_process); - } else { - // Nothing to grant for a pid we could not read. The kernel's exec - // path already removed the predecessor's entry, so this only - // clears the daemon-side bookkeeping that would otherwise credit - // an allow event to a rule for a process that no longer exists. - self.drop_grant(pid); - } } /// Compiles the process-wide rules into the kernel's own table. @@ -1226,54 +645,10 @@ impl VerdictSink { if let Err(e) = clear(&mut self.map.lock(), pid) { warn!(pid, "could not evict the in-kernel verdict: {e}"); } - // The kernel evicted the grant itself on the exit path; this drops - // the credit record so a recycled pid is never credited to a rule - // that granted its predecessor. - if let Some(fast) = self.fast.as_ref() { - let _ = clear(&mut fast.map.lock(), pid); - fast.granted_by.lock().remove(&pid); - } - } -} - -/// `CLOCK_BOOTTIME` in nanoseconds - the clock `bpf_ktime_get_boot_ns` -/// reads, which counts through suspend. The deadline it feeds must be sixty -/// wall-clock seconds, not sixty awake ones. -fn boottime_ns() -> anyhow::Result { - let mut ts = libc::timespec { - tv_sec: 0, - tv_nsec: 0, - }; - // SAFETY: a valid pointer to a timespec on our own stack; the call writes - // it and nothing else. - let rc = unsafe { libc::clock_gettime(libc::CLOCK_BOOTTIME, &mut ts) }; - if rc != 0 { - return Err(std::io::Error::last_os_error()).context("clock_gettime(CLOCK_BOOTTIME)"); - } - Ok(u64::try_from(ts.tv_sec).unwrap_or(0) * 1_000_000_000 - + u64::try_from(ts.tv_nsec).unwrap_or(0)) -} - -/// Whether a grant judged at `recorded` still belongs to the process now -/// holding its pid, given the start time read `now`. -/// -/// Only an exact match keeps a grant. `None` on either side is "no process": -/// a grant recorded without a start time cannot happen (`grant_if_still` -/// refuses it) but is treated as stale rather than trusted, and a pid with no -/// process now is a grant for whoever gets the pid next. Two live processes -/// never share a (pid, start time) pair within a boot. -fn grant_is_stale(recorded: Option, now: Option) -> bool { - match (recorded, now) { - (Some(a), Some(b)) => a != b, - _ => true, } } /// `/proc//exe`, with the kernel's `" (deleted)"` suffix stripped. -/// -/// One reader, because three callers want the same normalisation and two of -/// them are a guard and its counter-check - a difference between those two -/// would be a grant kept or withdrawn for a reason nobody wrote down. fn proc_exe(pid: u32) -> Option { let exe = std::fs::read_link(format!("/proc/{pid}/exe")).ok()?; let s = exe.to_string_lossy(); @@ -1295,7 +670,7 @@ fn proc_exe(pid: u32) -> Option { /// abstain and keeps the tri-state the sweep depends on. /// /// `None` means the process is gone or its /proc is unreadable, which callers -/// must treat as "no grant" rather than falling back to a guess. +/// must treat as "no answer" rather than falling back to a guess. fn proc_view(pid: u32) -> Option<(Process, Option)> { // `proc_exe` strips the kernel's " (deleted)" suffix - a package upgrade // under a running program, which process_resolve calls Tuesday on a rolling @@ -1351,9 +726,9 @@ fn proc_starttime(pid: u32) -> Option { // Same reason as `proc_uid`, and it matters more here: the kernel writes // comm unescaped into this file, and every guard built on this function // treats `None` as "no process". A program named with non-UTF-8 bytes was - // therefore never granted the fast path, and - once the deny pass started - // filtering on the start time - never given an in-kernel deny either, - // which is resync's whole job. Permanently, not transiently. + // therefore - once the deny pass started filtering on the start time - + // never given an in-kernel deny, which is resync's whole job. + // Permanently, not transiently. let stat = String::from_utf8_lossy(&std::fs::read(format!("/proc/{pid}/stat")).ok()?).into_owned(); // The comm field is parenthesised and may itself contain spaces and @@ -1399,22 +774,17 @@ fn is_absent(err: &aya::maps::MapError) -> bool { } /// Per-CPU counters, summed. See [`enforce_stat`]. +/// +/// Only the slots the daemon still reports. The kernel also counts the +/// legacy Fast Allow outcomes, and the slot layout stays as it is. #[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] pub(super) struct EnforceStats { - /// `connect()` calls allowed because the map held an allow. - pub allowed: u64, /// `connect()` calls refused in-kernel, before a packet existed. pub denied: u64, /// `connect()` calls with no entry, which went on to the packet path. pub unknown: u64, - /// Grants the kernel saw but did not honour because the deadline had - /// lapsed - a daemon that stopped heartbeating, seen from the kernel. - pub stale: u64, - /// Grants not applied because the socket carried a foreign mark - a VPN - /// or proxy marking its own sockets, left alone on purpose. - pub foreign_mark: u64, /// Decisions the kernel made and could not report, because the ring was - /// full. Non-zero means the daemon's view of the fast path undercounts. + /// full. pub report_dropped: u64, } @@ -1464,6 +834,7 @@ pub(super) fn prepare() -> anyhow::Result { unpin_other_versions(&ns); let dir = pin_dir(); std::fs::create_dir_all(&dir).with_context(|| format!("creating {}", dir.display()))?; + remove_legacy_pins(&dir); Ok(dir) } @@ -1520,28 +891,85 @@ pub(super) fn already_attached(dir: &Path) -> bool { dir.join("connect4").exists() && dir.join("connect6").exists() } -/// A directory `attach` creates beside the connect pins once *both* cookie -/// connect variants have attached, and nothing else. +/// Removes what a 0.4-0.6 daemon pinned for Fast Allow and nothing reads any +/// more: the two sendmsg links and the cookie-variant marker directory. /// -/// The inherited path needs to know which connect variant a previous daemon -/// left running, and the pin names do not say: `connect4` is `connect4` -/// whether it holds the cookie program or the basic twin. The sendmsg pins used -/// to be the evidence - written only after the cookie variants took - which -/// made a daemon that ran cookie variants *without* sendmsg (a kernel that -/// refuses `bpf_getsockopt` there) look, to its successor, like a basic-connect -/// daemon, and refuse the fast path on every restart. bpffs allows directories, -/// so the evidence is now its own object and says one thing only. -pub(super) const COOKIE_MARKER: &str = "cookie-variants"; - -/// True when a previous daemon left the marker: the pinned connect programs -/// are the cookie variants, which carry `mark_decision`. -pub(super) fn cookie_variants_pinned(dir: &Path) -> bool { - dir.join(COOKIE_MARKER).is_dir() +/// Removing a link pin drops the kernel's last reference and detaches the +/// program, so this is also what takes an older daemon's inert sendmsg hooks +/// off the cgroup, on the inherited path as much as on a fresh attach. Absent +/// is the ordinary case and says nothing. The legacy *maps* stay pinned: the +/// connect programs still read them, and [`disarm_legacy_fast_allow`] needs +/// to reach the pinned copies. +fn remove_legacy_pins(dir: &Path) { + for name in [LINK_SENDMSG4, LINK_SENDMSG6] { + let pin = dir.join(name); + match std::fs::remove_file(&pin) { + Ok(()) => debug!("removed the legacy {name} pin at {}", pin.display()), + Err(e) if e.kind() == io::ErrorKind::NotFound => {} + Err(e) => warn!( + "could not remove the legacy {name} pin at {}: {e}", + pin.display() + ), + } + } + let marker = dir.join(LEGACY_COOKIE_MARKER); + match std::fs::remove_dir(&marker) { + Ok(()) => debug!("removed the legacy marker at {}", marker.display()), + Err(e) if e.kind() == io::ErrorKind::NotFound => {} + Err(e) => warn!( + "could not remove the legacy marker at {}: {e}", + marker.display() + ), + } } -/// True when a previous daemon left both sendmsg programs pinned. -pub(super) fn sendmsg_pinned(dir: &Path) -> bool { - dir.join(LINK_SENDMSG4).exists() && dir.join(LINK_SENDMSG6).exists() +/// Withdraws whatever Fast Allow state the pinned maps still hold: a zero +/// deadline, the unarmed mark, and no grants. +/// +/// The daemon no longer grants, but the kernel object still carries the maps +/// (ABI v4) and the connect programs still consult them. They are pinned, so +/// a 0.4-0.6 daemon that died while armed left a mark the hooks would go on +/// setting across any number of restarts - a mark that can collide with the +/// fwmark selectors of kube-proxy, Tailscale or wg-quick. With `UNARMED` +/// written, `mark_decision` returns at its first array read. Best effort: a +/// map that is missing or will not take the write is logged and skipped. +pub(super) fn disarm_legacy_fast_allow(bpf: &mut Ebpf) { + if let Some(m) = bpf.map_mut(MAP_FAST_ALLOW_UNTIL) { + match aya::maps::Array::<_, u64>::try_from(m) { + Ok(mut until) => { + if let Err(e) = until.set(0, 0u64, 0) { + warn!("could not zero the legacy {MAP_FAST_ALLOW_UNTIL}: {e}"); + } + } + Err(e) => warn!("{MAP_FAST_ALLOW_UNTIL} is not an array: {e}"), + } + } + if let Some(m) = bpf.map_mut(MAP_FAST_ALLOW_MARK) { + match aya::maps::Array::<_, u32>::try_from(m) { + Ok(mut mark) => { + if let Err(e) = mark.set(0, cfc_ebpf_common::fast_allow::UNARMED, 0) { + warn!("could not unarm the legacy {MAP_FAST_ALLOW_MARK}: {e}"); + } + } + Err(e) => warn!("{MAP_FAST_ALLOW_MARK} is not an array: {e}"), + } + } + if let Some(m) = bpf.map_mut(MAP_FAST_ALLOW) { + match BpfHashMap::<_, u32, u32>::try_from(m) { + Ok(mut grants) => { + let pids: Vec = grants.keys().flatten().collect(); + for pid in pids { + match grants.remove(&pid) { + Err(e) if !is_absent(&e) => { + warn!(pid, "could not drop a legacy {MAP_FAST_ALLOW} grant: {e}") + } + _ => {} + } + } + } + Err(e) => warn!("{MAP_FAST_ALLOW} is not a hash map: {e}"), + } + } } /// Unpins any lone connect-link leftovers so a fresh attach starts clean. @@ -1552,19 +980,7 @@ pub(super) fn sendmsg_pinned(dir: &Path) -> bool { /// the packet path like any other unenforced moment - is the price of not /// being wedged forever. fn drop_half_attached(dir: &Path) { - // The marker goes with the pins it describes; a marker outliving them - // would tell the next daemon the cookie variants are running when nothing - // is. - let marker = dir.join(COOKIE_MARKER); - if marker.is_dir() { - if let Err(e) = std::fs::remove_dir(&marker) { - warn!( - "could not remove the stale cookie marker at {}: {e}", - marker.display() - ); - } - } - for name in ["connect4", "connect6", LINK_SENDMSG4, LINK_SENDMSG6] { + for name in ["connect4", "connect6"] { let pin = dir.join(name); if !pin.exists() { continue; @@ -1639,56 +1055,17 @@ fn attach_one( Ok(insns) } -/// What [`attach`] managed to put in place. -pub(super) struct AttachedPrograms { - /// Every program attached, with its verified instruction count. - pub programs: Vec<(String, Option)>, - /// Whether the kernel side of the fast path is in place, and if not, the - /// one sentence that says why - two different kernels give two different - /// answers, and reporting the wrong one sent a reader to the wrong - /// kernel version. - pub fast_path: FastPathCapability, -} - -// Defined in `ebpf.rs`, where the `Report` that carries it lives in every -// build; re-exported so the paths this module's callers use keep resolving. -pub(super) use super::FastPathCapability; - -impl FastPathCapability { - /// The one reason the fast path cannot run at all, or `None`. - pub(super) fn refusal(self) -> Option<&'static str> { - match self { - Self::BasicConnect => Some( - "the connect hooks fell back to the basic variants, which carry no mark \ - decision (usually no bpf_get_socket_cookie / bpf_setsockopt on sock_addr \ - programs; the log line beside this one has the kernel's actual answer)", - ), - Self::SendmsgUnavailable => Some("fast-allow requires both UDP sendmsg hooks"), - Self::Ready => None, - } - } - - /// A guarantee the path runs without, for the report, or `None`. - pub(super) fn caveat(self) -> Option<&'static str> { - match self { - Self::SendmsgUnavailable => Some( - "the cgroup/sendmsg hooks did not load or attach, so a mark forged onto an \ - unconnected UDP socket is not stripped (usually this kernel's verifier: \ - bpf_getsockopt/setsockopt on sendmsg needs a newer kernel than on connect - \ - 5.10 refuses, 6.12 accepts; the log line beside this one has the actual answer)", - ), - Self::Ready | Self::BasicConnect => None, - } - } -} - /// Attaches both connect programs, pinning them under `dir` when it is -/// `Some`, then the sendmsg pair when the cookie variants took. +/// `Some`, and returns every program attached with its verified instruction +/// count. /// /// `dir` is `None` when [`prepare`] failed: the programs still attach and still /// enforce, they just stop when this process does. That is strictly better than /// not attaching, and worse than pinning, so the caller says which happened. -pub(super) fn attach(bpf: &mut Ebpf, dir: Option<&Path>) -> anyhow::Result { +pub(super) fn attach( + bpf: &mut Ebpf, + dir: Option<&Path>, +) -> anyhow::Result)>> { let root = super::cgroup::v2_root() .ok_or_else(|| anyhow!("no cgroup2 mount in /proc/mounts (unified hierarchy required)"))?; // Read-only, for the same reason as the DNS attach: the kernel wants the @@ -1705,7 +1082,6 @@ pub(super) fn attach(bpf: &mut Ebpf, dir: Option<&Path>) -> anyhow::Result = Vec::with_capacity(4); - let mut cookie_variants = 0usize; for (name, basic, pin_name) in [ (PROG_CONNECT4, PROG_CONNECT4_BASIC, "connect4"), (PROG_CONNECT6, PROG_CONNECT6_BASIC, "connect6"), @@ -1723,7 +1099,6 @@ pub(super) fn attach(bpf: &mut Ebpf, dir: Option<&Path>) -> anyhow::Result) -> anyhow::Result i, @@ -1753,63 +1128,7 @@ pub(super) fn attach(bpf: &mut Ebpf, dir: Option<&Path>) -> anyhow::Result out.push((name.to_string(), i)), - Err(e) => { - warn!( - "{name} could not load or attach ({e:#}); the fast path runs without the sendmsg hooks - a mark forged onto an unconnected UDP socket is not stripped on this kernel" - ); - fast_path = FastPathCapability::SendmsgUnavailable; - break; - } - } - } - if fast_path != FastPathCapability::Ready { - // Leave no lone sendmsg pin behind for the next start to trip on. - if let Some(d) = dir { - for p in [LINK_SENDMSG4, LINK_SENDMSG6] { - let _ = std::fs::remove_file(d.join(p)); - } - } - } - } - Ok(AttachedPrograms { - programs: out, - fast_path, - }) + Ok(out) } /// Removes entries for pids that no longer exist. @@ -1863,11 +1182,8 @@ pub(super) fn stats(map: &PerCpuArray<&MapData, u64>) -> anyhow::Result anyhow::Result<()> { Ok(()) } -/// Arms the kernel side of the fast path and returns the mark it will set. -/// -/// The nftables side is not here: the table is normally not loaded yet when -/// the daemon starts (`colony-firewall-nft.service` waits for the daemon), -/// so the set is written by the heartbeat task, which retries until it can - -/// see `nft_arm_state`. A mark in the map with no element in the set is a -/// wasted `setsockopt`, never a bypass: the ruleset accepts a value only while -/// it is in the set, and the caller flushed the set unconditionally before -/// this ran. -/// -/// The deadline does stay zero until the heartbeat's first beat - but because -/// `VerdictSink::arm` zeroes it, not by itself. `FAST_ALLOW_UNTIL` is a pinned -/// map, so after an unclean death it holds whatever the previous daemon last -/// wrote, up to a minute into the future; this comment asserted the zero for -/// two releases before anything wrote it. -/// -/// The mark is 32 random bits, never zero (`UNARMED`). Random per start -/// because since kernel 5.17 `SO_MARK` needs only `CAP_NET_RAW`, which docker -/// grants by default: a value anyone could read out of a package would be a -/// bypass token for any such process on the host's network. -fn arm_kernel_side(sink: &enforce::VerdictSink, configured: Option) -> anyhow::Result { - let mark = match configured { - Some(m) if m == cfc_ebpf_common::fast_allow::UNARMED => { - return Err(anyhow!( - "[ebpf] fast_allow_mark = 0 is not a mark: zero is what every socket \ - nothing has marked carries, and accepting it would accept everything" - )) - } - Some(m) => { - if let Some(who) = collides_with(m) { - // Their machine, their call - but not silently. - warn!( - "the configured fast-allow mark 0x{m:08x} is one {who} selects on; \ - traffic this daemon marks may be routed or dropped by that rule" - ); - } - m - } - None => pick_mark(|| uuid::Uuid::new_v4().as_u128() as u32).ok_or_else(|| { - anyhow!( - "could not draw a fast-allow mark that avoids the fwmark selectors this \ - host is likely to use; set [ebpf] fast_allow_mark to choose one by hand" - ) - })?, - }; - sink.arm(mark) - .context("writing the fast-allow mark to the kernel")?; - Ok(mark) -} - -/// fwmark selectors this machine is likely to already have, as -/// (mask, value, who) - a candidate `m` collides when `m & mask == value`. -/// -/// The mark is one 32-bit word shared by everything on the host, and the -/// dangerous consumers are the ones that select on a *mask*: they do not need -/// to guess our value, only to share a bit with it. A uniformly random word -/// therefore collides at a rate the other rule's mask decides, freshly at every -/// daemon start, which turns this into an intermittent and very hard to -/// attribute network fault - the fast path is off by default, so the operator's -/// first evidence is that turning it on breaks their VPN one boot in N. -/// -/// The two kube-proxy entries are why this list is not optional. Their masks -/// are a single bit, so a random word matches one of them **half the time**, -/// and `0x8000/0x8000` is the mark kube-proxy attaches to packets it then -/// DROPs. On such a node the previous code broke every fast-allowed flow on -/// roughly every other daemon start. -/// -/// This list is not, and cannot be, complete: nothing enumerates the fwmark -/// users of a Linux host. It is the documented ones, and `[ebpf] -/// fast_allow_mark` is the answer for a machine with a selector it misses. -const KNOWN_SELECTORS: &[(u32, u32, &str)] = &[ - // kube-proxy: masquerade, and drop. - (0x0000_4000, 0x0000_4000, "kube-proxy (masquerade)"), - (0x0000_8000, 0x0000_8000, "kube-proxy (drop)"), - // Tailscale's ip rules: "came from tailscale0", and "bypass tailscale". - (0x00ff_0000, 0x0008_0000, "Tailscale"), - (0x00ff_0000, 0x0004_0000, "Tailscale (bypass)"), - // wg-quick's `ip rule not fwmark lookup `: an exact-word - // compare, so this one costs a single value out of four billion. Listed - // because excluding it is free and the failure - the tunnel's own table - // stops being consulted for our traffic - is silent. - (0xffff_ffff, 0x0000_ca6c, "wg-quick"), -]; - -/// The first entry of [`KNOWN_SELECTORS`] that would match `mark`. -fn collides_with(mark: u32) -> Option<&'static str> { - KNOWN_SELECTORS - .iter() - .find(|(mask, value, _)| mark & mask == *value) - .map(|(_, _, who)| *who) -} - -/// Draws a mark that is neither `UNARMED` nor something in -/// [`KNOWN_SELECTORS`]. -/// -/// Rejection sampling rather than a claimed range, because a range is the -/// thing that must not be predictable: `SO_MARK` needs only CAP_NET_RAW since -/// 5.17, so a value an attacker can enumerate is a bypass token. Roughly a -/// quarter of the word survives the sieve - the two single-bit kube-proxy -/// masks account for almost all of it - which leaves about thirty bits of -/// entropy and takes four draws on average. -fn pick_mark(mut draw: impl FnMut() -> u32) -> Option { - // Bounded so a caller whose `draw` is degenerate cannot hang the daemon. - // About a quarter of the word survives the sieve, so 64 consecutive - // rejections is not chance - it is a broken source of randomness. - for _ in 0..64 { - let candidate = draw(); - if candidate != cfc_ebpf_common::fast_allow::UNARMED && collides_with(candidate).is_none() { - return Some(candidate); - } - } - // And then nothing, rather than a fallback. - // - // The obvious fallback - walk upward from 1 until the sieve passes - was - // worse than no fast path at all: it does not depend on `draw`, so it is - // the *same* value on every machine that reaches it. A published constant - // is precisely the bypass token the random draw exists to avoid, and - // `SO_MARK` needs only CAP_NET_RAW since 5.17. The path stays off, with - // the reason in `cfc status`, and every connection keeps taking the queue - - // which is the behaviour this whole feature degrades to anyway. - None -} - -/// The (deadline, heartbeat) pair a daemon should use, in seconds. -/// -/// With every guarantee in place the deadline is a backstop and the full pair -/// applies. With any reduced - see [`fast_path_decision`] - it is ten times -/// shorter and refreshed five times as often, and where exit detection is -/// imprecise that same beat is also the cadence of the stale-grant sweep. -fn deadline_pair(reduced: bool) -> (u64, u64) { - use cfc_ebpf_common::fast_allow as fa; - if reduced { - (fa::DEADLINE_SECS_REDUCED, fa::HEARTBEAT_SECS_REDUCED) - } else { - (fa::DEADLINE_SECS, fa::HEARTBEAT_SECS) - } -} - -/// What the eligibility decision is made from, so the decision can be a pure -/// function with a test rather than a ladder of `else if` that only a live -/// kernel exercises. -#[derive(Debug, Clone, Copy)] -struct LadderFacts { - config_on: bool, - has_maps: bool, - exit_tracking: bool, - exit_precise: bool, - lifecycle_pinned: bool, - capability: enforce::FastPathCapability, -} - -/// Legacy status rendering remains readable for older clients. -#[cfg(test)] -const REDUCED_IMPRECISE_EXIT: &str = "exit is detected by thread-group leader only"; - -/// Legacy capability checks remain conservative, but runtime grants are disabled -/// for every configuration: a socket mark does not identify its current sender. -const FAST_ALLOW_DISABLED: &str = - "Fast Allow is disabled: socket marks cannot verify the current sender; use normal NFQUEUE filtering"; - -fn fast_path_decision(f: &LadderFacts) -> Result, &'static str> { - if !f.config_on { - return Err("[ebpf] fast_allow is not set"); - } - if !f.has_maps { - return Err("the loaded object has no fast-allow maps"); - } - if !f.exit_tracking || !f.exit_precise || !f.lifecycle_pinned { - return Err("fast-allow requires precise, pinned exec/exit hooks"); - } - if let Some(why) = f.capability.refusal() { - return Err(why); - } - Err(FAST_ALLOW_DISABLED) -} - -/// One attempt at the nftables side, at startup, reported as the state it -/// leaves the path in. On a daemon *restart* the table is already loaded and -/// this comes back `Live` at once; on a boot it comes back waiting, and the -/// heartbeat task finishes the job. -fn nft_arm_state(mark: u32, deadline_secs: u64, reduced: Option) -> FastAllow { - match super::nft_set::arm(mark) { - Ok(()) => FastAllow::Live { - deadline_secs, - reduced, - }, - Err(e) => nft_arm_state_from_error(&e), - } -} - -/// The reported state for a failed nftables arm. A missing *table* is the -/// ruleset unavailable and reads as waiting; a missing *set* is an operator-visible fact - a snippet that -/// predates the feature - and carries the fix; anything else is quoted. -fn nft_arm_state_from_error(e: &anyhow::Error) -> FastAllow { - match e.downcast_ref::() { - Some(super::nft_set::Absent::Table) => FastAllow::Off( - "waiting for the nftables table (colony-firewall-nft.service has not installed filtering)" - .to_string(), - ), - Some(super::nft_set::Absent::Set) => FastAllow::Off(format!("{e}")), - None => FastAllow::Off(format!("could not arm nftables: {e:#}")), - } -} - /// Classifies a failure from `EbpfLoader::load` - parsing the ELF, creating /// maps, applying relocations. /// @@ -401,18 +196,6 @@ impl Drop for Attached { for t in &self.tasks { t.abort(); } - // A clean stop disarms now rather than letting the deadline lapse: - // zero deadline and unarmed mark in the kernel, the set flushed in - // nftables. Best effort - a daemon on its way out has only the log. - if let Some(sink) = &self._sink { - sink.disarm(); - if let Err(e) = super::nft_set::disarm_for_shutdown() { - tracing::warn!( - "could not flush the fast-allow mark from nftables on shutdown: {e:#}" - ); - } - super::set_fast_allow_level(FastAllow::Off("the daemon stopped".to_string())); - } } } @@ -422,32 +205,12 @@ impl Drop for Attached { /// missing file, a malformed ELF, a kernel that refuses the whole program set. /// Individual attach failures are recorded in the [`Report`] and leave the /// rest running. -// Eight injected dependencies, each a different thing the layer may read or -// feed and none of which it should own; a bag struct to satisfy the lint -/// The `[ebpf]` fast-path settings one load should honour. -/// -/// Two fields rather than two parameters: the argument list is already at the -/// lint's limit, and these two are one decision - whether the fast path runs, -/// and with which mark - taken from one config section. -#[derive(Clone, Copy, Debug, Default)] -pub(super) struct FastAllowCfg { - /// `[ebpf] fast_allow`. - pub on: bool, - /// `[ebpf] fast_allow_mark`, when the operator pinned one. `None` draws. - pub mark: Option, -} - -// would name nothing that the parameter list does not already name. -#[allow(clippy::too_many_arguments)] pub(super) fn load_and_attach( object_path: &Path, dns: DnsCache, table: KernelProcTable, engine: Option, trust: Trust, - observed: tokio::sync::broadcast::Sender, - stats: crate::stats::Stats, - fast_allow: FastAllowCfg, ) -> Result<(Attached, Report), LoadError> { let mut report = Report { mode: crate::config::EbpfMode::On, @@ -455,36 +218,6 @@ pub(super) fn load_and_attach( ..Report::default() }; - // Whatever a previous daemon left accepted in the nftables set, drop it - - // first, before anything can return. - // - // This lived further down for a while, next to the code that arms, and - // that was wrong twice over. It ran only when this daemon was *eligible* - // and armed; and even after being made unconditional it still sat behind - // every early return in this function - a missing object (which is the - // single most common outcome on a default install), an untrusted one, a - // failed load. So a daemon that crashed while armed and came back to any - // of those left its predecessor's mark sitting in the set: accepted by the - // ruleset, refreshed by nobody, removed by nothing short of the table - // going away. That is a standing bypass token rather than a stale entry - - // every process that was ever fast-allowed can read the value back off its - // own socket with `getsockopt(SO_MARK)`, and setting it again needs only - // CAP_NET_RAW. - // - // Flushing before knowing whether this daemon will arm is the right order: - // an empty set accepts nothing, which is the safe state to pass through. - if let Err(e) = super::nft_set::disarm_for_start() { - // Logged as well as noted, because the note alone reaches nobody on - // the paths that matter most: every early return below builds a - // `LoadError` and drops this `Report` on the floor, and a failed flush - // followed by a failed load is exactly the shape that leaves a - // predecessor's mark accepted with no daemon to explain it. - tracing::error!("could not disable previous Fast Allow state: {e:#}; old marks may still bypass filtering; run systemctl reload colony-firewall-nft and inspect the journal before relying on filtering"); - report.notes.push(format!( - "could not disable previous Fast Allow state: {e:#}; old marks may still bypass filtering; reload colony-firewall-nft before relying on filtering" - )); - } - // Vet before read, so a file we would refuse is never even pulled into // memory, and so the "not there at all" case is distinguishable from the // "there but not ours" one. @@ -595,11 +328,10 @@ pub(super) fn load_and_attach( // Attribution rather than enforcement, but the same restart // split applies; see the SOCK_PIDS paragraph below. (enforce::MAP_SOCK_PIDS, dir.join(enforce::MAP_SOCK_PIDS)), - // The fast path's four. Pinned for the same restart reason - // as everything above, and it is the pinning that makes the - // deadline load-bearing: the programs keep these alive after - // the daemon dies, so only `FAST_ALLOW_UNTIL` running out - // stops the marks. + // The legacy Fast Allow maps. Nothing grants any more, but + // the kernel object still reads them, and they must be the + // pinned ones or the startup disarm writes a fresh map that + // no inherited program sees. (enforce::MAP_FAST_ALLOW, dir.join(enforce::MAP_FAST_ALLOW)), ( enforce::MAP_FAST_ALLOW_UNTIL, @@ -859,6 +591,12 @@ pub(super) fn load_and_attach( }) .map_err(|e| LoadError::new(classify_load(&e), e))?; + // Fast Allow is gone from the daemon, but the kernel object still carries + // its maps (ABI v4) and they are pinned, so a 0.4-0.6 daemon that died + // while armed left a mark the connect hooks would go on setting, past any + // restart. Disarm before anything attaches, whatever else comes up. + enforce::disarm_legacy_fast_allow(&mut bpf); + // --- attach, each independently ------------------------------------ let exec_pin = pin_dir.as_deref().map(|d| d.join(enforce::LINK_EXEC)); @@ -893,35 +631,15 @@ pub(super) fn load_and_attach( &mut exit_pinned, ); report.exit_tracking = record_attach(&mut report, PROG_EXIT, "sched_process_exit", r); - // Both clears have to survive this daemon for a grant to be safe past its - // death, so this is one flag, not two. + // Both clears have to survive this daemon for its denials to stay + // honest past its death, so this is one flag, not two. report.lifecycle_pinned = exec_pinned && exit_pinned; let r = attach_dns(&mut bpf); report.dns_capture = record_attach(&mut report, PROG_DNS, "cgroup_skb/ingress", r); // --- enforcement ---------------------------------------------------- - // Whether the kernel side of the fast path is in place. On the inherited - // path the pin names do not say which connect variant is running; the - // previous daemon's cookie marker does (see `enforce::COOKIE_MARKER`), and - // the sendmsg pins say whether that daemon had those hooks too. - let mut fast_path = enforce::FastPathCapability::BasicConnect; report.enforcement = if inherited { - fast_path = match pin_dir.as_deref() { - Some(d) if enforce::cookie_variants_pinned(d) => { - if enforce::sendmsg_pinned(d) { - enforce::FastPathCapability::Ready - } else { - enforce::FastPathCapability::SendmsgUnavailable - } - } - // No marker: a basic-connect daemon, or one that could not create - // the marker and said so. Either way nothing here is known to mark, - // so the packet path decides - fail closed, and the reason names - // the marker so the fix (a restart with the pins removed) is - // legible. - _ => enforce::FastPathCapability::BasicConnect, - }; report.notes.push(format!( "in-kernel enforcement was already attached and pinned at {}; \ steering it rather than replacing it", @@ -930,9 +648,8 @@ pub(super) fn load_and_attach( Enforcement::Inherited } else { match enforce::attach(&mut bpf, pin_dir.as_deref()) { - Ok(attached) => { - fast_path = attached.fast_path; - for (name, insns) in attached.programs { + Ok(programs) => { + for (name, insns) in programs { if let Some(n) = insns { report.verified_insns.push((name, n)); } @@ -953,13 +670,6 @@ pub(super) fn load_and_attach( } }; - // A fact of the eligibility ladder, recorded whether or not the ladder - // runs below: a layer handed no engine still reports what the kernel let - // it have, which is what the kernel matrix reads. `None` where nothing - // attached - `fast_path` still says BasicConnect then, and that would be - // a claim about hooks that do not exist. - report.fast_path_capability = report.enforcement.is_live().then_some(fast_path); - // The socket-cookie -> pid map, for O(1) attribution - the pinned one // when there is a pin directory, which is what lets an inherited connect // program's writes land somewhere this daemon can read. Taken whenever it @@ -1009,26 +719,15 @@ pub(super) fn load_and_attach( .and_then(|m| aya::maps::PerCpuArray::<_, u64>::try_from(m).map_err(Into::into)) .and_then(|m| enforce::stats(&m)) { - // Every counter, in the guard and in the message. Two of them - // used to be read and then left out of both, so the one state an - // operator most wants named - a foreign mark keeping the fast path - // permanently disengaged for some program - could not be reached - // from the note at all. - Ok(s) - if s.denied > 0 - || s.allowed > 0 - || s.unknown > 0 - || s.stale > 0 - || s.foreign_mark > 0 - || s.report_dropped > 0 => - { + // The counters the daemon still acts on. The kernel keeps + // counting the legacy Fast Allow slots too, and the stat layout + // stays as it is, but those are only ever zero now. + Ok(s) if s.denied > 0 || s.unknown > 0 || s.report_dropped > 0 => { report.notes.push(format!( "in-kernel enforcement carried over: {} connect() refused, \ - {} fast-allowed, {} passed to the packet path, {} grants \ - ignored as stale, {} sockets left alone for carrying \ - another marker's mark, {} decisions the report ring could \ - not hold, since the pins were made", - s.denied, s.allowed, s.unknown, s.stale, s.foreign_mark, s.report_dropped + {} passed to the packet path, {} decisions the report ring \ + could not hold, since the pins were made", + s.denied, s.unknown, s.report_dropped )) } Ok(_) => {} @@ -1066,121 +765,19 @@ pub(super) fn load_and_attach( // enforcement did not come up, or when the caller has no rule engine to // consult (the live tests); in both cases the map simply stays empty and // every connect falls through to the packet path. - // The mark the kernel side was armed with, when it was: the heartbeat - // task below needs it to finish the nftables half of arming. Alongside it, - // the deadline pair that task must write and pace itself by - the full one, - // or the shortened one for a kernel whose lifecycle links could not be - // pinned. Both are decided inside the block below and used after it. - let mut armed_mark: Option = None; - let mut deadline_secs = cfc_ebpf_common::fast_allow::DEADLINE_SECS; - let mut heartbeat_secs = cfc_ebpf_common::fast_allow::HEARTBEAT_SECS; - // And the reason the guarantee is weaker, if it is, for every `Live` the - // heartbeat will ever publish. Hoisted rather than read back out of - // `report.fast_allow` at spawn time, because on a boot that field is - // `Off("waiting for the nftables table")` when the task starts - the - // common case, not an edge - and deriving from it lost the reason on - // every first arm. - let mut reduced_because: Option = None; let sink = match (report.enforcement.is_live(), engine) { (true, Some(engine)) => { match enforce::VerdictSink::new(&mut bpf, engine.clone(), table.clone()) { - Ok(mut sink) => { - // Whatever the previous daemon granted, it granted under - // its rules. The map is pinned, so those grants are still - // here; nothing below writes a grant until this is done. - let dropped = sink.flush_fast_allow(); - if dropped > 0 { - debug!( - "dropped {dropped} fast-allow grants inherited from a previous daemon" - ); - } - - // The decision is a pure function of the facts, so it has a - // test; this is only the gathering. `enforcement` is not - // among the facts - see `fast_path_decision` for why the - // Process-mode refusal was backwards. - let facts = LadderFacts { - config_on: fast_allow.on, - has_maps: sink.has_fast_path(), - exit_tracking: report.exit_tracking, - exit_precise: report.exit_precise, - lifecycle_pinned: report.lifecycle_pinned, - capability: fast_path, - }; - let (off, reduced): (Option<&str>, Vec<&str>) = match fast_path_decision(&facts) - { - Ok(reduced) => (None, reduced), - Err(why) => (Some(why), Vec::new()), - }; - - // Weaker guarantees are said, once each, in the log and the - // report, and carried in the status so `live` never hides - // them. Both select the short deadline pair; where exit - // detection is imprecise the heartbeat also sweeps grants, - // which is what makes that reduction boundable at all. - (deadline_secs, heartbeat_secs) = deadline_pair(!reduced.is_empty()); - reduced_because = if off.is_none() && !reduced.is_empty() { - for why in &reduced { - let note = format!( - "fast-allow runs with a weaker guarantee: {why}; grants lapse within {deadline_secs}s instead of {}s", - cfc_ebpf_common::fast_allow::DEADLINE_SECS - ); - warn!("{note}"); - report.notes.push(note); - } - Some(reduced.join("; ")) - } else { - None - }; - if off.is_none() { - if let Some(caveat) = fast_path.caveat() { - warn!("fast-allow: {caveat}"); - report.notes.push(format!("fast-allow: {caveat}")); - } - } - - let (state, mark_opt) = match off { - Some(why) => { - if fast_allow.on { - warn!("{FAST_ALLOW_DISABLED}"); - report.notes.push(FAST_ALLOW_DISABLED.to_string()); - } - sink.withdraw_fast_path(); - (FastAllow::Off(why.to_string()), None) - } - None => match arm_kernel_side(&sink, fast_allow.mark) { - Ok(mark) => ( - nft_arm_state(mark, deadline_secs, reduced_because.clone()), - Some(mark), - ), - Err(e) => { - sink.withdraw_fast_path(); - (FastAllow::Off(format!("could not arm: {e:#}")), None) - } - }, - }; - report.fast_allow = Some(state); - armed_mark = mark_opt; - + Ok(sink) => { let sink = std::sync::Arc::new(sink); // resync rather than compile_rules alone: on the inherited // path the pinned map holds the previous daemon's // verdicts, made under the previous daemon's rules, and // this is the reconciliation that makes them this - // daemon's. - // - // Which of its parts does that work is worth being exact - // about, because a comment here once claimed the orphan - // sweep did all of it and that was only half true. The - // proc table is empty at this point - `set_live` has not - // run yet - so the live loop no-ops, and the orphan sweep - // reconciles the *denials* the previous daemon left in - // `VERDICTS`. It cannot reconcile grants: `flush_fast_allow` - // above has just emptied the map the sweep would walk, on - // purpose. Re-seeding the grants is `sweep_fast_allow`'s - // job, at the end of `resync`, and it walks /proc rather - // than any map - which is the only way to reach a process - // that was already running when this daemon started. + // daemon's. The proc table is empty at this point - + // `set_live` has not run yet - so the live loop no-ops, + // and it is the orphan sweep that reconciles the denials + // the previous daemon left in `VERDICTS`. sink.resync(); let weak = std::sync::Arc::downgrade(&sink); engine.set_on_change(Box::new(move || { @@ -1200,32 +797,6 @@ pub(super) fn load_and_attach( } _ => None, }; - if report.fast_allow.is_none() { - report.fast_allow = Some(FastAllow::Off( - if report.enforcement.is_live() { - "no decision engine was handed to the layer" - } else { - "in-kernel enforcement is not live" - } - .to_string(), - )); - } - - // Published here, and only here, because after this point the heartbeat - // task below is running and publishing states of its own. - // - // `start` used to do it, once `load_and_attach` returned - which is *after* - // that task exists. On a daemon restart the table is already loaded, so the - // very first thing the heartbeat does is arm and publish `Live`; `start` - // then overwrote it with the state decided here, "waiting for the nftables - // table". And nothing ever corrected it: the heartbeat only publishes while - // it is not armed. `cfc status` said the fast path was waiting for a table - // that had been there all along, for the life of the daemon, while the path - // was in fact live. - if let Some(state) = report.fast_allow.clone() { - super::set_fast_allow_level(state); - } - // Exec without exit tracking would let entries age out on the TTL alone, // which is a materially weaker pid-reuse story. Refuse the combination // rather than quietly serving it. @@ -1303,35 +874,6 @@ pub(super) fn load_and_attach( } } - // The eligibility ladder ran before any of the consumers above existed, - // and two of them can retract what it assumed: a ring consumer that fails - // to start turns `exec_tracking` off, and the exit one turns both off. So - // the fast path could be armed, reported `live`, and marking sockets while - // `on_exec` - its only per-execve writer - could never run, and while - // nothing on the daemon side evicted a grant. - // - // Correct it here rather than moving the decision, because the decision - // needs the sink and the sink is what these consumers borrow. Flush what - // was granted, empty the set so the ruleset accepts nothing, and leave - // `armed_mark` unset so the heartbeat task below is never spawned - with - // no heartbeat the kernel stops honouring grants within one deadline, and - // with no element in the set it stops mattering immediately. - if armed_mark.is_some() && !(report.exec_tracking && report.exit_tracking) { - let why = "the exec/exit ring consumers did not start, so nothing would grant or evict"; - warn!("fast-allow withdrawn after arming: {why}"); - if let Err(e) = super::nft_set::disarm() { - report - .notes - .push(format!("could not flush the fast-allow set: {e:#}")); - } - if let Some(sink) = sink.as_ref() { - sink.flush_fast_allow(); - } - armed_mark = None; - report.fast_allow = Some(FastAllow::Off(why.to_string())); - super::set_fast_allow_level(FastAllow::Off(why.to_string())); - } - // Denials refused in the kernel never reach NFQUEUE, so this consumer is // the only thing standing between "CFC blocked it" and "the connection just // failed". It is a log line rather than a prompt on purpose: the user @@ -1363,219 +905,6 @@ pub(super) fn load_and_attach( } } - // The fast path's two tasks, only while it is live: the heartbeat that - // keeps the kernel honouring grants, and the consumer that keeps the - // rest of the daemon honest about flows the packet path never sees. - if let (Some(mark), Some(sink)) = (armed_mark, sink.as_ref()) { - // Heartbeat, and the nftables side of arming. The two are one task on - // purpose: `colony-firewall-nft.service` is After= this daemon and - // waits for it to be active, so at daemon start the table is not - // loaded yet and the set cannot be written - on every boot, not as - // an edge case. The kernel side is armed (mark written) but the - // deadline stays zero, so nothing is honoured, until the element is - // in the set; then every tick refreshes the deadline. If this task - // ever stops - abort on shutdown, a wedged runtime, the daemon dying - // - the kernel stops honouring grants within one deadline. That is - // the design, not a failure mode. - let beat = sink.clone(); - let mut armed = matches!(report.fast_allow, Some(FastAllow::Live { .. })); - // What the status must keep saying every time this task re-arms, and - // whether each beat also sweeps grants. - let reduced_for_status = reduced_because.clone(); - let sweep_grants = !report.exit_precise; - tasks.push(tokio::spawn(async move { - let mut tick = tokio::time::interval(std::time::Duration::from_secs(heartbeat_secs)); - let mut last_reason: Option = None; - // Ticks between two checks that the set still holds the mark. - // Expressed in ticks, so it has to follow the tick length: one - // minute either way, whichever deadline pair is in force. - // Clamped at both ends: a divisor of zero would panic, and a - // heartbeat longer than the check period would make this zero, - // which reads as "check on every tick" - a fork and exec every - // beat, forever. - let checks_every: u32 = ((60 / heartbeat_secs.max(1)) as u32).max(1); - let mut since_check: u32 = 0; - loop { - tick.tick().await; - if !armed { - // Off the async threads: this execs nft and waits on it. - let attempt = tokio::task::spawn_blocking(move || super::nft_set::arm(mark)) - .await - .unwrap_or_else(|e| Err(anyhow!("arming task failed: {e}"))); - let state = match attempt { - Ok(()) => { - armed = true; - tracing::info!( - "fast-allow armed: the nftables set now holds this daemon's mark" - ); - FastAllow::Live { - deadline_secs, - reduced: reduced_for_status.clone(), - } - } - Err(e) => nft_arm_state_from_error(&e), - }; - // Say each reason once, not on every beat. - let reason = state.describe(); - if last_reason.as_deref() != Some(reason.as_str()) { - if !armed { - tracing::info!("fast-allow {reason}"); - } - last_reason = Some(reason); - } - super::set_fast_allow_level(state); - if !armed { - continue; - } - } - if let Err(e) = beat.beat(deadline_secs) { - // The number, not "a minute": on a kernel whose lifecycle - // links could not be pinned this deadline is six seconds, - // and a warning that names the wrong one sends a reader - // looking for a window that closed long ago. - tracing::warn!( - "fast-allow heartbeat failed: {e:#}; grants lapse within {deadline_secs}s" - ); - } - - // On a kernel without `group_dead` the exit program evicts on - // leader exit only, so a process whose leader exits first and - // dies later keeps its grant while this daemon lives and - // refreshes the deadline. This is what bounds that: every - // beat, every granted pid is re-dated against the start time - // it was granted with. Off the async threads - it reads /proc - // once per granted pid, and granted pids are the few a lasting - // rule allows outright. - if sweep_grants { - let sweeper = beat.clone(); - match tokio::task::spawn_blocking(move || sweeper.sweep_stale_grants()).await { - Ok(0) => {} - Ok(n) => tracing::debug!(dropped = n, "swept stale fast-allow grants"), - // A panic inside the sweep; the grants it did not reach - // are re-checked next beat, so this is worth a line and - // nothing more. - Err(e) => tracing::debug!("fast-allow grant sweep did not run: {e}"), - } - } - - // Armed is not a fact that stays true, and this loop used to - // treat it as one: once the element went in, the only thing it - // ever did again was refresh the deadline. `systemctl restart - // nftables`, or any `nft -f` that reloads the machine's - // ruleset, recreates `table inet colony_firewall` with an - // empty set - and the daemon went on marking sockets, went on - // crediting rule hits from ALLOW_EVENTS, and went on telling - // `cfc status` that the fast path was live, while every one of - // those flows was in fact taking the queue and being counted - // a second time by the packet path. - // - // Checked once a minute rather than every tick: this is a - // fork and exec, the window it leaves is a minute of an - // over-optimistic status line, and nothing unsafe happens in - // it - the failure is the set accepting *less* than the daemon - // thinks, never more. A minute either way, so a shortened - // heartbeat does not turn this into a fork every two seconds. - since_check += 1; - if since_check >= checks_every { - since_check = 0; - match tokio::task::spawn_blocking(move || super::nft_set::holds(mark)).await { - Ok(Ok(true)) => {} - Ok(Ok(false)) => { - armed = false; - last_reason = None; - tracing::warn!( - "the fast-allow mark is no longer in the nftables set (the \ - ruleset was reloaded); re-arming" - ); - super::set_fast_allow_level(FastAllow::Off( - "the nftables set no longer holds this daemon's mark; re-arming" - .to_string(), - )); - } - // Could not ask. Say nothing and keep the current - // state: a failed probe is not evidence either way, - // and disarming on it would take the path down on a - // transient. - Ok(Err(e)) => tracing::debug!("could not check the fast-allow set: {e:#}"), - Err(e) => tracing::debug!("fast-allow set check did not run: {e}"), - } - } - } - })); - - // ALLOW_EVENTS: one record per flow the kernel waved past the queue. - // Credited to the rule that granted, counted where NFQUEUE counts, - // and fed to the same observed stream - so the busiest allow rule - // does not read as dead and the enforcing heuristic does not flip - // to "not enforcing" while the firewall is doing its job. - let credit = sink.clone(); - let engine_hits = sink.engine().clone(); - let observed_tx = observed.clone(); - let stats_tx = stats.clone(); - // The same reverse-DNS seam the packet path uses. Without it a - // fast-allowed flow is the one kind of connection whose destination - // never gets a name: the packet path attaches whatever is cached and - // enqueues a lookup for next time, and this consumer - which exists - // precisely because these flows never reach that path - did neither. - // So `cfc log` and the live feed showed bare addresses for exactly the - // programs a user had trusted enough to allow outright, and the cache - // was never warmed for their destinations either, so the *next* flow - // to the same host had no name to attach. - let dns_hosts = dns.clone(); - match spawn_ring(&mut bpf, enforce::MAP_ALLOW_EVENTS, move |bytes| { - let Some(ev) = decode::(bytes) else { - return; - }; - stats_tx.record_allow(); - let verdict = match credit.granted_by(ev.pid) { - Some(rule) => { - engine_hits.record_hit(rule); - cfc_core::Verdict::from_rule(cfc_core::Action::Allow, rule) - } - // A grant this daemon did not make (a previous one's, in the - // window before the startup flush). Reported, credited to no - // rule rather than to the wrong one. - None => cfc_core::Verdict::from_policy(cfc_core::Action::Allow), - }; - let protocol = match ev.protocol { - 6 => cfc_core::Protocol::Tcp, - 17 => cfc_core::Protocol::Udp, - other => cfc_core::Protocol::Other(other), - }; - let unspecified = if ev.family == 4 { - std::net::IpAddr::V4(std::net::Ipv4Addr::UNSPECIFIED) - } else { - std::net::IpAddr::V6(std::net::Ipv6Addr::UNSPECIFIED) - }; - let dst = ev.destination(); - let mut connection = cfc_core::Connection::new( - protocol, - cfc_core::Direction::Outbound, - unspecified, - 0, - dst.ip(), - dst.port(), - ); - if let Some((host, verified)) = dns_hosts.cached_named(dst.ip()) { - connection = connection.with_host_verified(host, verified); - } - dns_hosts.enqueue_lookup(dst.ip()); - let process = crate::process_resolve::resolve(ev.pid); - let _ = observed_tx.send(crate::nfqueue::ObservedConnection { - connection, - process, - verdict, - }); - }) { - Ok(task) => tasks.push(task), - Err(e) => report.notes.push(format!( - "{} consumer not started: {e:#}; fast-allowed flows will be \ - unreported and their rules uncredited", - enforce::MAP_ALLOW_EVENTS - )), - } - } - if report.dns_capture { let cache = dns.clone(); // One scratch answer for the life of the consumer. `for_each_answer` @@ -1709,14 +1038,10 @@ fn attach_tracepoint( // evicting after the daemon dies" property, which is what best-effort was // meant to mean. // - // `pinned_out` carries that outcome to the caller, because one feature does - // depend on it. The fast path's eligibility ladder asks for `Pinned` - // enforcement and exit tracking, on the reasoning that a grant is always - // cleared even if the daemon dies - and that reasoning is the *pin's*, not - // the attach's. On a kernel with no BPF_LINK_TYPE_PERF_EVENT the connect - // programs stay pinned and go on marking sockets while the exec and exit - // clears die with the daemon, which the ladder could not see because this - // function used to return the same `Ok` either way. + // `pinned_out` carries that outcome to the caller, which reports it as + // `lifecycle_pinned`: on a kernel with no BPF_LINK_TYPE_PERF_EVENT the + // connect programs stay pinned while the exec and exit clears die with + // the daemon, and this function used to return the same `Ok` either way. let pinned = prog .take_link(id) .map_err(anyhow::Error::new) @@ -1949,18 +1274,6 @@ mod tests { assert!(process_group_is_gone(u32::MAX)); } - #[test] - fn fast_allow_is_refused_even_with_every_hook_available() { - let facts = LadderFacts { - config_on: true, - has_maps: true, - exit_tracking: true, - exit_precise: true, - lifecycle_pinned: true, - capability: enforce::FastPathCapability::Ready, - }; - assert!(fast_path_decision(&facts).is_err()); - } use std::net::Ipv4Addr; use std::os::unix::fs::PermissionsExt as _; @@ -2025,150 +1338,6 @@ mod tests { assert!(!dir_is_safe(0, 0o040775), "group-writable counts too"); } - /// The regression this sieve exists for: kube-proxy selects on - /// `0x8000/0x8000` and DROPs what matches. A uniformly random word has - /// that bit set half the time, so on a Kubernetes node the previous - /// draw broke every fast-allowed flow on roughly every other start. - /// The property that makes a late tick harmless: several beats fit inside - /// one deadline, for *both* pairs. A ratio of one would mean a single - /// delayed heartbeat lapses the fast path on a perfectly healthy daemon. - #[test] - fn several_beats_fit_inside_every_deadline() { - for reduced in [false, true] { - let (deadline, heartbeat) = deadline_pair(reduced); - assert!(heartbeat > 0, "a zero heartbeat would spin"); - assert!( - deadline >= heartbeat * 3, - "reduced={reduced}: {deadline}s deadline against a {heartbeat}s beat leaves \ - no room for a late tick" - ); - } - } - - /// The unpinned pair exists to shrink the window a dead daemon leaves, so - /// it has to actually be shorter - and the pinned one has to be the value - /// every document quotes. - #[test] - fn the_reduced_deadline_is_the_shorter_one() { - let (full, _) = deadline_pair(false); - let (reduced, beat) = deadline_pair(true); - assert_eq!(full, cfc_ebpf_common::fast_allow::DEADLINE_SECS); - assert!( - reduced < full, - "a reduced guarantee must not get the longer deadline: {reduced} vs {full}" - ); - // The heartbeat must speed up with it, or the ratio above breaks. - assert!(beat < cfc_ebpf_common::fast_allow::HEARTBEAT_SECS); - } - - /// The policy, over the cases that decide it. Refusals are where nothing - /// could mark or nothing could evict; everything weaker but boundable - /// reduces; a missing sendmsg hook is a note. `Enforcement` is not an - /// input at all, which is itself the assertion. - #[test] - fn the_ladder_never_arms_socket_mark_authorization() { - use enforce::FastPathCapability as Cap; - for config_on in [false, true] { - for capability in [Cap::Ready, Cap::SendmsgUnavailable, Cap::BasicConnect] { - for lifecycle_pinned in [false, true] { - for exit_precise in [false, true] { - assert!(fast_path_decision(&LadderFacts { - config_on, - has_maps: true, - exit_tracking: true, - exit_precise, - lifecycle_pinned, - capability, - }) - .is_err()); - } - } - } - } - } - - /// `cfc status` must not say plain `live` on a kernel where the guarantee - /// is weaker - that is the whole reason the deadline is carried. - #[test] - fn a_shortened_deadline_is_visible_in_the_status_line() { - let full = FastAllow::Live { - deadline_secs: cfc_ebpf_common::fast_allow::DEADLINE_SECS, - reduced: None, - }; - assert_eq!(full.describe(), "live"); - - let short = FastAllow::Live { - deadline_secs: cfc_ebpf_common::fast_allow::DEADLINE_SECS_REDUCED, - reduced: Some(REDUCED_IMPRECISE_EXIT.to_string()), - }; - let said = short.describe(); - assert_ne!( - said, "live", - "a weaker guarantee must not read as the full one" - ); - assert!( - said.contains(&cfc_ebpf_common::fast_allow::DEADLINE_SECS_REDUCED.to_string()), - "the status line must name the number: {said}" - ); - assert!( - said.contains("leader only"), - "two different weaknesses give the same six seconds, so the status must say \ - which: {said}" - ); - } - - #[test] - fn a_mark_sharing_a_bit_with_a_known_selector_is_refused() { - assert_eq!(collides_with(0x0000_8000), Some("kube-proxy (drop)")); - assert_eq!(collides_with(0xdead_8000), Some("kube-proxy (drop)")); - assert_eq!(collides_with(0x0000_4000), Some("kube-proxy (masquerade)")); - assert_eq!(collides_with(0x0008_0000), Some("Tailscale")); - assert_eq!(collides_with(0x1208_0000), Some("Tailscale")); - assert_eq!(collides_with(0x0004_0000), Some("Tailscale (bypass)")); - // wg-quick's value is caught, though by kube-proxy's masquerade bit - // rather than by its own entry: 0xca6c has bit 14 set. Its entry is - // kept anyway - it documents the selector, and it is what would catch - // the value if the kube-proxy masks ever moved. - assert!(collides_with(0x0000_ca6c).is_some()); - - // And values no selector here claims. - assert_eq!(collides_with(0x0000_0a6c), None); - assert_eq!(collides_with(0x0003_3331), None); - } - - #[test] - fn a_drawn_mark_is_never_unarmed_and_never_collides() { - // A deterministic walk over the space rather than a real rng: the - // property is about the sieve, and a test that draws randomly would - // pass or fail randomly. - let mut seed = 0x1234_5678u32; - let mut draw = || { - seed = seed.wrapping_mul(1_664_525).wrapping_add(1_013_904_223); - seed - }; - for _ in 0..2000 { - let mark = pick_mark(&mut draw).expect("a healthy source always yields one"); - assert_ne!(mark, cfc_ebpf_common::fast_allow::UNARMED); - assert_eq!( - collides_with(mark), - None, - "drew a colliding mark 0x{mark:08x}" - ); - } - } - - /// A source that only ever offers unusable values must not hang the - /// daemon - and must not be answered with a *constant* either, which is - /// what an earlier fallback did: a value that does not depend on the draw - /// is the same on every machine that reaches it, which is the published - /// bypass token the random draw exists to avoid. Refusing leaves the fast - /// path off, which is where it degrades to anyway. - #[test] - fn a_degenerate_draw_arms_nothing_rather_than_a_constant() { - assert_eq!(pick_mark(|| 0x0000_8000), None); - assert_eq!(pick_mark(|| cfc_ebpf_common::fast_allow::UNARMED), None); - } - #[test] fn a_world_writable_object_is_refused_but_only_under_refuse() { let dir = tempfile::tempdir().expect("tempdir"); @@ -2184,9 +1353,6 @@ mod tests { KernelProcTable::new(), None, Trust::Refuse, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - Default::default(), ) .err() .expect("a world-writable object must not be loaded"); @@ -2201,9 +1367,6 @@ mod tests { KernelProcTable::new(), None, Trust::Warn, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - Default::default(), ) .err() .expect("`not an ELF` cannot load either way"); @@ -2226,9 +1389,6 @@ mod tests { KernelProcTable::new(), None, Trust::Refuse, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - Default::default(), ) .err() .expect("there is no object there"); @@ -2532,9 +1692,6 @@ mod tests { KernelProcTable::new(), None, Trust::Warn, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - Default::default(), ) .expect("the unit's capability set must be enough to load the object"); @@ -2572,37 +1729,33 @@ mod tests { "{} must be pinned alongside the verdict maps", sock_pids_pin.display() ); - // The fast path's maps and links. Unpinned, a restart would split - // them from the programs still attached (the SOCK_PIDS lesson), and - // an unpinned deadline map would be one no restarted daemon could - // ever refresh. + // The legacy Fast Allow maps stay pinned while the kernel object + // carries them (ABI v4): the startup disarm has to reach the maps the + // pinned programs read, and an unpinned copy would split them across + // a restart (the SOCK_PIDS lesson). for name in [ enforce::MAP_FAST_ALLOW, enforce::MAP_FAST_ALLOW_UNTIL, enforce::MAP_FAST_ALLOW_MARK, enforce::MAP_ALLOW_EVENTS, - enforce::LINK_SENDMSG4, - enforce::LINK_SENDMSG6, ] { let pin = enforce::pin_dir().join(name); assert!(pin.exists(), "{} must be pinned", pin.display()); } - // And the rung the fast path's safety argument stands on: the exec and - // exit tracepoint links pinned, not merely attached. + // The exec and exit tracepoint links pinned, not merely attached. // // Only the pin makes their clears outlive the daemon, and the connect // programs' links are pinned separately - so on a kernel where these - // two cannot be, the marking survives a dead daemon while the clearing - // does not. The loader used to throw the pin outcome away and the - // ladder could not see the difference; this is the assertion that - // stops it being thrown away again. Only this test can make it: the - // matrix guests have no bpffs. + // two cannot be, denials survive a dead daemon while their eviction + // does not. The loader used to throw the pin outcome away; this is + // the assertion that stops it being thrown away again. Only this test + // can make it: the matrix guests have no bpffs. for name in [enforce::LINK_EXEC, enforce::LINK_EXIT] { let pin = enforce::pin_dir().join(name); assert!( pin.exists(), - "{} must be pinned, or the fast path's clears die with the daemon", + "{} must be pinned, or eviction dies with the daemon", pin.display() ); } @@ -2611,9 +1764,65 @@ mod tests { "the report must say both lifecycle links pinned when they did: {:?}", report.notes ); + drop(attached); + + // The legacy disarm, on the path that needs it: a pinned MARK left + // armed by a 0.4-0.6 daemon that died, met by a restart on the + // inherited path with no engine. Only a bpffs host can show this. + { + let dir = enforce::pin_dir(); + let mark = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW_MARK)) + .expect("reopen the pinned FAST_ALLOW_MARK"); + let mut mark = aya::maps::Array::<_, u32>::try_from(aya::maps::Map::Array(mark)) + .expect("FAST_ALLOW_MARK is an array"); + mark.set(0, 0x0003_3331, 0).expect("arm the legacy mark"); + let until = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW_UNTIL)) + .expect("reopen the pinned FAST_ALLOW_UNTIL"); + let mut until = aya::maps::Array::<_, u64>::try_from(aya::maps::Map::Array(until)) + .expect("FAST_ALLOW_UNTIL is an array"); + until.set(0, u64::MAX, 0).expect("set a legacy deadline"); + let grants = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW)) + .expect("reopen the pinned FAST_ALLOW"); + let mut grants = BpfHashMap::<_, u32, u32>::try_from(aya::maps::Map::HashMap(grants)) + .expect("FAST_ALLOW is a hash map"); + grants.insert(1, 1, 0).expect("leave a legacy grant"); + } + let (attached, report) = load_and_attach( + Path::new(&path), + DnsCache::new(), + KernelProcTable::new(), + None, + Trust::Warn, + ) + .expect("the restart must load too"); + assert_eq!( + report.enforcement, + Enforcement::Inherited, + "{:?}", + report.notes + ); + { + let dir = enforce::pin_dir(); + let mark = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW_MARK)).expect("mark"); + let mark = + aya::maps::Array::<_, u32>::try_from(aya::maps::Map::Array(mark)).expect("array"); + assert_eq!( + mark.get(&0, 0).expect("read"), + cfc_ebpf_common::fast_allow::UNARMED, + "a restart must disarm a legacy mark left in the pinned map" + ); + let until = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW_UNTIL)).expect("until"); + let until = + aya::maps::Array::<_, u64>::try_from(aya::maps::Map::Array(until)).expect("array"); + assert_eq!(until.get(&0, 0).expect("read"), 0); + let grants = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW)).expect("grants"); + let grants = + BpfHashMap::<_, u32, u32>::try_from(aya::maps::Map::HashMap(grants)).expect("hash"); + assert_eq!(grants.keys().count(), 0, "legacy grants must be emptied"); + } println!( - "seven programs attached; connect, sendmsg and lifecycle links plus \ - every fast-path map pinned, without CAP_SYS_ADMIN" + "five programs attached; connect and lifecycle links plus the \ + legacy maps pinned and disarmed, without CAP_SYS_ADMIN" ); drop(attached); @@ -2662,9 +1871,6 @@ mod tests { table.clone(), None, Trust::Warn, - tokio::sync::broadcast::channel(8).0, - crate::stats::Stats::new(), - Default::default(), ) .expect("load"); for note in &report.notes { @@ -2714,80 +1920,9 @@ mod tests { "cgroup/connect4|6 must load and attach on every matrix kernel: {:?}", report.notes ); - // The fast path's kernel side rides with the cookie connect variants, - // but not all the way: 5.10 verifies bpf_setsockopt on connect hooks - // and refuses it on sendmsg ones, so "cookie verified implies sendmsg - // verified" is false on the matrix floor and is not asserted. What is - // asserted is consistency: the two sendmsg programs verify together - // or not at all. - println!("fast_allow = {:?}", report.fast_allow); - let measured = |name: &str| report.verified_insns.iter().any(|(p, _)| p == name); - assert_eq!( - measured(enforce::PROG_SENDMSG4), - measured(enforce::PROG_SENDMSG6), - "the sendmsg pair must verify together or not at all: {:?}", - report.verified_insns - ); - // Off, and off for the reason this test's own setup dictates: it - // hands the layer no decision engine, so there is no sink to grant - // from. The first version of this assertion expected the config - // reason and learned on every matrix kernel at once that the test's - // inputs never reach the config check - an assertion about a state - // the test does not produce is the class of mistake this file exists - // to catch in the code, not to commit in the tests. - match &report.fast_allow { - Some(FastAllow::Off(why)) => assert!( - why.contains("no decision engine"), - "fast-allow is off for a reason this setup does not produce: {why}" - ), - other => panic!("fast-allow should be off here (no engine), got {other:?}"), - } - // The eligibility ladder's facts, observed on this kernel. The decision - // in this report is `Off` because the test hands the layer no engine, - // so the decision a rule set that grants would get is taken here from - // the facts the loader gathered - the ladder is a pure function with - // its own tests - and printed for the matrix summary. `config_on` and - // `has_maps` are this test's inputs, not observations: the object under - // test carries the maps, and the question is what the kernel would - // allow with the feature switched on. `lifecycle_pinned` is observed - // but says as much about the host as about the kernel: the qemu - // guests mount no bpffs, so it is false there on every kernel and - // the decision printed there carries the reduced deadline where a - // real host of the same kernel would pin. `exit_precise` and the - // capability are the kernel's own answers. - let capability = report - .fast_path_capability - .expect("enforcement is live, asserted above, so the capability is recorded"); - let facts = LadderFacts { - config_on: true, - has_maps: true, - exit_tracking: report.exit_tracking, - exit_precise: report.exit_precise, - lifecycle_pinned: report.lifecycle_pinned, - capability, - }; - let decision = fast_path_decision(&facts); - let described = match &decision { - Ok(reduced) if reduced.is_empty() => "live".to_string(), - Ok(reduced) => format!( - "live, grants lapse within {}s ({})", - deadline_pair(true).0, - reduced.join("; ") - ), - Err(why) => format!("off: {why}"), - }; println!( - "fast path on this kernel: {described} [exit_precise={} lifecycle_pinned={} capability={}]", - report.exit_precise, - report.lifecycle_pinned, - capability.as_str() - ); - // Runtime grants remain disabled regardless of the capabilities this - // kernel verifies. The matrix still records those facts for Deny and - // attribution compatibility; it must never claim Fast Allow is live. - assert!( - decision.is_err(), - "Fast Allow must be disabled on every matrix kernel: {decision:?}" + "lifecycle on this kernel: exit_precise={} lifecycle_pinned={}", + report.exit_precise, report.lifecycle_pinned ); // Where the matrix has already shown what a kernel answers, the answer // is asserted, so a kernel that changes its mind is caught here and not @@ -2795,26 +1930,10 @@ mod tests { // line printed above is what the first run of a new matrix entry // contributes to this table. match kernel_major_minor() { - Some((5, 10)) => { - assert!( - !report.exit_precise, - "5.10 has no group_dead in sched_process_exit" - ); - assert_eq!( - capability, - enforce::FastPathCapability::SendmsgUnavailable, - "5.10 takes the connect hooks and refuses bpf_getsockopt on the sendmsg ones" - ); - } - Some((5, 15)) | Some((6, 12)) => { + Some((5, 10)) | Some((5, 15)) | Some((6, 12)) => { assert!( !report.exit_precise, - "5.15 and 6.12 have no group_dead in sched_process_exit" - ); - assert_eq!( - capability, - enforce::FastPathCapability::Ready, - "5.15 already takes the sendmsg hooks 5.10 refuses, and so does 6.12" + "5.10, 5.15 and 6.12 have no group_dead in sched_process_exit" ); } Some((6, 18)) | Some((7, 1)) => { @@ -2822,18 +1941,11 @@ mod tests { report.exit_precise, "group_dead is in sched_process_exit from 6.18 on" ); - assert_eq!(capability, enforce::FastPathCapability::Ready); } other => { println!("kernel {other:?}: no recorded answer for this one, observation only") } } - // Where the sendmsg pair did not verify, the report must say so in the - // fast-path terms - the note is the only trace a kernel like 5.10 - // leaves, and it must not be mistaken for the basic-connect fallback. - if measured(enforce::PROG_CONNECT4) && !measured(enforce::PROG_SENDMSG4) { - println!("sendmsg hooks refused by this kernel; fast path unavailable here"); - } // `bpf_prog_info.verified_insns` exists since kernel 5.16; before // that, an empty report is the correct answer, not a recording // failure. On a kernel that does report counts, every program that diff --git a/crates/cfc-daemon/src/ebpf/nft_set.rs b/crates/cfc-daemon/src/ebpf/nft_set.rs index 69002d2..98ca840 100644 --- a/crates/cfc-daemon/src/ebpf/nft_set.rs +++ b/crates/cfc-daemon/src/ebpf/nft_set.rs @@ -1,78 +1,33 @@ -//! The one thing the daemon does to nftables: put its fast-allow mark into -//! the `fast_allow` set the snippet declares empty, and take it out again. +//! The daemon's two questions for nftables: is `table inet colony_firewall` +//! loaded, and, once at start, flush the legacy `fast_allow` set. //! -//! Until this module existed the daemon never wrote to the ruleset. It sat on -//! the far end of NFQUEUE 0 and `colony-firewall-nft.service` owned every -//! rule; that boundary was kept on purpose. It moves for exactly one reason, -//! given in full in [`cfc_ebpf_common::fast_allow`]: the mark the connect -//! hooks set must be a per-start random value, so it cannot be a literal in -//! the snippet, so something at runtime has to tell nftables what it is. This -//! is that something, and it is kept to two statements: `add element` when -//! the fast path comes up, `flush set` at every start and at shutdown. +//! The daemon does not write the ruleset. It sits on the far end of NFQUEUE 0 +//! and `colony-firewall-nft.service` owns every rule. The flush is the one +//! exception, and it is a removal: Fast Allow (0.4.0 to 0.6) put a per-start +//! random mark into that set, and a daemon that crashed while armed left it +//! there, accepted by a ruleset nothing reloaded. Fast Allow is gone; the +//! flush stays until no supported upgrade path starts from a release that had +//! it. //! //! # Why a child process //! //! `nft(8)` is run as a child, the way the provenance backend runs `rpm -qa`, //! and with the same discipline: a fixed program path, a deadline with a kill //! behind it, `LC_ALL=C`, stderr captured into the error and never parsed as -//! data. The alternative - speaking nf_tables netlink from the daemon - is a -//! batching protocol with its own cache semantics, for two statements the -//! package already `Requires: nftables` to make. One fork at start and one at -//! stop, both off the packet path. -//! -//! # What this module refuses to do -//! -//! Create the table or the set. Those belong to the snippet and its unit; a -//! daemon that made them on demand would be a daemon that quietly builds a -//! ruleset nobody loaded, and on a host without the snippet that ruleset -//! would be a fail-closed table with the wrong owner. A missing set is -//! reported as exactly that, with the fix. A missing table is reported as the -//! ordering it usually is: `colony-firewall-nft.service` is `After=` the -//! daemon, so at daemon start the table is normally not there *yet*, and at -//! stop (`PartOf=`) it is normally already gone. [`Absent`] tells the two -//! apart so the loader can retry the first, and [`disarm`] treats both as -//! nothing left to flush. -//! -//! # The value is a secret -//! -//! A forger holding `CAP_NET_RAW` but not `CAP_NET_ADMIN` can read neither the -//! ruleset nor the BPF map, and that is the whole argument for a random value. -//! The journal and `cfc status` are readable by more people than the ruleset -//! is, so the mark never appears in a log line or an error: nft echoes the -//! failing command back on stderr, and that echo is redacted before it goes -//! anywhere. -//! -//! There is a third read path, and naming two of them made this argument look -//! stronger than it is: a process the fast path has **granted** can read the -//! value straight off its own socket with `getsockopt(SO_MARK)`. Nothing -//! prevents that and nothing should - that process is allowed by a rule, which -//! is why the kernel marked it. What follows is the scope of the secret: it -//! holds against processes that have never been granted, and not against one -//! that has and then, say, execs into something a rule denies. The kernel -//! clears `FAST_ALLOW` on exec, but it cannot unmark a socket the old program -//! already passed on. The mark being redrawn at every daemon start is what -//! bounds that, and it is why [`disarm`] runs unconditionally at start rather -//! than only on the arming path - a value left accepted in the set that -//! nothing refreshes is one every past grantee still knows. - -// The only caller is the loader, which is behind the `ebpf` feature. The -// module itself stays in every build so its tests run in the default suite, -// for the same reason `tracefs` does. -#![cfg_attr(not(feature = "ebpf"), allow(dead_code))] +//! data. Speaking nf_tables netlink from the daemon would be a batching +//! protocol with its own cache semantics, for two commands the package +//! already `Requires: nftables` to make. One fork at start and one a minute +//! for the probe, both off the packet path. -use std::fmt; use std::io::Read as _; use std::path::Path; use std::process::{Command, ExitStatus, Stdio}; use std::time::{Duration, Instant}; -use anyhow::{anyhow, bail}; -use cfc_ebpf_common::fast_allow; +use anyhow::anyhow; use tracing::debug; -/// The family and table the snippet declares, and the set inside it. Named -/// once here and spelled into every command, so a rename in the snippet fails -/// the `list set` probe rather than silently arming nothing. +/// The family and table the snippet declares, and the legacy set inside it. const FAMILY: &str = "inet"; const TABLE: &str = "colony_firewall"; const SET: &str = "fast_allow"; @@ -88,135 +43,16 @@ const NFT_CANDIDATES: [&str; 2] = ["/usr/sbin/nft", "/usr/bin/nft"]; /// /// nft holds the nf_tables transaction lock for the length of its batch, and /// waits for it when another process - a large `nft -f`, a container runtime -/// rewriting its chains - holds it first. A daemon that hangs at start or -/// stop behind that lock is worse than one whose fast path stays off, and at -/// shutdown a hang here would run into the unit's stop timeout. Five seconds -/// is far past any healthy command and far short of that timeout. +/// rewriting its chains - holds it first. A daemon that hangs at start behind +/// that lock is worse than one whose legacy flush is logged as failed. Five +/// seconds is far past any healthy command and far short of the unit's +/// start timeout. const NFT_TIMEOUT: Duration = Duration::from_secs(5); /// How often the deadline is re-checked while waiting; same value and same /// reasoning as the rpm query's. const NFT_POLL_INTERVAL: Duration = Duration::from_millis(50); -/// Adds `mark` to `set fast_allow` in `table inet colony_firewall`. -/// -/// Errors when the set does not exist (a snippet that predates it), when the -/// table is not loaded, when `nft` is missing, or when the command fails; the -/// loader turns any error into `FastAllow::Off(reason)` and never arms the -/// kernel side without it. The first two are an [`Absent`], reachable through -/// `downcast_ref`, because one of them is the normal state right after -/// startup and deserves a retry rather than a reason. -/// -/// Refuses [`fast_allow::UNARMED`] outright: zero is the mark every socket -/// carries when nothing has marked it, and a zero element in the set would -/// accept every unmarked packet on the machine. -/// Serialises [`arm`] against the shutdown flush, and refuses an arm once that -/// flush has begun. -/// -/// Both run `nft`, and the heartbeat's arm runs inside `spawn_blocking`, which -/// `JoinHandle::abort` cannot cancel: aborting the heartbeat task leaves any -/// `nft add element` already in flight running to completion on the blocking -/// pool. So `Drop for Attached` could flush the set and *then* have that add -/// put the element back - leaving a mark accepted by the ruleset with no -/// daemon alive to refresh a deadline, honour a revocation, or ever remove it. -/// -/// The bool inside is "shutdown has started". Holding the lock across the -/// check and the command is what makes the two orders both end flushed: if the -/// heartbeat holds it, the shutdown flush waits and runs last; if the shutdown -/// flush holds it, the heartbeat then sees the flag and does not arm. -/// -/// It says "has started", not "has happened", so it belongs to one layer's -/// lifetime and [`disarm_for_start`] clears it. Leaving it set was a real bug -/// for as long as it existed: this is a process-global, so the first `Attached` -/// dropped would have refused every arm for the rest of that process - every -/// later test in one test binary, and any reload of the eBPF layer that did -/// not also restart the daemon. -static SHUTDOWN: std::sync::Mutex = std::sync::Mutex::new(false); - -/// Locks [`SHUTDOWN`], ignoring poisoning: the flag is a single bool that no -/// panic can leave inconsistent, and refusing to shut down cleanly because -/// some other thread panicked would be the worse failure. -fn shutdown_gate() -> std::sync::MutexGuard<'static, bool> { - SHUTDOWN.lock().unwrap_or_else(|e| e.into_inner()) -} - -/// Puts `mark` in the set, and only `mark`. -/// -/// Flushes first, so the set is left holding exactly this daemon's value -/// rather than this one added to whatever a predecessor left. Refuses once -/// [`disarm_for_shutdown`] has run - see [`SHUTDOWN`]. -pub(super) fn arm(mark: u32) -> anyhow::Result<()> { - if mark == fast_allow::UNARMED { - bail!( - "refusing to arm the fast path with mark {}: that is the mark of every \ - socket nothing has marked, and accepting it would accept everything", - mark_literal(mark) - ); - } - let gate = shutdown_gate(); - if *gate { - bail!("not arming the fast-allow set: the daemon is shutting down"); - } - arm_commands(mark, run) -} - -/// The ordered transaction steps, injectable without executing nft in tests. -fn arm_commands(mark: u32, mut run: impl FnMut(Op) -> Result<(), Failed>) -> anyhow::Result<()> { - match run(Op::ListSet) { - Ok(()) => {} - Err(failed) if failed.is_no_such_object() => { - // Table or set? nft says "No such file or directory" for both and - // only moves the caret. One more probe tells them apart, and it - // runs on this path alone. - let absent = match run(Op::ListTable) { - Ok(()) => Absent::Set, - Err(failed) if failed.is_no_such_object() => Absent::Table, - Err(failed) => return Err(failed.into_error(Op::ListTable)), - }; - return Err(anyhow::Error::new(absent)); - } - Err(failed) => return Err(failed.into_error(Op::ListSet)), - } - // Flush before adding, so this leaves the set holding *exactly* this - // daemon's mark rather than adding to whatever is already there. - // - // The startup flush is not enough on its own. It runs once, and on a boot - // it runs when the table does not exist yet - `colony-firewall-nft.service` - // is ordered after this daemon - so it flushes nothing. The heartbeat then - // retries this function on every heartbeat until the table appears, and a - // plain `add element` at that point would leave a crashed predecessor's - // mark accepted alongside this daemon's, for as long as the ruleset lives. - // A mark that is still accepted but that nothing refreshes is the worst - // shape this set can be in: every process that ever held it can read it - // back with `getsockopt(SO_MARK)` and set it again. - run(Op::FlushSet).map_err(|failed| failed.into_error(Op::FlushSet))?; - let op = Op::AddElement(mark); - run(op).map_err(|failed| failed.into_error(op))?; - debug!("fast-allow mark added to set {FAMILY} {TABLE} {SET}"); - Ok(()) -} - -/// Whether the ruleset still accepts `mark`. -/// -/// Arming is not a fact that stays true. `systemctl restart nftables`, a -/// `nft -f` that reloads the machine's ruleset, or anything else that -/// recreates `table inet colony_firewall` leaves the set empty while this -/// daemon goes on marking sockets and `cfc status` goes on saying `live`. The -/// heartbeat calls this so that state is noticed and re-armed rather than -/// reported. -/// -/// A missing table or set answers `false`, not an error: they are the same -/// answer for the caller - the mark is not accepted - and the caller's re-arm -/// path already classifies which of the two it is. -pub(super) fn holds(mark: u32) -> anyhow::Result { - let op = Op::GetElement(mark); - match run(op) { - Ok(()) => Ok(true), - Err(failed) if failed.is_no_such_object() => Ok(false), - Err(failed) => Err(failed.into_error(op)), - } -} - /// Whether `table inet colony_firewall` is loaded at all. /// /// This is the question `cfc status`'s `enforcing` is really asking. Without @@ -239,144 +75,43 @@ pub(super) fn table_loaded() -> anyhow::Result { } } -/// Flushes `set fast_allow`, so that no value (this daemon's or a previous -/// one's) is accepted by the ruleset. -/// -/// This is the plain flush: the late withdrawal in `load_and_attach` calls it -/// when the ring consumers failed after arming. The start and shutdown -/// flushes are [`disarm_for_start`] and [`disarm_for_shutdown`], which also -/// move the gate - an earlier version of this comment named them as this -/// function's callers, and they are not. +/// Flushes the legacy `set fast_allow`, so that no mark an older daemon left +/// there is accepted by the ruleset. /// /// A missing table or a missing set is success: there is nothing in either -/// that could accept a mark. The nft table intentionally remains loaded across -/// daemon restarts and stops; only an explicit nft-unit stop removes it. -pub(super) fn disarm() -> anyhow::Result<()> { - let _gate = shutdown_gate(); - flush() -} - -/// [`disarm`], beginning a new layer lifetime. -/// -/// For the one call at the top of `load_and_attach`. Clears the flag a -/// previous `Attached`'s shutdown set: that flag means "the layer that owned -/// this set is going away", and by here it has gone. -pub(super) fn disarm_for_start() -> anyhow::Result<()> { - begin_lifetime(); - flush() -} - -/// The flag half of [`disarm_for_start`], split out so the regression it -/// exists for can be tested without running `nft`. -fn begin_lifetime() { - *shutdown_gate() = false; -} - -/// [`disarm`], and no [`arm`] after it. -/// -/// For `Drop for Attached` only. Sets the flag the heartbeat's in-flight arm -/// will see, under the same lock, so the set cannot be re-armed behind a -/// daemon that has already stopped. -pub(super) fn disarm_for_shutdown() -> anyhow::Result<()> { - let mut gate = shutdown_gate(); - *gate = true; - flush() -} - -fn flush() -> anyhow::Result<()> { +/// that could accept a mark. At boot the table is normally not loaded yet, +/// because `colony-firewall-nft.service` is ordered after the daemon. +pub(super) fn flush() -> anyhow::Result<()> { match run(Op::FlushSet) { Ok(()) => { - debug!("fast-allow set flushed"); + debug!("legacy fast_allow set flushed"); Ok(()) } Err(failed) if failed.is_no_such_object() => { - debug!("fast-allow set not present, nothing to flush"); + debug!("legacy fast_allow set not present, nothing to flush"); Ok(()) } Err(failed) => Err(failed.into_error(Op::FlushSet)), } } -/// Why [`arm`] found nothing to arm. -/// -/// Returned as the error itself rather than as context, so the loader can -/// `downcast_ref::()` and treat the two differently: a missing table -/// is the expected state right after the daemon starts (the nft unit is -/// ordered after it) and is worth retrying; a missing set will not fix itself -/// and names its fix. -#[derive(Debug, Clone, Copy, PartialEq, Eq)] -pub(super) enum Absent { - /// `table inet colony_firewall` is not loaded. - Table, - /// The table is loaded but carries no `fast_allow` set. - Set, -} - -impl fmt::Display for Absent { - fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { - match self { - Absent::Table => write!( - f, - "table {FAMILY} {TABLE} is not loaded, so there is no {SET} set to arm; \ - colony-firewall-nft.service loads it once the daemon is up" - ), - Absent::Set => write!( - f, - "the loaded nftables snippet predates {SET}; reinstall \ - systemd/nftables-snippet.conf and restart colony-firewall-nft.service" - ), - } - } -} - -impl std::error::Error for Absent {} - -/// The commands this module issues. Three do the work; `ListTable` exists -/// only to say which of two things is missing. +/// The commands this module issues. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum Op { - /// `nft list set inet colony_firewall fast_allow`: does the set exist? - /// Its output is discarded; the exit status is the answer. - ListSet, - /// `nft list table inet colony_firewall`, run only after `ListSet` failed. + /// `nft list table inet colony_firewall`: is the table loaded? Its output + /// is discarded; the exit status is the answer. ListTable, - /// `nft add element inet colony_firewall fast_allow { 0x }`. - AddElement(u32), /// `nft flush set inet colony_firewall fast_allow`. FlushSet, - /// `nft get element inet colony_firewall fast_allow { 0x }`: is - /// this daemon's mark still accepted? Status-only, like `ListSet`, which - /// is why it is a `get` and not a `list` the caller would have to parse. - GetElement(u32), } /// The argument vector for `op`, without the program. -/// -/// Pure, and the part of this module that is tested without nft: what the -/// tests pin is that every command names the snippet's table and set, and -/// that the mark is spelled one way. fn argv(op: Op) -> Vec { let words: &[&str] = match op { - Op::ListSet => &["list", "set", FAMILY, TABLE, SET], Op::ListTable => &["list", "table", FAMILY, TABLE], - Op::AddElement(_) => &["add", "element", FAMILY, TABLE, SET], Op::FlushSet => &["flush", "set", FAMILY, TABLE, SET], - Op::GetElement(_) => &["get", "element", FAMILY, TABLE, SET], }; - let mut argv: Vec = words.iter().map(|w| w.to_string()).collect(); - if let Op::AddElement(mark) | Op::GetElement(mark) = op { - argv.push(format!("{{ {} }}", mark_literal(mark))); - } - argv -} - -/// The mark as a `type mark` element: `0x` and exactly eight hex digits. -/// -/// One fixed spelling, because nft echoes the failing command line back on -/// stderr verbatim and [`redact`] removes the literal by exact match; a -/// literal that could be spelled two ways could be leaked one of them. -fn mark_literal(mark: u32) -> String { - format!("0x{mark:08x}") + words.iter().map(|w| w.to_string()).collect() } /// One nft command that did not succeed. @@ -397,13 +132,11 @@ impl Failed { matches!(self, Failed::Nft { stderr, .. } if stderr_names_no_such_object(stderr)) } - /// Folds into an error whose text is safe to log: the mark is redacted - /// from the command line and from nft's echo of it. fn into_error(self, op: Op) -> anyhow::Error { - let command = redact(op, format!("nft {}", argv(op).join(" "))); + let command = format!("nft {}", argv(op).join(" ")); match self { Failed::Nft { status, stderr } => { - anyhow!("{command} failed ({status}): {}", redact(op, stderr).trim()) + anyhow!("{command} failed ({status}): {}", stderr.trim()) } Failed::Run(e) => e.context(format!("running {command}")), } @@ -418,16 +151,6 @@ fn stderr_names_no_such_object(stderr: &str) -> bool { stderr.contains("No such file or directory") } -/// Replaces the mark literal with a placeholder. Only [`Op::AddElement`] and -/// [`Op::GetElement`] carry the mark; every other command's text is returned -/// as it is. -fn redact(op: Op, text: String) -> String { - match op { - Op::AddElement(mark) | Op::GetElement(mark) => text.replace(&mark_literal(mark), ""), - _ => text, - } -} - /// The first of [`NFT_CANDIDATES`] that exists. fn locate_nft() -> anyhow::Result<&'static str> { NFT_CANDIDATES @@ -506,89 +229,6 @@ fn run(op: Op) -> Result<(), Failed> { #[cfg(test)] mod tests { - - /// One test, not three: the flag is a process-global, so separate tests - /// touching it would race each other under the parallel harness. - /// - /// Neither half reaches `nft`. `arm` checks the gate before it runs - /// anything, and `begin_lifetime` is the flag half of `disarm_for_start` - /// split out for exactly this. - #[test] - fn a_shutdown_refuses_arming_until_a_new_layer_begins() { - let mark = 0x0003_3331; - - *shutdown_gate() = true; - let e = arm(mark).expect_err("an arm after shutdown must be refused"); - assert!( - e.to_string().contains("shutting down"), - "refused for the wrong reason: {e}" - ); - - // The regression: the flag was set by shutdown and cleared by nothing, - // so the first `Attached` dropped refused every arm for the rest of - // the process - every later test in one binary, and any reload of the - // layer that did not also restart the daemon. - begin_lifetime(); - assert!( - !*shutdown_gate(), - "a new layer lifetime must clear a previous shutdown's flag" - ); - - // Deliberately not "and now an arm succeeds": getting past the gate is - // the only thing left to check, and checking it means letting `arm` - // run nft - which on a machine that *does* have the table loaded would - // put a live element in a live ruleset from a unit test. The gate is - // two lines and the flag above is the whole of its state. - } - - #[test] - fn a_failed_flush_never_adds_a_mark() { - let mut attempted = Vec::new(); - let result = arm_commands(0x0003_3331, |op| { - attempted.push(op); - if op == Op::FlushSet { - Err(Failed::Run(anyhow!("flush refused"))) - } else { - Ok(()) - } - }); - assert!(result.is_err()); - assert_eq!(attempted, [Op::ListSet, Op::FlushSet]); - } - - /// The element check is status-only, like the set probe: `get element` - /// exits non-zero with ENOENT when the element is absent, so nothing here - /// has to parse nft's output. - #[test] - fn the_element_check_names_the_element_and_carries_the_mark() { - assert_eq!( - argv(Op::GetElement(0x0012_3456)), - vec![ - "get", - "element", - "inet", - "colony_firewall", - "fast_allow", - "{ 0x00123456 }", - ] - ); - } - - /// Both mark-carrying commands must be redacted, not just the add. nft - /// echoes the failing command line back verbatim, so a `get` that fails - /// would otherwise put the mark in the journal - where every reader of the - /// journal could set it. - #[test] - fn the_element_check_redacts_the_mark_from_what_nft_echoes() { - let mark = 0xdead_0a6c; - let echoed = format!("Error: No such file or directory\nget element inet colony_firewall fast_allow {{ {} }}", mark_literal(mark)); - let safe = redact(Op::GetElement(mark), echoed); - assert!( - !safe.contains(&mark_literal(mark)), - "leaked the mark: {safe}" - ); - assert!(safe.contains("")); - } use super::*; use std::os::unix::process::ExitStatusExt as _; @@ -597,7 +237,6 @@ mod tests { // moves, from under the table name to under the set name. const NO_TABLE_LIST: &str = "Error: No such file or directory\nlist set inet colony_firewall fast_allow\n ^^^^^^^^^^^^^^^\n"; const NO_SET_FLUSH: &str = "Error: No such file or directory\nflush set inet colony_firewall fast_allow\n ^^^^^^^^^^\n"; - const NO_SET_ADD: &str = "Error: No such file or directory\nadd element inet colony_firewall fast_allow { 0x1234abcd }\n ^^^^^^^^^^\n"; const SYNTAX_ERROR: &str = "Error: syntax error, unexpected newline\nexpected any of: , last\nlist set inet colony_firewall\n ^\n"; const NOT_PERMITTED: &str = "Error: Operation not permitted\nlist set inet colony_firewall fast_allow\n^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^^\n"; @@ -606,46 +245,11 @@ mod tests { } #[test] - fn add_element_spells_the_mark_as_eight_hex_digits() { - assert_eq!( - argv(Op::AddElement(0x1234_abcd)), - [ - "add", - "element", - "inet", - "colony_firewall", - "fast_allow", - "{ 0x1234abcd }" - ] - ); - // A small value is padded rather than shortened: one spelling only, - // which is what makes the redaction an exact match. - assert_eq!(argv(Op::AddElement(7)).last().unwrap(), "{ 0x00000007 }"); - assert_eq!(mark_literal(u32::MAX), "0xffffffff"); - } - - #[test] - fn every_set_command_names_the_snippets_table_and_set() { - for op in [Op::ListSet, Op::AddElement(1), Op::FlushSet] { - let argv = argv(op); - assert_eq!( - &argv[2..5], - ["inet", "colony_firewall", "fast_allow"], - "{op:?} does not address the snippet's set" - ); - } + fn list_and_flush_use_the_verbs_nft_understands() { assert_eq!( argv(Op::ListTable), ["list", "table", "inet", "colony_firewall"] ); - } - - #[test] - fn list_and_flush_use_the_verbs_nft_understands() { - assert_eq!( - argv(Op::ListSet), - ["list", "set", "inet", "colony_firewall", "fast_allow"] - ); assert_eq!( argv(Op::FlushSet), ["flush", "set", "inet", "colony_firewall", "fast_allow"] @@ -654,7 +258,7 @@ mod tests { #[test] fn a_missing_table_and_a_missing_set_both_read_as_absent() { - for stderr in [NO_TABLE_LIST, NO_SET_FLUSH, NO_SET_ADD] { + for stderr in [NO_TABLE_LIST, NO_SET_FLUSH] { assert!( stderr_names_no_such_object(stderr), "not classified as absent:\n{stderr}" @@ -679,45 +283,4 @@ mod tests { let failed = Failed::Run(anyhow!("No such file or directory")); assert!(!failed.is_no_such_object()); } - - #[test] - fn the_unarmed_value_is_refused_before_nft_runs() { - // Zero is what every unmarked socket reads; accepting it would accept - // everything. This must fail on a machine without nft, which is why - // the guard comes before any command is built. - let e = arm(fast_allow::UNARMED).expect_err("mark 0 must be refused"); - assert!(e.to_string().contains("0x00000000"), "{e}"); - } - - #[test] - fn an_add_element_failure_never_names_the_mark() { - let mark = 0x1234_abcd; - let failed = Failed::Nft { - status: exit_status(1), - stderr: NO_SET_ADD.to_string(), - }; - let text = format!("{:#}", failed.into_error(Op::AddElement(mark))); - assert!(!text.contains("1234abcd"), "leaked: {text}"); - assert!(text.contains(""), "{text}"); - assert!(text.contains("exit status: 1"), "{text}"); - - let failed = Failed::Run(anyhow!("spawning /usr/sbin/nft: permission denied")); - let text = format!("{:#}", failed.into_error(Op::AddElement(mark))); - assert!(!text.contains("1234abcd"), "leaked: {text}"); - assert!(text.contains(""), "{text}"); - } - - #[test] - fn absent_names_its_fix_and_survives_the_error_chain() { - let set = Absent::Set.to_string(); - assert!(set.contains("nftables-snippet.conf"), "{set}"); - assert!(set.contains("colony-firewall-nft.service"), "{set}"); - let table = Absent::Table.to_string(); - assert!(table.contains("not loaded"), "{table}"); - assert!(table.contains("colony-firewall-nft.service"), "{table}"); - - // What the loader relies on to tell "retry" from "operator". - let e = anyhow::Error::new(Absent::Table); - assert_eq!(e.downcast_ref::(), Some(&Absent::Table)); - } } diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index da0db1f..3974121 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -880,9 +880,6 @@ impl Firewall for FirewallService { enforcement: crate::ebpf::enforcement_level() .map_or("starting", |l| l.as_str()) .to_string(), - fast_allow: crate::ebpf::fast_allow_level() - .map(|f| f.describe()) - .unwrap_or_default(), })) } diff --git a/crates/cfc-daemon/src/main.rs b/crates/cfc-daemon/src/main.rs index b00345d..7f260f5 100644 --- a/crates/cfc-daemon/src/main.rs +++ b/crates/cfc-daemon/src/main.rs @@ -41,8 +41,8 @@ const RUNTIME_SHUTDOWN_GRACE: Duration = Duration::from_secs(5); /// moved. /// How often to ask nftables whether the table that feeds NFQUEUE is loaded. /// -/// One minute, matching the fast-allow set check: both are a fork and an exec, -/// and both bound how long `cfc status` may be stale by the same amount. +/// One minute: it is a fork and an exec, and it bounds how long `cfc status` +/// may be stale. const NFT_PRESENCE_INTERVAL: std::time::Duration = std::time::Duration::from_secs(60); const PROVENANCE_WARM_INTERVAL: std::time::Duration = std::time::Duration::from_secs(120); @@ -289,10 +289,8 @@ async fn run() -> anyhow::Result<()> { // A packet counter cannot tell "nothing is filtered" from "nothing is // happening" - an idle laptop looks identical to an unprotected one. So // ask nftables instead. Once a minute, on the blocking pool because it is - // a fork and an exec, which is the same cadence and the same reasoning as - // the fast-allow set check that already runs there. An error leaves the - // previous answer standing: "could not ask" must never render as "the - // firewall is gone". + // a fork and an exec. An error leaves the previous answer standing: "could + // not ask" must never render as "the firewall is gone". { let stats = stats.clone(); tokio::spawn(async move { @@ -365,12 +363,12 @@ async fn run() -> anyhow::Result<()> { // sock_diag + /proc alone, which is exactly what the daemon does when the // layer is unavailable anyway. // - // The loader flushes a predecessor's fast-allow mark at the top of every - // load. With the layer switched off in the config that flush is never - // reached, and the nftables set outlives daemons - so it is done here for - // exactly that case. Not under --dry-run, which touches nothing. - if !args.dry_run && !cfg.ebpf.enabled.wants_load() { - ebpf::flush_stale_fast_allow(); + // Flush the legacy fast_allow nftables set before the layer comes up, + // whatever its mode and whether or not it is compiled in: a mark an older + // daemon left there would otherwise stay accepted. Not under --dry-run, + // which touches nothing. + if !args.dry_run { + ebpf::flush_legacy_fast_allow_set(); } // Held for the daemon's lifetime: dropping it detaches the programs. @@ -395,8 +393,6 @@ async fn run() -> anyhow::Result<()> { // direction of the dependency the same as everywhere else: the eBPF // layer is handed what it may read, and owns nothing the daemon needs. Some(engine.clone()), - observed_tx.clone(), - stats.clone(), ); _ebpf.report.log(); // Publish it so `cfc status` can say where enforcement lives without anyone diff --git a/crates/cfc-proto/proto/cfc.proto b/crates/cfc-proto/proto/cfc.proto index e71d75c..f69ad7e 100644 --- a/crates/cfc-proto/proto/cfc.proto +++ b/crates/cfc-proto/proto/cfc.proto @@ -251,17 +251,10 @@ message StatusResponse { // "pinned" and "inherited" are the two that survive the daemon. string enforcement = 15; - // Whether process-wide allows are skipping the queue: "live", "off: ", - // or empty when the daemon predates this field or startup has not answered - // yet. - // - // It carries its reason because the path has several ways to be inert - // with nothing else changing - the switch left off, an attach inherited - // from a build without it, a kernel whose verifier lacks bpf_setsockopt on - // sock_addr, exit tracking without group_dead, an nftables set the snippet - // does not declare - and an allow that quietly takes the slow path looks - // exactly like one that is working, only later. - string fast_allow = 16; + // Field 16 was `fast_allow`, the state of the removed Fast Allow path. + // Never reuse the number or the name. + reserved 16; + reserved "fast_allow"; } message SetPausedRequest { From 4fcd0df12dec94d8b20a94b19f68a0cf27a10d80 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:07:28 +0200 Subject: [PATCH 003/125] build(ebpf): stop requiring the sendmsg programs The loader no longer looks up cfc_sendmsg4/6, so the object check stops demanding them. The FAST_ALLOW* maps and ALLOW_EVENTS stay required: they are still pinned by name so startup can disarm what an older release armed. The kernel matrix summary drops its "fast path" row. --- .github/workflows/ebpf.yml | 13 ------------- crates/xtask/src/main.rs | 6 ++---- 2 files changed, 2 insertions(+), 17 deletions(-) diff --git a/.github/workflows/ebpf.yml b/.github/workflows/ebpf.yml index fed9e30..8e76eb4 100644 --- a/.github/workflows/ebpf.yml +++ b/.github/workflows/ebpf.yml @@ -513,19 +513,6 @@ jobs: grep -o 'verified_insns: [^ ]* = [0-9]*' guest.log | while read -r _ p _ n; do echo "| \`${p}\` verified insns | ${n} |" done || true - # The fast path's facts as this guest observed them and the - # decision the eligibility ladder takes on them with the feature - # on. Not what `cfc status` would say on a real host of this - # kernel: the guest mounts no bpffs, so `lifecycle_pinned` is - # false here on every kernel and the deadline shown is the reduced - # one wherever a real host would pin. The other two facts - - # `exit_precise` and the capability - are the kernel's own. - # `tr -d '\r'`: the guest console writes CRLF, and a CR inside a - # table cell ends the markdown row early. `|| true` as above: a - # guest that died before printing this already failed on the marker. - grep -o 'fast path on this kernel: .*' guest.log | tr -d '\r' | tail -1 | while read -r line; do - echo "| fast path | ${line#fast path on this kernel: } |" - done || true } >> "${GITHUB_STEP_SUMMARY}" # Wall clock, not just instruction count. They are different diff --git a/crates/xtask/src/main.rs b/crates/xtask/src/main.rs index c69df89..404a94d 100644 --- a/crates/xtask/src/main.rs +++ b/crates/xtask/src/main.rs @@ -106,8 +106,6 @@ const REQUIRED_SYMBOLS: &[&str] = &[ // sock_addr programs; the loader tries the cookie ones first "cfc_connect4_basic", "cfc_connect6_basic", - "cfc_sendmsg4", - "cfc_sendmsg6", // maps "EXEC_EVENTS", "EXIT_EVENTS", @@ -122,8 +120,8 @@ const REQUIRED_SYMBOLS: &[&str] = &[ // "is it worth hashing?" guard "EXE_RULES", "EXE_RULES_ON", - // the fast path: the grant map, the deadline the daemon's heartbeat - // refreshes, the mark to set, and the ring the grants are reported on + // legacy Fast Allow maps: pinned by name so the daemon can disarm what an + // older release armed; gone with the kernel side at the next ABI bump "FAST_ALLOW", "FAST_ALLOW_UNTIL", "FAST_ALLOW_MARK", From 7ca86e4cee009e06c066fe3639049730949d3548 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:09:46 +0200 Subject: [PATCH 004/125] docs: describe Fast Allow as removed Fast Allow opened bypasses because a socket mark cannot prove which process sends, and its userspace side is gone. Say so in the README, architecture, roadmap, TODO, troubleshooting and hardening notes, and reduce the sample config to a note that the legacy keys are ignored. Packaging comments now describe nft as a one-shot flush at start plus the table probe. The VM bench drops its fast-N states and no longer reads the removed fast_allow status key. --- CHANGELOG.md | 10 ++++++ README.md | 5 +-- TODO.md | 28 +++++++-------- docs/ARCHITECTURE.md | 20 ++++++----- docs/HARDENING.md | 2 +- docs/ROADMAP.md | 20 ++++------- docs/TROUBLESHOOTING.md | 7 ++-- packaging/rpm/colony-firewall-control.spec | 4 +-- packaging/selinux/TESTING.md | 1 - packaging/selinux/colony_firewall.te | 37 +++++++------------- scripts/bench-latency.sh | 40 +++++++--------------- scripts/vm-bench/README.md | 24 +++++-------- scripts/vm-bench/init | 5 ++- scripts/vm-bench/plan.sh | 22 +++++------- scripts/vm-bench/report.py | 6 +--- systemd/colony-firewalld.service | 9 ++--- systemd/daemon.toml.sample | 15 +++----- 17 files changed, 108 insertions(+), 147 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 23beae5..1e166db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,16 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +### Removed + +- The Fast Allow userspace path, disabled since 0.7.0 because a socket mark + cannot prove which process sends and so opened bypasses. `cfc --json status` + no longer has a `fast_allow` key, `StatusResponse` field 16 is reserved, and + the `[ebpf] fast_allow` and `fast_allow_mark` keys are ignored with a + warning. For hosts upgrading from 0.4-0.6, startup still flushes the legacy + nftables set, disarms the legacy pinned maps and removes the old sendmsg + link pins. + ## [0.7.0] - 2026-09-30 ### Added diff --git a/README.md b/README.md index 14f247e..eac714c 100644 --- a/README.md +++ b/README.md @@ -279,8 +279,9 @@ Applications with `CAP_NET_RAW` can use AF_PACKET outside the `inet OUTPUT` hook. Raw IP packets can also coincide with another socket's tuple; socket attribution does not prove their origin. Use explicit application confinement or OS containment for those cases. -Fast Allow is disabled even when `fast_allow = true` is configured; allowed -flows use the normal NFQUEUE path. +Fast Allow was removed: a socket mark cannot prove which process sends, so it +opened bypasses. The old `[ebpf] fast_allow` and `fast_allow_mark` keys are +ignored with a warning, and allowed flows use the normal NFQUEUE path. Then confirm it is really filtering: diff --git a/TODO.md b/TODO.md index 09e472c..52f4374 100644 --- a/TODO.md +++ b/TODO.md @@ -20,15 +20,17 @@ longer lifts anything; `nft delete table` no longer lifts the denies it holds. Two pieces of it are deliberately not done, and both are real work rather than oversights: -**1a. Fast Allow is disabled.** Socket marks do not attest the current sender, -and grants can outlive their intended executable or rule. Every configuration, -including `fast_allow = true`, uses NFQUEUE for allowed flows. The nft snippet -no longer accepts the legacy set, startup clears old state, and upgrades reload -active nft units atomically. Reintroducing an in-kernel Allow requires a design -that verifies current socket ownership and revocation; the old mark protocol is -not a supported security boundary. - -The previous latency measurements describe the disabled implementation. The +**1a. Fast Allow was removed.** It let a process a lasting Allow covered skip +NFQUEUE by marking its sockets. A socket mark does not attest the current +sender, and grants could outlive their intended executable or rule, so it +opened bypasses. It was disabled in 0.7.0 and its userspace side has since been +removed; allowed flows use NFQUEUE. The nft snippet no longer accepts the +legacy set, startup flushes it and disarms the legacy pinned maps the kernel +object still carries until an ABI bump, and upgrades reload active nft units +atomically. Reintroducing an in-kernel Allow needs sender attestation: a design +that verifies current socket ownership and revocation. + +The previous latency measurements describe the removed implementation. The remaining NFQUEUE cost still warrants measurement and optimization, with the same application-policy semantics. @@ -45,14 +47,12 @@ be resolved to addresses in advance. Mostly done in `8db949b` and `b05eefc`: the SELinux module, the RPM provenance backend, the `.spec`, and a 5.10 entry in the kernel matrix that sits *below* -RHEL 9's backported 5.14. The fast-allow branch adds 5.15 above it, so the pair +RHEL 9's backported 5.14. The matrix also carries 5.15 above it, so the pair brackets the RHEL kernel: what both allow, 5.14 allows unless Red Hat took it out; what only 5.15 allows, 5.14 has only if they backported it; what both refuse, 5.14 may still have through a backport. Where the two disagree is the -list of things to check on a Rocky host rather than assume. The first 5.15 run -named one such thing: 5.15 already accepts `bpf_getsockopt` on the sendmsg hooks -that 5.10 refuses, so whether RHEL 9's 5.14 does is exactly what a Rocky host -has to answer; neither kernel has `group_dead`. +list of things to check on a Rocky host rather than assume. Neither kernel has +`group_dead`. What remains needs a real enforcing machine - except 2b, which turned out to be doable from CI after all: diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index d1817d0..9ddb5e5 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -115,7 +115,7 @@ lands just after the worker committed to a fresh wait. `scripts/vm-bench` attributes it - 4.90 ms of 5.67 at 300 flows, 5.24 ms of 7.61 at 3000, by building the same daemon with the constant at 200 us and measuring both in one boot. These are historical measurements, not a current performance guarantee. -Fast Allow is disabled, so allowed flows also pay the queue round trip. +Fast Allow was removed, so allowed flows also pay the queue round trip. **Prompt deduplication** requires the same UID, executable path, image digest, destination IP, destination port and protocol. Source address and port are @@ -392,19 +392,23 @@ inert as one built with `--no-default-features`. | `tracepoint/sched/sched_process_exit` | `sched:sched_process_exit` | evicts only on confirmed thread-group death | | `cgroup_skb/ingress` | cgroup v2 root | copies received DNS response payloads for diagnostics, never policy identity | | `cgroup/connect4`, `cgroup/connect6` | cgroup v2 root, link **pinned** | refuse `connect()` for pids the daemon has denied outright, before a packet exists | -| `cgroup/sendmsg4`, `cgroup/sendmsg6` | cgroup v2 root, link pinned | legacy mark-clearing support; Fast Allow stays disabled | **In-kernel denials.** The connect hooks refuse an executable denied process-wide with `EPERM`. Pinned denials outlive the daemon. Conditional rules, prompts and Allow decisions remain on the normal NFQUEUE path. -**Fast Allow is disabled in every runtime configuration.** A socket mark cannot +**Fast Allow was removed.** It marked the sockets of a process a lasting Allow +covered so that nftables accepted them ahead of the queue. A socket mark cannot prove the current sender's identity, and lifecycle checks do not repair that -property. `fast_allow = true` produces a warning and no grants or heartbeat. -The nft snippet has no mark-set accept rule. Startup flushes legacy accepted -marks, and package upgrades reload active nft units with one atomic transaction -to remove old acceptance rules. A failed cleanup emits an error and requires -operator action before filtering can be relied upon. +property, so it opened bypasses; it was disabled in 0.7.0 and its userspace +side is gone. The `[ebpf] fast_allow` keys still parse and only log a warning. +The kernel object still carries the Fast Allow maps until an ABI bump, so +startup flushes the legacy nft set once, disarms the pinned maps (unarmed mark, +zero deadline, no grants) and removes the old `sendmsg4`/`sendmsg6` link pins, +which detaches those hooks. The nft snippet has no mark-set accept rule, and +package upgrades reload active nft units with one atomic transaction. A failed +flush emits an error and requires operator action before filtering can be +relied upon. **Compatibility exit handling.** When `sched_process_exit` exposes `group_dead`, the kernel evicts only on confirmed process death. Without that field, it diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 0d804f4..42bc88b 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -357,7 +357,7 @@ to shrink what a code-execution bug could reach: | Directive | Why | |------------------------------------|---------------------------------| -| `CapabilityBoundingSet`, `AmbientCapabilities` | Seven capabilities, not full root: `CAP_NET_ADMIN` for NFQUEUE and for flushing legacy Fast Allow state, `CAP_NET_RAW` for Reject injection, `CAP_SYS_PTRACE` for reading other processes' `/proc`, `CAP_BPF` + `CAP_PERFMON` for the eBPF layer, `CAP_CHOWN` for the control socket's group, and `CAP_DAC_READ_SEARCH` for the `/proc/*/fd` walk attribution falls back to. The count and the list have to agree: this said seven and named five, and the two it left out are exactly the pair the SELinux policy was once missing - with the fail-closed ruleset, a daemon that cannot read `/proc` attributes nothing and the machine loses outbound traffic | +| `CapabilityBoundingSet`, `AmbientCapabilities` | Seven capabilities, not full root: `CAP_NET_ADMIN` for NFQUEUE, the nftables table probe and the one-shot flush of the legacy Fast Allow set, `CAP_NET_RAW` for Reject injection, `CAP_SYS_PTRACE` for reading other processes' `/proc`, `CAP_BPF` + `CAP_PERFMON` for the eBPF layer, `CAP_CHOWN` for the control socket's group, and `CAP_DAC_READ_SEARCH` for the `/proc/*/fd` walk attribution falls back to. The count and the list have to agree: this said seven and named five, and the two it left out are exactly the pair the SELinux policy was once missing - with the fail-closed ruleset, a daemon that cannot read `/proc` attributes nothing and the machine loses outbound traffic | | `NoNewPrivileges` | No regaining privileges via setuid binaries | | `SystemCallFilter=@system-service` | seccomp; the biggest blast-radius reduction available | | `SystemCallFilter=bpf perf_event_open` | The two syscalls the eBPF layer needs, named individually | diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 0ba6d03..6471335 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -3,9 +3,9 @@ Tracking the port from opensnitch (Go daemon + Python Qt UI) to Rust. Phases 0-3 are done and have since been through a hardening pass -(Phase 3.5). The eBPF backend, the system tray and the whitelist fast -path (opt-in, `[ebpf] fast_allow`; its latency win is still to be -measured, see `TODO.md` 1a) have since landed too. What is left is +(Phase 3.5). The eBPF backend and the system tray have since landed +too; the whitelist fast path landed and was later removed (see +`TODO.md` 1a). What is left is VirusTotal lookups, publishing to the AUR, and one end-to-end test that is still manual. @@ -165,16 +165,10 @@ kernel 7.1.8. 1,000,000-instruction complexity limit (see `crates/cfc-ebpf/README.md` for the full write-up) - [x] Whitelist fast path for already-allowed flows (`[ebpf] - fast_allow`). Not the shape first imagined: a cgroup *egress* - hook runs after NF_INET_LOCAL_OUT and cannot short-circuit - NFQUEUE, but the `connect()` hook runs before any packet exists. - A process a lasting Allow rule covers gets an entry in a kernel - map; its TCP connects are marked with SO_MARK at connect time - and `meta mark @fast_allow accept` takes them before the queue - rule. TCP only, grants evicted on exec and exit, the whole path - bounded by a heartbeat deadline so a dead daemon strips itself - out. Two degradations shorten the deadline instead of turning - the feature off; see `docs/ARCHITECTURE.md` + fast_allow`): removed. Allowed TCP connects were marked with + SO_MARK at connect time and accepted ahead of the queue rule, and + a socket mark does not attest which process sends, so it opened + bypasses; see `TODO.md` 1a ## Phase 5 - Polish diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index fd5093f..16a3f6c 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -19,9 +19,10 @@ run `systemctl daemon-reload`, then `systemctl reenable colony-firewalld colony-firewall-nft` (and the inbound unit only if already enabled), and `systemctl reload colony-firewall-nft` (and the inbound unit if active) before relying on the new rules. Reenable installs the native network-manager -requirements on existing deployments. A startup error saying -previous Fast Allow state could not be disabled means old acceptance may still -exist; resolve that error and inspect the loaded table. The nft units load +requirements on existing deployments. A startup error saying the +legacy fast_allow nftables set could not be flushed means a mark left by an +older release may still be accepted; resolve that error and inspect the loaded +table. The nft units load before the daemon. Failed daemon initialization leaves filtering installed; a failed nft load blocks the daemon and the enabled NetworkManager or systemd-networkd requirements. This covers those managers' startup after diff --git a/packaging/rpm/colony-firewall-control.spec b/packaging/rpm/colony-firewall-control.spec index 51504b7..5acf394 100644 --- a/packaging/rpm/colony-firewall-control.spec +++ b/packaging/rpm/colony-firewall-control.spec @@ -59,7 +59,7 @@ Install one - `cargo xtask build-ebpf`, dropped at /usr/lib/colony-firewall/cfc-ebpf.o, in the directory this package creates for it - and the daemon adds process attribution, DNS display enrichment, and in-kernel connect(2) denial. Pinned denials survive a daemon crash. Fast Allow -is disabled; allowed connections continue through NFQUEUE. +was removed; allowed connections go through NFQUEUE. The ruleset is fail-closed. If the daemon is not running, new outbound connections are dropped rather than allowed. @@ -79,7 +79,7 @@ SELinux policy module for Colony Firewall Control. Confines the daemon to what it actually needs: netlink_netfilter and raw sockets, bpf() and perf_event_open(), the bpffs pin directory, other domains' /proc entries for attribution, a read-only rpm query for package provenance, -and running nft(8) to clear Fast Allow state left by older installations. CAP_SYS_ADMIN is deliberately not granted; that it is +and running nft(8) to probe the table and flush the legacy Fast Allow set. CAP_SYS_ADMIN is deliberately not granted; that it is unnecessary is covered by a test rather than assumed. %prep diff --git a/packaging/selinux/TESTING.md b/packaging/selinux/TESTING.md index fff726f..42ef5bc 100644 --- a/packaging/selinux/TESTING.md +++ b/packaging/selinux/TESTING.md @@ -109,7 +109,6 @@ chance to be needed. | bpf/perf ring 0 | **degraded by design in the RPM**: no eBPF object ships (see the spec's `%build` comment), so the journal says `ring0=unavailable degrade=object_missing` once at startup and the bpf/perf/bpffs/tracefs rules are never reached. That log line *is* the expected result. To exercise the group for real: build the object (`cargo xtask build-ebpf`, pinned nightly + bpf-linker), drop it at `/usr/lib/colony-firewall/cfc-ebpf.o`, restart, and expect `ring0=active` | with the object installed: `degrade=not_permitted` where `object_missing` was, and AVCs on `bpf`, `perf_event`, `tracefs_t`/`debugfs_t` or `bpf_t` | | /proc attribution walk | `curl` from a second user account; the prompt must name curl's real path and pid | every prompt says `exe= pid=0`; AVCs from `domain_read_all_domains_state` targets. Enforcing, this is the outage mode: no exe rule can ever match | | rpm provenance | automatic: one `rpm -qa` at startup and after any `dnf install`. Install any small package, wait ~2 minutes, then check a prompt or `cfc log` shows package names | everything reports `Unpackaged` plus one provenance warning in the journal; AVC on `rpm_exec_t` or `rpm_var_lib_t` | -| nft, the fast-allow set | needs ring 0 up (the object installed as in the bpf/perf row) and `[ebpf] fast_allow = true`, then both units running. Fast-allow armed: `cfc status` shows fast_allow live, `sudo nft list set inet colony_firewall fast_allow` shows one element. Then set `fast_allow = false` and `systemctl restart colony-firewalld`: the set is empty while the table is still loaded, which is the unconditional start-up flush. (Do **not** test this by stopping the daemon - `colony-firewall-nft.service` is `PartOf=` it and tears the whole table down first, so `list set` answers "No such file or directory" and tells you nothing about the flush.) | `cfc status` shows fast_allow off with an nft error as the reason, and the set stays empty; AVC on `iptables_exec_t` (execute) - the `netlink_netfilter_socket` nft needs is the filtering group's, already exercised by the first row | | control socket, unconfined client | `cfc status` and `cfc rules list` as a normal logged-in user in the `colony-firewall` group (not root, not sudo) | connection refused/denied; AVC with the client's domain (`unconfined_t`) and `colony_firewall_runtime_t` | | sqlite WAL in /var/lib | answer any prompt with a persistent choice (**a**, then `3`=always), then `ls /var/lib/colony-firewall/` - `rules.db-wal` and `rules.db-shm` must exist while the daemon runs | the `map` denial is the quiet one: no error anywhere, just journal-mode SQLite and a 2.5x write regression. An AVC with class `file` permission `map` on `colony_firewall_var_lib_t` is the tell | diff --git a/packaging/selinux/colony_firewall.te b/packaging/selinux/colony_firewall.te index 672db78..3a6c05e 100644 --- a/packaging/selinux/colony_firewall.te +++ b/packaging/selinux/colony_firewall.te @@ -21,8 +21,8 @@ policy_module(colony_firewall, 0.2.0) # resolve struct offsets without CO-RE; # * walks other processes' /proc entries to attribute a socket to a program; # * runs rpm(8) once per package-database generation, for provenance; -# * runs nft(8) at start and at stop, to put its fast-allow mark into one -# nftables set and take it out again; +# * runs nft(8) once at start, to flush a legacy Fast Allow set, and once a +# minute, to ask whether the filtering table is loaded; # * serves a unix socket under /run/colony-firewall to the CLI, tray and GUI. # # Every one of those is a separate way to be denied, and the failure modes @@ -190,28 +190,17 @@ optional_policy(` # check on the same file rather than the only one. libs_read_lib_files(colony_firewalld_t) -# The fast-allow path. The connect hooks mark the sockets of a process the -# daemon has ruled allowed process-wide, and the shipped snippet accepts that -# mark ahead of the queue - but only for values in a set the snippet declares -# empty. The daemon runs nft(8) to put its per-start random value in and to -# flush the set again, executed in this domain like rpm below rather than -# transitioning to iptables_t, which may rewrite the whole ruleset. nft then -# opens a netlink_netfilter socket, already allowed in the filtering group -# above: it is the socket class NFQUEUE itself lives on. -# -# Five subcommands, not the two an earlier version of this comment claimed: -# `flush set` and `add element` do the work, `list set` and `list table` are -# the probe that tells a missing set from a missing table, and `get element` -# is the periodic check that the mark is still accepted after a ruleset -# reload. Nor is it twice in a daemon's life: the nft unit starts *after* the -# daemon, so arming is a retry on every heartbeat - ten seconds, or two on a -# kernel that cannot pin the exec/exit tracepoint links - until the table -# appears, and once armed the check runs every sixty. On a host where this is -# denied that is an AVC every heartbeat, for as long as the daemon runs with -# fast_allow = true - loud on purpose, but worth knowing before reading a log. -# -# A denial costs the fast path and nothing else. The daemon reports it as off -# with the reason in `cfc status`, and every connection keeps taking the queue. +# nft(8), for two things. Once at start, `flush set` empties the legacy +# fast_allow set: Fast Allow was removed, and a 0.4-0.6 daemon that crashed +# while armed can have left its mark accepted there. Once a minute, `list +# table` tells `cfc status` whether the filtering table is loaded. Executed in +# this domain like rpm below rather than transitioning to iptables_t, which may +# rewrite the whole ruleset. nft then opens a netlink_netfilter socket, already +# allowed in the filtering group above: it is the socket class NFQUEUE itself +# lives on. +# +# A denial costs the flush (an error in the journal at start) and the probe +# (`enforcing` keeps its previous answer); filtering itself is unaffected. # # Fedora labels /usr/bin/nft iptables_exec_t (RHEL: /usr/sbin/nft); the same # type covers both spellings, and the daemon tries both paths. diff --git a/scripts/bench-latency.sh b/scripts/bench-latency.sh index a3a2d15..4e8b20f 100755 --- a/scripts/bench-latency.sh +++ b/scripts/bench-latency.sh @@ -10,9 +10,8 @@ # # Two directions, because they do not take the same path: # out host -> namespace. Every SYN leaves through the host's output chain, -# where `inet colony_firewall` queues `ct state new` to the daemon - -# unless the client is fast-allowed, in which case its mark takes it -# past the queue. This is the direction the fast path exists for. +# where `inet colony_firewall` queues `ct state new` to the daemon. +# This is the direction that meets the queue. # in namespace -> host. The SYN arrives on the host's input path, which # `inet colony_firewall_inbound` filters only where that opt-in unit # is loaded. With it absent, this direction never meets a queue - but @@ -27,10 +26,9 @@ # from 10.199.0.0/24 there, or read `in` as unavailable on that host. # # The script never touches nftables, the daemon or its rules. It measures the -# machine as it finds it, prints what `cfc status` says the fast path is, and -# leaves the comparison to whoever runs it more than once: with the client -# covered by a lasting Allow rule (fast path), by a flow-scoped one (queue), -# and with the table absent (nothing). The client CFC attributes is python3, +# machine as it finds it, and leaves the comparison to whoever runs it more +# than once: with the client covered by an Allow rule (queue) and with the +# table absent (nothing). The client CFC attributes is python3, # so the rule to write is for python3's resolved path (`readlink -f # "$(command -v python3)"`); the first connect of a run is the one that prompts. # @@ -44,11 +42,11 @@ # SQLite bench, and the next one that writes should remember it. # # Needs root - it creates a namespace and a veth pair - plus iproute2 and -# python3. A VM is the right place: the point of measuring is to arm the fast -# path, and arming a firewall on a development host has consequences. +# python3. A VM is the right place: the point of measuring is to arm the +# firewall, and arming one on a development host has consequences. # # sudo scripts/bench-latency.sh both directions, 200 connects -# sudo scripts/bench-latency.sh -n 1000 -d out -l "fast path live" +# sudo scripts/bench-latency.sh -n 1000 -d out -l "queue armed" # sudo scripts/bench-latency.sh --json >> runs.jsonl one JSON object per direction set -euo pipefail @@ -245,19 +243,6 @@ wait_listening "" "$PORT" # each probe is allowed to fail: the bench is also how one measures a machine # with no CFC on it at all. KERNEL="$(uname -r)" -FAST_ALLOW="cfc not installed" -if command -v cfc >/dev/null 2>&1; then - # cfc exits non-zero when the daemon is down; under pipefail that failed - # the whole pipeline after python had already printed, and the `|| echo` - # that used to follow printed the same words a second time. - FAST_ALLOW="$( (cfc status --json 2>/dev/null || true) | python3 -c ' -import json, sys -try: - print(json.load(sys.stdin).get("fast_allow", "not reported")) -except Exception: - print("daemon not reachable") -')" -fi TABLES="nft not installed" if command -v nft >/dev/null 2>&1; then # `|| true` on the whole pipeline: a machine with no colony table makes @@ -269,7 +254,6 @@ if command -v nft >/dev/null 2>&1; then fi { echo "kernel: $KERNEL" - echo "fast-allow: $FAST_ALLOW" echo "nft tables: $TABLES" echo "link: $HOST_IF ($HOST_IP) <-> $NS:$NS_IF ($NS_IP), port $PORT" echo "per run: $COUNT connects after $WARMUP warm-up, ${TIMEOUT}s timeout each" @@ -281,13 +265,13 @@ run_direction() { # run_direction local dir="$1" ns="" target="$NS_IP" raw if [[ "$dir" == in ]]; then ns="$NS"; target="$HOST_IP"; fi raw="$(in_ns "$ns" python3 -c "$CLIENT" "$target" "$PORT" "$COUNT" "$TIMEOUT" "$WARMUP")" - python3 - "$raw" "$dir" "$LABEL" "$KERNEL" "$FAST_ALLOW" "$COUNT" "$JSON" "$PORT" <<'PY' + python3 - "$raw" "$dir" "$LABEL" "$KERNEL" "$COUNT" "$JSON" "$PORT" <<'PY' import json, sys r = json.loads(sys.argv[1]) -port_num = int(sys.argv[8]) +port_num = int(sys.argv[7]) r.update(direction=sys.argv[2], label=sys.argv[3], kernel=sys.argv[4], - fast_allow=sys.argv[5], connects=int(sys.argv[6])) -if sys.argv[7] == "1": + connects=int(sys.argv[5])) +if sys.argv[6] == "1": print(json.dumps(r)) sys.exit(0) f = r["failed"] diff --git a/scripts/vm-bench/README.md b/scripts/vm-bench/README.md index b5436ce..d275824 100644 --- a/scripts/vm-bench/README.md +++ b/scripts/vm-bench/README.md @@ -2,8 +2,7 @@ `scripts/bench-latency.sh` measures connect latency over a veth pair. It answers nothing on its own, because the interesting comparison needs CFC -*armed* - the queue rule loaded, the daemon deciding, the fast path granting - -and arming a fail-closed firewall on a development machine has consequences. +*armed* - the queue rule loaded, the daemon deciding - and arming a fail-closed firewall on a development machine has consequences. This directory boots a throwaway VM instead. It assembles an initramfs from the host's own kernel modules, `nftables`, `iproute2`, `python3` and the release @@ -28,9 +27,8 @@ isolates one cost. | state | what is running | what the difference against the previous one buys | |---|---|---| | `floor` | nothing: no daemon, no table | the veth link and `connect()` itself | -| `queue-N` | the daemon, the table, a lasting Allow, `fast_allow = false` | the NFQUEUE round trip, at N flows | +| `queue-N` | the daemon, the table, a lasting Allow | the NFQUEUE round trip, at N flows | | `poll200us-N` | the same, with a daemon built with a shorter `RECV_POLL_INTERVAL` | how much of that round trip is the worker's idle beat | -| `fast-N` | `fast_allow = true`, the client covered by a lasting Allow | the fast path against the queue | Both directions run in every state and they answer different questions. `out` leaves through the host's output chain and meets the queue. `in` is generated @@ -38,12 +36,10 @@ inside the network namespace, whose own output chain carries no colony table, so it never meets a queue - but its client sits in the root cgroup and still runs the connect hooks, which makes `in` the cost of the eBPF layer alone. -Two things are recorded beside every measurement rather than assumed: -`cfc status`'s own account of the fast path, and `id_sequence` from -`/proc/net/netfilter/nfnetlink_queue` - one increment per packet the kernel -actually handed to userspace. A state calling itself `fast` whose queue saw one -packet per connect did not take the fast path, and no latency figure says that -on its own. +One thing is recorded beside every measurement rather than assumed: +`id_sequence` from `/proc/net/netfilter/nfnetlink_queue` - one increment per +packet the kernel actually handed to userspace. A state whose queue saw no +packets did not measure the queue, and no latency figure says that on its own. `ALT_DAEMON=/path/to/colony-firewalld` carries a second daemon into the same image, measured in the same boot under conditions that differ in nothing else. @@ -57,15 +53,13 @@ Run on 2026-09-06, Linux 7.2.2, KVM, four vCPUs, 3000 flows unless said. | state | 300 flows | 3000 flows | |---|---|---| | no firewall | 0.0158 ms | 0.0162 ms | -| fast path | 0.0268 ms | 0.0269 ms | | queue, 200 us idle beat | 0.7703 ms | 2.3646 ms | | queue, the shipped 5 ms beat | 5.6745 ms | 7.6083 ms | -Read across, and three things fall out. +That run also measured Fast Allow, since removed because a socket mark does +not prove which process sends (see `TODO.md` 1a). Read across, and two things +fall out. -- **The fast path saves 5.6 ms per new flow at 300 flows and 7.6 ms at 3000**, - and costs 0.011 ms over having no firewall at all. Its own cost does not grow - with load, because those flows never reach the daemon. - **A full `RECV_POLL_INTERVAL` is paid per queued flow, not half of one.** `crates/cfc-daemon/src/nfqueue.rs` predicts "up to one interval (mean: half that)", which is right for random arrivals and wrong for a client that diff --git a/scripts/vm-bench/init b/scripts/vm-bench/init index 66672f2..5953a74 100755 --- a/scripts/vm-bench/init +++ b/scripts/vm-bench/init @@ -13,9 +13,8 @@ mount -t sysfs sysfs /sys mount -t tmpfs tmpfs /tmp mount -t tmpfs tmpfs /run mount -t cgroup2 cgroup2 /sys/fs/cgroup -# bpffs, unlike the CI guests: with it the exec/exit links pin, which is what -# lets the fast path run with its full sixty-second deadline rather than the -# shortened one. The measurement should see the feature as a host sees it. +# bpffs, unlike the CI guests: with it the connect and exec/exit links pin, +# so the measurement sees the in-kernel layer as a host sees it. mount -t bpf bpf /sys/fs/bpf mount -t tracefs tracefs /sys/kernel/tracing 2>/dev/null diff --git a/scripts/vm-bench/plan.sh b/scripts/vm-bench/plan.sh index 58917e8..618bc67 100755 --- a/scripts/vm-bench/plan.sh +++ b/scripts/vm-bench/plan.sh @@ -2,10 +2,9 @@ # Why does a queued flow cost what it costs, and does that cost depend on load? # # The first full run said 17.8 ms per queued flow at 3000 flows and 5.5 ms at -# 40, with the two rounds 15.0 and 20.7 ms apart - and the state that does -# strictly MORE work (`armed`: the fast path live but the rule ineligible) came -# out faster than the one that does less. None of that is a per-packet -# constant. Two candidate explanations, and this run separates them: +# 40, with the two rounds 15.0 and 20.7 ms apart - and a state that did +# strictly MORE work came out faster than one that did less. None of that is a +# per-packet constant. Two candidate explanations, and this run separates them: # # 1. a fixed cost per flow, dominated by RECV_POLL_INTERVAL (5 ms), the beat # the NFQUEUE worker idles on. Testable by changing the constant: the same @@ -56,7 +55,6 @@ enabled = false [ebpf] enabled = "on" object_path = "/cfc-ebpf.o" -fast_allow = $1 EOF } @@ -84,31 +82,30 @@ stop_daemon() { probe_layer() { say "what the in-kernel layer comes up as here" - write_cfg true + write_cfg start_daemon /usr/bin/colony-firewalld info || return 1 nft -f "$SNIPPET"; write_rules cfc --socket "$SOCK" rules import --replace /tmp/rules.json >/dev/null 2>&1 sleep 4 - for k in ring0 enforcement degrade fast_path exec_tracking exit_tracking dns_capture ppid_from_btf; do + for k in ring0 enforcement degrade exec_tracking exit_tracking dns_capture ppid_from_btf; do v="$(grep -oE "$k=[A-Za-z_-]+" "$LOG" | tail -1)" [ -n "$v" ] && ctx "layer $v" done - ctx "layer $(cfc --socket "$SOCK" status --json | python3 -c 'import json,sys; d=json.load(sys.stdin); print("status_fast_allow=%s status_enforcing=%s" % (d["fast_allow"], d["enforcing"]))')" + ctx "layer $(cfc --socket "$SOCK" status --json | python3 -c 'import json,sys; d=json.load(sys.stdin); print("status_enforcing=%s" % d["enforcing"])')" stop_daemon } -measure() { # $1 label $2 n $3 mode(none|queue|fast) $4 binary +measure() { # $1 label $2 n $3 mode(none|queue) $4 binary local label="$1" n="$2" mode="$3" bin="${4:-/usr/bin/colony-firewalld}" q0 q1 say "state: $label n=$n mode=$mode daemon=$(basename "$bin")" if [ "$mode" != none ]; then [ -x "$bin" ] || { echo "SKIP $label: $bin is not in this image"; return 0; } - if [ "$mode" = fast ]; then write_cfg true; else write_cfg false; fi + write_cfg start_daemon "$bin" || { echo "FAIL $label"; return 1; } nft -f "$SNIPPET" || { echo "FAIL $label nft"; stop_daemon; return 1; } write_rules cfc --socket "$SOCK" rules import --replace /tmp/rules.json >/dev/null 2>&1 sleep 4 - ctx "$label fast_allow=$(cfc --socket "$SOCK" status --json 2>/dev/null | python3 -c 'import json,sys; print(json.load(sys.stdin).get("fast_allow","?"))' 2>/dev/null || echo unreachable)" fi ctx "$label before sockets=$(sockets) conntrack=$(ctcount)" q0="$(qseq)" @@ -137,8 +134,5 @@ done for n in "$SMALL" "$LARGE"; do drain; measure "poll200us-$n" "$n" queue /usr/bin/colony-firewalld-alt done -for n in "$SMALL" "$LARGE"; do - drain; measure "fast-$n" "$n" fast /usr/bin/colony-firewalld -done [ "$SMALL" != "$LARGE" ] && { drain; measure "floor-$LARGE" "$LARGE" none; } say "done" diff --git a/scripts/vm-bench/report.py b/scripts/vm-bench/report.py index 65a7729..75d7937 100755 --- a/scripts/vm-bench/report.py +++ b/scripts/vm-bench/report.py @@ -56,7 +56,7 @@ def counts(prefix): return sorted(int(k.rsplit("-", 1)[1]) for k in out if k.rsplit("-", 1)[0] == prefix and k.rsplit("-", 1)[1].isdigit()) - q, f, fl, po = (counts(x) for x in ("queue", "fast", "floor", "poll200us")) + q, fl, po = (counts(x) for x in ("queue", "floor", "poll200us")) print("\nreadings (p50 of the `out` direction, the one that meets the queue)") if len(q) > 1: pair(f"queue-{q[-1]}", f"queue-{q[0]}", @@ -64,10 +64,6 @@ def counts(prefix): for n in po: pair(f"poll200us-{n}", f"queue-{n}", f"{n} flows: a 200us idle beat against the 5ms one") - for n in f: - pair(f"fast-{n}", f"queue-{n}", f"{n} flows: the fast path against the queue") - for n in sorted(set(f) & set(fl)): - pair(f"fast-{n}", f"floor-{n}", f"{n} flows: what the fast path costs over nothing") if len(fl) > 1: pair(f"floor-{fl[-1]}", f"floor-{fl[0]}", f"the floor itself, {fl[-1]} flows against {fl[0]}") diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index b0dca18..2e1ead4 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -210,10 +210,11 @@ MemoryDenyWriteExecute=true # else about this unit is compatible with that child - ProtectSystem=strict # leaves /var/lib/rpm readable, and rpm needs no writable-executable memory. If # it ever is not compatible, the failure is a warning and an empty provenance -# index, never a failure to filter. The other child is nft(8), run at start -# and at stop to flush legacy Fast Allow state. The same fork/execve and -# CAP_NET_ADMIN netlink operations are needed. Failed cleanup is reported -# explicitly because an older installed ruleset may still accept old marks. +# index, never a failure to filter. The other child is nft(8), run once at +# start to flush the legacy Fast Allow set and once a minute to probe the +# table. The same fork/execve and CAP_NET_ADMIN netlink operations are needed. +# A failed flush is reported explicitly because an older installed ruleset +# may still accept old marks. SystemCallFilter=@system-service SystemCallFilter=bpf perf_event_open SystemCallErrorNumber=EPERM diff --git a/systemd/daemon.toml.sample b/systemd/daemon.toml.sample index 46a0a88..8b5326e 100644 --- a/systemd/daemon.toml.sample +++ b/systemd/daemon.toml.sample @@ -206,10 +206,9 @@ enabled = true # # # # WHAT IT DOES NOT DO # # A process-wide Deny can refuse connect() in the kernel. Allow decisions and -# # conditional rules still use NFQUEUE. Fast Allow is disabled in every -# # configuration because a socket mark cannot identify the current sender. -# # Neither this layer nor NFQUEUE confines inherited sockets, local relays or -# # packet-layer traffic with CAP_NET_RAW. +# # conditional rules still use NFQUEUE. Neither this layer nor NFQUEUE +# # confines inherited sockets, local relays or packet-layer traffic with +# # CAP_NET_RAW. # # # # REQUIREMENTS - all of which degrade to a warning, never to a failure to # # start. If any of this is missing the daemon logs what it could not do and @@ -238,12 +237,8 @@ enabled = true # # /usr/lib/colony-firewall/cfc-ebpf.o. Point it at # # `cargo xtask ebpf-path` output when working on the programs themselves. # object_path = "/usr/lib/colony-firewall/cfc-ebpf.o" -# # Compatibility settings retained for existing configuration files. -# # Fast Allow is disabled. Setting true logs a warning and creates no grants; -# # allowed connections continue through NFQUEUE. Keep this false. -# fast_allow = false -# # An old configured mark has no runtime effect while Fast Allow is disabled. -# fast_allow_mark = 0x00033331 +# # fast_allow and fast_allow_mark are legacy keys from the removed Fast Allow +# # path: ignored, and a warning is logged if either is set. Delete them. # Needs a restart (the socket is secured once, right after bind). [ipc] From 784772a92db82a9060b8a60bfbb910c7c5c5db3b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:10:30 +0200 Subject: [PATCH 005/125] test(ebpf): keep the legacy disarm fixture inert The root test arms the pinned FAST_ALLOW_MARK to prove the next load disarms it, while the inherited connect hooks are still attached. Use a deadline already in the past and a grant for a pid no process can hold, so nothing on the test host is marked in that window. --- crates/cfc-daemon/src/ebpf/loader.rs | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index b5654f7..a88cc90 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -1769,6 +1769,9 @@ mod tests { // The legacy disarm, on the path that needs it: a pinned MARK left // armed by a 0.4-0.6 daemon that died, met by a restart on the // inherited path with no engine. Only a bpffs host can show this. + // The fixture is inert while it sits there: a deadline already in + // the past (the hooks honour nothing) and a grant for a pid no + // process can hold. { let dir = enforce::pin_dir(); let mark = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW_MARK)) @@ -1780,12 +1783,12 @@ mod tests { .expect("reopen the pinned FAST_ALLOW_UNTIL"); let mut until = aya::maps::Array::<_, u64>::try_from(aya::maps::Map::Array(until)) .expect("FAST_ALLOW_UNTIL is an array"); - until.set(0, u64::MAX, 0).expect("set a legacy deadline"); + until.set(0, 1, 0).expect("set a lapsed legacy deadline"); let grants = MapData::from_pin(dir.join(enforce::MAP_FAST_ALLOW)) .expect("reopen the pinned FAST_ALLOW"); let mut grants = BpfHashMap::<_, u32, u32>::try_from(aya::maps::Map::HashMap(grants)) .expect("FAST_ALLOW is a hash map"); - grants.insert(1, 1, 0).expect("leave a legacy grant"); + grants.insert(u32::MAX, 1, 0).expect("leave a legacy grant"); } let (attached, report) = load_and_attach( Path::new(&path), From a6ea048cbeaca31228f0e69f36ef4325cb2f35ac Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:14:58 +0200 Subject: [PATCH 006/125] ci: add an armed e2e job with real NFQUEUE verdicts The smoke job drives a --dry-run daemon, so it never proves a packet is dropped. The new e2e workflow loads the shipped nftables snippet inside a throwaway network namespace, runs colony-firewalld there and checks with curl that an allow rule passes, a deny rule and an unmatched flow are silently dropped, the audit log names the deciding rule, rules persist across a restart, and new flows stay dropped after SIGTERM and SIGKILL. Nothing outside the namespace is filtered. --- .github/workflows/e2e.yml | 37 ++++++++ scripts/armed-e2e.sh | 189 ++++++++++++++++++++++++++++++++++++++ 2 files changed, 226 insertions(+) create mode 100644 .github/workflows/e2e.yml create mode 100755 scripts/armed-e2e.sh diff --git a/.github/workflows/e2e.yml b/.github/workflows/e2e.yml new file mode 100644 index 0000000..28b4e82 --- /dev/null +++ b/.github/workflows/e2e.yml @@ -0,0 +1,37 @@ +name: e2e + +on: + push: + branches: [main] + pull_request: + branches: [main] + +permissions: + contents: read + +env: + CARGO_TERM_COLOR: always + +jobs: + armed-e2e: + name: armed e2e (real NFQUEUE verdicts in a netns) + runs-on: ubuntu-latest + timeout-minutes: 30 + steps: + - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 + with: + persist-credentials: false + - uses: dtolnay/rust-toolchain@02cb101ec7c40f2c49e1d9714d64511d8e1b74de # master + with: + toolchain: stable + - name: Install build and test deps + run: | + sudo apt-get update + sudo apt-get install -y --no-install-recommends \ + protobuf-compiler libnfnetlink-dev libnetfilter-queue-dev nftables jq + - uses: Swatinem/rust-cache@6323deb102c322ba6fcbdcafc7e3dddab59af2b6 # v2.9.2 + with: + key: armed-e2e + - name: Armed e2e (build, netns, shipped nft snippet, daemon, curl) + timeout-minutes: 25 + run: ./scripts/armed-e2e.sh diff --git a/scripts/armed-e2e.sh b/scripts/armed-e2e.sh new file mode 100755 index 0000000..e963acc --- /dev/null +++ b/scripts/armed-e2e.sh @@ -0,0 +1,189 @@ +#!/usr/bin/env bash +# Armed end-to-end test: real NFQUEUE verdicts, not --dry-run. +# +# Everything filtered lives in a throwaway network namespace, so the host's +# own traffic (the CI runner's connection to GitHub, an SSH session) never +# meets the fail-closed table. Two namespaces joined by a veth pair: +# FW the shipped nftables snippet, colony-firewalld and curl +# SRV three HTTP servers, no filtering +# Needs sudo (passwordless on GitHub runners). Never runs nft outside FW. + +set -euo pipefail + +ROOT="$(cd "$(dirname "$0")/.." && pwd)" +DAEMON="${ROOT}/target/debug/colony-firewalld" +CFC="${ROOT}/target/debug/cfc" +FW="cfc-e2e-fw-$$" +SRV="cfc-e2e-srv-$$" +SRV_IP=10.200.0.2 +ALLOW_PORT=8080 # allow rule +DENY_PORT=8081 # deny rule +UNMATCHED_PORT=8082 # no rule: balanced profile, nobody subscribed -> Deny +W="$(mktemp -d "${RUNNER_TEMP:-/tmp}/cfc-e2e.XXXXXX")" +SOCK="${W}/cfc.sock" + +fail() { echo "FAIL: $*" >&2; exit 1; } +say() { printf '\n=== %s ===\n' "$*"; } +in_fw() { sudo ip netns exec "${FW}" "$@"; } +in_srv() { sudo ip netns exec "${SRV}" "$@"; } +cfc() { sudo "${CFC}" --socket "${SOCK}" "$@"; } +table_loaded() { in_fw nft list table inet colony_firewall >/dev/null; } +# Kernel truth, not a log line: is anything bound to NFQUEUE 0 in FW? +queue_bound() { + in_fw awk '$1 == 0 { b = 1 } END { exit !b }' \ + /proc/net/netfilter/nfnetlink_queue 2>/dev/null +} + +# curl from inside FW as the unprivileged runner user. Prints the HTTP code +# and returns curl's exit status. +probe() { + in_fw setpriv --reuid="$(id -u)" --regid="$(id -g)" --clear-groups \ + curl -sS --noproxy '*' -o /dev/null -w '%{http_code}' \ + --connect-timeout 3 --max-time 6 "http://${SRV_IP}:$1/" +} +expect_200() { + local code + code="$(probe "$1")" || fail "port $1: curl failed, expected HTTP 200" + [[ "${code}" == 200 ]] || fail "port $1: got HTTP ${code}, expected 200" +} +# 28 = connect timeout: the SYN vanished. 7 (refused) would mean an RST came +# back, i.e. something answered instead of dropping. +expect_drop() { + local rc=0 + probe "$1" >/dev/null 2>&1 || rc=$? + [[ "${rc}" -eq 28 ]] || fail "port $1: curl exit ${rc}, expected 28 (silently dropped)" +} + +start_daemon() { + # The inner sh writes its own PID and then execs the daemon, so the file + # names the daemon itself (pkill -x cannot: comm is cut to 15 characters). + # shellcheck disable=SC2016 # $$, $0 and $@ belong to the inner sh + in_fw sh -c 'echo $$ >"$0"; exec "$@"' "${W}/daemon.pid" \ + "${DAEMON}" --debug --config "${W}/daemon.toml" --socket "${SOCK}" \ + >"${W}/$1.log" 2>&1 & + DAEMON_JOB=$! + for _ in $(seq 1 150); do + queue_bound && cfc status >/dev/null 2>&1 && return 0 + sleep 0.2 + done + fail "daemon did not bind NFQUEUE 0 and its socket within 30s" +} +stop_daemon() { + local pid + pid="$(sudo cat "${W}/daemon.pid")" + sudo kill -"$1" "${pid}" + for _ in $(seq 1 100); do + sudo kill -0 "${pid}" 2>/dev/null || break + sleep 0.2 + done + if sudo kill -0 "${pid}" 2>/dev/null; then fail "daemon survived SIG$1 for 20s"; fi + wait "${DAEMON_JOB}" 2>/dev/null || true + sudo rm -f "${W}/daemon.pid" +} + +cleanup() { + local rc=$? + if sudo test -s "${W}/daemon.pid"; then + sudo kill -KILL "$(sudo cat "${W}/daemon.pid")" 2>/dev/null || true + fi + sudo ip netns pids "${SRV}" 2>/dev/null | xargs -r sudo kill 2>/dev/null || true + sudo ip netns del "${FW}" 2>/dev/null || true + sudo ip netns del "${SRV}" 2>/dev/null || true + if [[ "${rc}" -ne 0 ]]; then + for f in "${W}"/*.log; do + [[ -e "${f}" ]] || continue + echo "--- ${f} (last 200 lines)" + tail -n 200 "${f}" + done + fi + sudo rm -rf "${W}" +} +trap cleanup EXIT + +sudo -v +command -v jq >/dev/null || fail "jq is required" + +say "Building colony-firewalld (no eBPF) and cfc" +cargo build --locked -p cfc-daemon -p cfc-cli --no-default-features + +say "Namespaces, veth pair, HTTP servers" +sudo modprobe -a nfnetlink_queue nft_queue +sudo ip netns add "${FW}" +sudo ip netns add "${SRV}" +sudo ip link add fw0 netns "${FW}" type veth peer name srv0 netns "${SRV}" +in_fw ip addr add 10.200.0.1/24 dev fw0 +in_fw ip link set fw0 up +in_srv ip addr add "${SRV_IP}/24" dev srv0 +in_srv ip link set srv0 up +# FW's lo stays down on purpose: no loopback flows (the daemon's own reverse +# DNS to a 127.0.0.53 stub fails fast instead of being queued). +mkdir "${W}/www" +for p in "${ALLOW_PORT}" "${DENY_PORT}" "${UNMATCHED_PORT}"; do + in_srv python3 -m http.server "${p}" --bind "${SRV_IP}" \ + --directory "${W}/www" >"${W}/http-${p}.log" 2>&1 & +done + +say "Control: every port answers before the firewall exists" +for p in "${ALLOW_PORT}" "${DENY_PORT}" "${UNMATCHED_PORT}"; do + for _ in $(seq 1 50); do + [[ "$(probe "${p}" 2>/dev/null)" == 200 ]] && break + sleep 0.2 + done + expect_200 "${p}" +done + +say "Load the shipped snippet in FW (as README: nft -f from a checkout)" +in_fw nft -f "${ROOT}/systemd/nftables-snippet.conf" +table_loaded || fail "table inet colony_firewall not loaded in ${FW}" +queue_bound && fail "something is already bound to NFQUEUE 0" + +say "Fail-closed before the daemon ever started" +expect_drop "${ALLOW_PORT}" + +cat >"${W}/daemon.toml" </dev/null \ + || fail "no deny event on port ${DENY_PORT} attributed to rule ${DENY_ID}" +echo "${DENIES}" | jq -e --argjson p "${UNMATCHED_PORT}" \ + 'any(.[]; .dst_port == $p and .rule_id == null)' >/dev/null \ + || fail "no default-deny event on port ${UNMATCHED_PORT}" + +say "SIGTERM: clean stop keeps the table, new flows drop" +stop_daemon TERM +table_loaded || fail "a clean daemon stop removed the table" +queue_bound && fail "NFQUEUE 0 still bound after the daemon exited" +expect_drop "${ALLOW_PORT}" + +say "Restart: rules persisted" +start_daemon daemon-2 +expect_200 "${ALLOW_PORT}" +expect_drop "${DENY_PORT}" + +say "SIGKILL: crash keeps the table, new flows drop" +stop_daemon KILL +table_loaded || fail "table gone after SIGKILL" +queue_bound && fail "NFQUEUE 0 still bound after SIGKILL" +expect_drop "${ALLOW_PORT}" + +say "Armed e2e passed" From 144a2e4ef6dae54000cd1f7a43d1988cd410ba95 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:17:28 +0200 Subject: [PATCH 007/125] chore: refresh project status docs and CI hygiene SECURITY.md now says beta, states plainly that there has been no external security audit, and marks explicit application confinement as experimental, as does its README section. The README crate table lists all ten crates, including cfc-ebpf-common, cfc-ebpf and xtask. release.yml passes github.ref_name to run scripts through env instead of interpolating it, and drops the AUR-submittable wording: the project is not published on the AUR. Dependabot groups cargo minor+patch and GitHub Actions updates, and covers transitive cargo dependencies. --- .github/dependabot.yml | 11 +++++++++++ .github/workflows/release.yml | 30 ++++++++++++++++++------------ README.md | 30 +++++++++++++++++++----------- SECURITY.md | 10 +++++++--- 4 files changed, 55 insertions(+), 26 deletions(-) diff --git a/.github/dependabot.yml b/.github/dependabot.yml index 14d1df9..dd347b7 100644 --- a/.github/dependabot.yml +++ b/.github/dependabot.yml @@ -4,8 +4,19 @@ updates: directory: "/" schedule: interval: weekly + # Transitive crates too: a root daemon parsing untrusted packets is only + # as current as its whole dependency tree. + allow: + - dependency-type: all + groups: + # Minor and patch bumps arrive as one PR; majors stay separate. + cargo-minor: + update-types: ["minor", "patch"] - package-ecosystem: github-actions directory: "/" schedule: interval: weekly + groups: + actions: + patterns: ["*"] diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 2265047..69cd189 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -21,6 +21,9 @@ jobs: - name: Resolve version and verify it matches the tag id: version + env: + REF_TYPE: ${{ github.ref_type }} + REF_NAME: ${{ github.ref_name }} run: | VERSION="$(sed -n '/^\[workspace\.package\]/,/^\[/{s/^version *= *"\(.*\)"/\1/p}' Cargo.toml | head -n1)" if [ -z "${VERSION}" ]; then @@ -30,8 +33,8 @@ jobs: # The release is named after the tag, but every asset filename is # built from the Cargo version. A mismatch publishes "v0.3.0" # containing colony-firewall-control-0.2.0-*.tar.zst. Refuse. - if [ "${{ github.ref_type }}" = "tag" ] && [ "${{ github.ref_name }}" != "v${VERSION}" ]; then - echo "::error::tag ${{ github.ref_name }} does not match Cargo.toml [workspace.package] version ${VERSION} (expected tag v${VERSION}). Bump Cargo.toml or retag." + if [ "${REF_TYPE}" = "tag" ] && [ "${REF_NAME}" != "v${VERSION}" ]; then + echo "::error::tag ${REF_NAME} does not match Cargo.toml [workspace.package] version ${VERSION} (expected tag v${VERSION}). Bump Cargo.toml or retag." exit 1 fi echo "version=${VERSION}" >> "${GITHUB_OUTPUT}" @@ -189,7 +192,7 @@ jobs: crates/cfc-ebpf/target/bpfel-unknown-none/release/cfc-ebpf.o \ "${STAGE}/" - # Docs, for parity with what the AUR package puts in + # Docs, for parity with what pkg/PKGBUILD puts in # /usr/share/doc. `cfc status` points users at TROUBLESHOOTING.md # by name, so it has to actually ship. install -m644 \ @@ -236,8 +239,9 @@ jobs: SHA256SUMS RELEASE_BODY.md - # Build pkg/PKGBUILD for real against the freshly pushed tag, and produce - # the AUR-submittable artifacts (PKGBUILD with real checksums + .SRCINFO). + # Build pkg/PKGBUILD for real against the freshly pushed tag, and attach + # the Arch packaging recipe (PKGBUILD with real checksums + .SRCINFO) to + # the release. The project is not published on the AUR. # # Hard gate, no continue-on-error: at tag time the source= URL # (.../archive/v$pkgver.tar.gz) resolves, because GitHub generates the @@ -275,10 +279,12 @@ jobs: chown -R builder: . - name: Verify pkgver matches the tag + env: + REF_NAME: ${{ github.ref_name }} run: | PKGVER="$(sed -n 's/^pkgver=//p' pkg/PKGBUILD | head -n1)" - if [ "${{ github.ref_name }}" != "v${PKGVER}" ]; then - echo "::error::pkg/PKGBUILD pkgver=${PKGVER} does not match tag ${{ github.ref_name }}" + if [ "${REF_NAME}" != "v${PKGVER}" ]; then + echo "::error::pkg/PKGBUILD pkgver=${PKGVER} does not match tag ${REF_NAME}" exit 1 fi @@ -287,7 +293,7 @@ jobs: run: | runuser -u builder -- updpkgsums PKGBUILD # A remote (non-VCS) source with sha256sums=('SKIP') is not - # acceptable AUR practice: it disables integrity checking of the + # acceptable Arch packaging practice: it disables integrity checking of the # release tarball entirely. if grep -qE "^sha256sums=\(.*'SKIP'" PKGBUILD; then echo "::error::pkg/PKGBUILD still has a SKIP checksum after updpkgsums" @@ -308,8 +314,8 @@ jobs: # An undotted copy is what gets attached: GitHub renames dot-leading # asset filenames (`.SRCINFO` became `default.SRCINFO`), so the # published name never matched what the README told people to - # download. Ship a name GitHub keeps; the AUR checkout renames it - # back to `.SRCINFO` locally. + # download. Ship a name GitHub keeps; whoever builds from it renames + # it back to `.SRCINFO` locally. cp .SRCINFO SRCINFO - name: namcap the PKGBUILD @@ -359,7 +365,7 @@ jobs: exit 1 fi - - name: Upload AUR artifacts + - name: Upload Arch packaging artifacts uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: aur-assets @@ -394,7 +400,7 @@ jobs: files: | release-assets/colony-firewall-control-*.tar.zst release-assets/SHA256SUMS - - name: Attach AUR artifacts + - name: Attach Arch packaging artifacts if: github.ref_type == 'tag' uses: softprops/action-gh-release@efb35369e0ad2afab669f228072c1b0d510eae64 # v3.0.3 with: diff --git a/README.md b/README.md index eac714c..d400d1e 100644 --- a/README.md +++ b/README.md @@ -68,17 +68,22 @@ NFQUEUE in the kernel, per-app pop-ups in iced, gRPC IPC over a Unix socket. +--------------------------------------------------+ ``` -Seven workspace crates: - -| Crate | Role | -|---------------|----------------------------------------------------------| -| `cfc-core` | Shared types: `Rule`, `Verdict`, `Connection`, `Process` | -| `cfc-proto` | gRPC schema (tonic + tonic-prost) | -| `cfc-client` | Shared UDS gRPC client wrapper | -| `cfc-daemon` | Privileged daemon | -| `cfc-ui` | iced GUI | -| `cfc-cli` | Terminal control tool | -| `cfc-tray` | System-tray companion (StatusNotifierItem) | +Ten crates: nine workspace members, plus the kernel-side `cfc-ebpf`, which +is its own workspace (pinned nightly + bpf-linker, built by `cargo xtask +build-ebpf`) so stable builds never see it: + +| Crate | Role | +|-------------------|----------------------------------------------------------------------------| +| `cfc-core` | Shared types and rule matching: `Rule`, `Verdict`, `Connection`, `Process` | +| `cfc-proto` | gRPC schema (tonic + tonic-prost) | +| `cfc-client` | Shared UDS gRPC client wrapper | +| `cfc-daemon` | Privileged daemon | +| `cfc-ui` | iced GUI | +| `cfc-cli` | Terminal control tool | +| `cfc-tray` | System-tray companion (StatusNotifierItem) | +| `cfc-ebpf-common` | POD types and pure parsers shared by eBPF and userspace | +| `cfc-ebpf` | Kernel-side programs of the optional eBPF backend | +| `xtask` | Build automation (eBPF object build) | More docs: @@ -299,6 +304,9 @@ cfc status # "enforcing yes", and it warns on stderr when it is not ### Explicit application confinement +**Experimental.** This mode is new in 0.7.0, has not been externally +audited, and its interface and platform requirements may change. + `cfc applications run` starts a separate, headless application tree with an empty network permission list. Administrators may approve exact numeric peer addresses with `--allow IP`. Permissions apply to the entire tree across diff --git a/SECURITY.md b/SECURITY.md index 1de02eb..0c9dc9a 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -1,10 +1,14 @@ # Security Policy -Colony Firewall Control is **alpha software**. It runs a daemon as root with +Colony Firewall Control is **beta software**. It runs a daemon as root with `CAP_NET_ADMIN` and makes allow/deny decisions about your network traffic, so security reports are taken seriously -- but expectations should match the -project's maturity: there has been no external audit, and interfaces may -change without notice. +project's maturity: **there has been no external security audit yet**, and +interfaces may change without notice. + +The explicit application confinement mode (`cfc applications run`, new in +0.7.0) is **experimental**, and its interface and platform requirements may +change. ## Supported Versions From 53db3f1d27f3ea7860786785512b74a6cb65f9fe Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:26:00 +0200 Subject: [PATCH 008/125] fix(nft): let new loopback flows through when no daemon listens Add `oifname "lo" ct state new queue num 0 bypass` before the final queue rule. The daemon still judges loopback flows while it runs; when nothing listens on the queue, local IPC such as the systemd-resolved stub keeps working, and every other new flow stays fail-closed. Docs, packaging notes and doc comments now describe the ruleset as fail-closed for everything except new loopback flows. The VM bench gains a loopback UDP echo measurement (floor vs armed), and the armed e2e asserts that a new loopback TCP flow succeeds after the daemon is killed. --- README.md | 11 ++-- TODO.md | 9 ++-- crates/cfc-daemon/src/config.rs | 21 ++++---- crates/cfc-ebpf-common/src/lib.rs | 4 +- docs/ARCHITECTURE.md | 3 +- docs/HARDENING.md | 14 +++-- docs/TROUBLESHOOTING.md | 58 ++++++++++---------- packaging/rpm/colony-firewall-control.spec | 5 +- packaging/selinux/README.md | 2 +- packaging/selinux/TESTING.md | 10 ++-- packaging/selinux/colony_firewall.te | 5 +- pkg/README.md | 2 +- pkg/colony-firewall-control.install | 5 +- scripts/armed-e2e.sh | 22 ++++++-- scripts/vm-bench/README.md | 8 ++- scripts/vm-bench/plan.sh | 61 +++++++++++++++++++--- systemd/colony-firewalld.service | 5 +- systemd/daemon.toml.sample | 6 +-- systemd/nftables-snippet.conf | 14 +++-- 19 files changed, 183 insertions(+), 82 deletions(-) diff --git a/README.md b/README.md index d400d1e..504fd5e 100644 --- a/README.md +++ b/README.md @@ -275,8 +275,10 @@ managers, or a later external ruleset flush. Early unmatched flows use established and related traffic retains its connection-wide authorization. Passed or inherited sockets are not reauthorized for each sending executable. A current descriptor holder does not prove which process sent a packet. -New direct loopback flows follow explicit rules; unmatched local IPC is allowed -without prompting. An allowed local resolver or proxy can still relay remote +While the daemon runs, new direct loopback flows follow explicit rules; +unmatched local IPC is allowed without prompting. While no daemon listens on +the queue, new loopback flows are allowed (`queue ... bypass` on `lo` only), so +the systemd-resolved stub and other local services keep working. An allowed local resolver or proxy can still relay remote traffic. CFC cannot establish the originating application's identity from remote flows delegated through local brokers, including AF_UNIX and D-Bus. @@ -295,8 +297,9 @@ cfc status # "enforcing yes", and it warns on stderr when it is not ``` > **WARNING - remote / SSH machines:** the shipped nftables snippet is -> fail-closed. If the daemon is down while the rule is loaded, **all new -> outbound connections drop**, and a mistake can lock you out of a box you +> fail-closed for everything except new loopback flows, which are allowed +> while no daemon listens. If the daemon is down while the rule is loaded, +> **all new non-loopback outbound connections drop**, and a mistake can lock you out of a box you > only reach over SSH. Read > [docs/TROUBLESHOOTING.md](docs/TROUBLESHOOTING.md) - specifically the > SSH exemption and dead-man's-switch patterns - *before* enabling diff --git a/TODO.md b/TODO.md index 52f4374..7aa0eaf 100644 --- a/TODO.md +++ b/TODO.md @@ -243,15 +243,16 @@ What defeats it completely: |---|---| | **Root** | narrower than it was, and still open. `nft delete table` no longer lifts the denials held in the kernel - those need `rm -rf /sys/fs/bpf/colony-firewall` as well, and anything not yet decided still falls through to a ruleset root can flush. CFC *is* root; it cannot confine root. | | **Code inside an allowed process** | a browser extension, a script under an allowed interpreter, `ptrace`/`LD_PRELOAD` injection. Structural to every application firewall. Making Allow persistent (`72964b5`) improved usability and widened this. | -| **Loopback** | `oifname "lo" accept`, deliberately - filtering it stalls the systemd-resolved stub. Anything that can reach a local service which egresses is attributed to that service. | +| **Loopback** | `oifname "lo" ct state new queue num 0 bypass`: the daemon judges new loopback flows while it runs (unmatched local IPC is allowed without prompting), and they are allowed unfiltered while no daemon listens, so the systemd-resolved stub survives a dead daemon. Anything that can reach a local service which egresses is attributed to that service. | | **DNS tunnelling** | the resolver must be allowed for anything to work. CFC *observes* answers; it does not inspect or block queries. | | **Inherited or passed socket descriptors** | Existing connection authorization is not rechecked for each sending executable; socket attribution is ambiguous when ownership is shared. | | **CAP_NET_RAW packet sockets** | Packet-layer egress can bypass the IP OUTPUT hook. Layer-2 confinement is outside the shipped rules. | | **Prompt fatigue** | demonstrated on this machine: ten Firefox prompts in a row, all denied, browser lost. A malicious installer generating thirty prompts trains the user to click Allow. | -And one tradeoff worth stating plainly: the ruleset is **fail-closed** (`ct -state new queue num 0`, no `bypass`). Killing the daemon drops all new outbound -traffic. That is the right choice for confidentiality and the wrong one for +And one tradeoff worth stating plainly: the ruleset is **fail-closed for +everything except new loopback flows, which are allowed while no daemon +listens** (the final `ct state new queue num 0` has no `bypass`). Killing the +daemon drops all new non-loopback outbound traffic. That is the right choice for confidentiality and the wrong one for availability - anything that can crash the daemon takes the machine's network with it. diff --git a/crates/cfc-daemon/src/config.rs b/crates/cfc-daemon/src/config.rs index cfb0afd..a5df462 100644 --- a/crates/cfc-daemon/src/config.rs +++ b/crates/cfc-daemon/src/config.rs @@ -156,9 +156,10 @@ impl Profile { /// /// The outbound table cannot lock an operator out of a remote machine: it /// hooks `output` on `ct state new` only, so an inbound SSH session's - /// replies are `ct state established` and are never queued. New loopback - /// flows follow explicit policy; unmatched local IPC is allowed without - /// prompting. Rules can still be added with `cfc-cli` from that + /// replies are `ct state established` and are never queued. While the + /// daemon runs, new loopback flows follow explicit policy and unmatched + /// local IPC is allowed without prompting; while no daemon listens, the + /// snippet's `bypass` on `lo` allows them. Rules can still be added with `cfc-cli` from that /// session. What it *does* mean on a fresh headless install is that /// outbound traffic — package updates, NTP, backups — is denied until /// rules exist for it. @@ -379,9 +380,10 @@ impl<'de> Deserialize<'de> for EbpfMode { /// `Auto`, matching how `profile` already treats an unknown value, and for /// a reason specific to this daemon: a config parse error propagates out of /// `Config::load` and the process exits *before* `READY=1`. The nftables - /// ruleset is `ct state new queue num 0` with no `bypass`, so a loaded - /// table with no daemon behind it blackholes every new outbound connection - /// on the machine. A typo in an enrichment layer's switch must not cost + /// ruleset is fail-closed for everything except new loopback flows, which + /// are allowed while no daemon listens: the final `ct state new queue num + /// 0` has no `bypass`, so a loaded table with no daemon behind it + /// blackholes every new non-loopback outbound connection on the machine. A typo in an enrichment layer's switch must not cost /// someone their network. fn deserialize>(d: D) -> Result { struct V; @@ -644,9 +646,10 @@ enabled = " Auto ""# /// A typo must not be able to take the machine's network away. /// /// A config parse error propagates out of `Config::load` and the daemon - /// exits *before* `READY=1`. `systemd/nftables-snippet.conf` is - /// `ct state new queue num 0` with **no** `bypass`, so a loaded table with - /// no daemon behind it drops every new outbound connection. Refusing to + /// exits *before* `READY=1`. `systemd/nftables-snippet.conf` ends with + /// `ct state new queue num 0` with **no** `bypass` (only the loopback rule + /// above it has one), so a loaded table with no daemon behind it drops + /// every new non-loopback outbound connection. Refusing to /// start over a misspelled enrichment-layer switch would turn a one-letter /// mistake into an outage, so an unknown value warns and falls back - /// exactly as `profile` already does. diff --git a/crates/cfc-ebpf-common/src/lib.rs b/crates/cfc-ebpf-common/src/lib.rs index d27a16c..e100a49 100644 --- a/crates/cfc-ebpf-common/src/lib.rs +++ b/crates/cfc-ebpf-common/src/lib.rs @@ -113,7 +113,9 @@ const _: () = { /// has seen `exec`, so a default deny here would blackhole every process that /// started before the daemon did - including the ones that bring the network /// up. The fail-closed guarantee stays where it already was, in the nftables -/// ruleset (`ct state new queue num 0`, no `bypass`). +/// ruleset: fail-closed for everything except new loopback flows, which are +/// allowed while no daemon listens (the final `ct state new queue num 0` has +/// no `bypass`). /// /// The point of this layer is the *opposite* direction: a deny written here /// keeps being enforced after the daemon is gone, because the link is pinned. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 9ddb5e5..4da8559 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -50,7 +50,8 @@ headless machine gets a say. ``` kernel (nftables OUTPUT hook) | - | loopback / established,related / daemon refusal packets accepted + | established,related / daemon refusal packets accepted + | oifname lo ct state new queue num 0 bypass (accepted if no daemon) | ct state new queue num 0 | all other traffic dropped v diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 42bc88b..f0c208a 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -184,8 +184,10 @@ is a separate launch mode. instead of, traditional access controls. - **eBPF / unprivileged user namespaces**: a sufficiently privileged user can bypass NFQUEUE entirely with `unshare -rn` and a custom net namespace. -- **Local relays and DNS**: explicit rules apply to new direct loopback flows. - Unmatched local IPC is allowed without prompting. An authorized local +- **Local relays and DNS**: while the daemon runs, explicit rules apply to + new direct loopback flows and unmatched local IPC is allowed without + prompting. While no daemon listens on the queue, new loopback flows are + allowed unfiltered (`bypass` on the `lo` rule only). An authorized local resolver or proxy can relay remote traffic, which is attributed to that service. CFC cannot establish the originating application's identity from remote flows delegated through AF_UNIX or D-Bus brokers. Existing local @@ -420,8 +422,9 @@ to `cgroupfs`. The other half of the security posture is the nftables side, not the daemon: whether the kernel drops or accepts new connections when nobody -is answering the queue. The shipped snippet is fail-closed, which is the -safer default and also the one that can lock you out of a remote box. +is answering the queue. The shipped snippet is fail-closed for everything +except new loopback flows, which are allowed while no daemon listens. That +is the safer default and also the one that can lock you out of a remote box. The full matrix - daemon up or down, table loaded or not, with and without `bypass` - is in [TROUBLESHOOTING.md](TROUBLESHOOTING.md#fail-open-vs-fail-closed-matrix). @@ -429,7 +432,8 @@ Read it before enabling enforcement on a machine you only reach over SSH. `[nfqueue] fail_open` must be `false`; `true` is rejected. Queue overflow must drop traffic instead of bypassing policy and durable refusal auditing. -The nftables `bypass` keyword governs missing listeners and is not shipped. +The nftables `bypass` keyword governs missing listeners; the shipped snippet +uses it only on the loopback rule (`oifname "lo"`). ## When something stops working diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 16a3f6c..24ff744 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -37,11 +37,15 @@ this check. `CFC_INBOUND_FORCE=1` remains the explicit console override. ## Testing over SSH without locking yourself out -The shipped nftables snippet is **fail-closed**: `queue num 0` without the -`bypass` keyword means that if nothing is listening on NFQUEUE 0 (daemon -stopped, crashed, or not yet started), the kernel drops every *new* -outbound connection. Your established SSH session survives (`ct state new` -only matches new flows), but the moment it drops you cannot open a new one. +The shipped nftables snippet is **fail-closed for everything except new +loopback flows, which are allowed while no daemon listens**: the final +`queue num 0` without the `bypass` keyword means that if nothing is +listening on NFQUEUE 0 (daemon stopped, crashed, or not yet started), the +kernel drops every *new* non-loopback outbound connection. Only the rule +just above it, `oifname "lo" ct state new queue num 0 bypass`, lets new +loopback flows through in that state. Your established SSH session +survives (`ct state new` only matches new flows), but the moment it drops +you cannot open a new one. Three layers of protection, use all of them the first time: @@ -98,8 +102,8 @@ cfc status ``` If `systemctl` shows the unit dead while the nftables rule is loaded, you -are in the fail-closed state described above: packets are queued to NFQUEUE -0 and nobody answers. Start the daemon or delete the table. +are in the fail-closed state described above: non-loopback packets are +queued to NFQUEUE 0 and nobody answers. Start the daemon or delete the table. **Is the nftables table actually loaded?** @@ -120,7 +124,8 @@ If they differ, packets queue to a number nobody consumes - same lockout as a dead daemon. **The fail-open alternative.** If you would rather lose filtering than -lose the network when the daemon is down, add the `bypass` keyword: +lose the network when the daemon is down, add the `bypass` keyword to the +final queue rule too: ``` ct state new queue num 0 bypass @@ -237,27 +242,26 @@ them apart from a bad argument (2) or a missing rule (3). The snippet's `output` hook matches loopback traffic too. On systems using systemd-resolved, every DNS query goes to the stub resolver at -`127.0.0.53:53` - over loopback - so each lookup gets intercepted and can -prompt, time out, or (under `strict`) be denied. The symptom is DNS that -is slow, flaky, or dead while direct-by-IP connections work. - -Exempt loopback above the queue rule: +`127.0.0.53:53` - over loopback. The shipped ruleset queues new loopback +flows with their own rule, just above the final queue rule: ``` -table inet colony_firewall { - chain output { - type filter hook output priority 0; policy accept; - oifname lo accept - ct state new queue num 0 - } -} +oifname "lo" ct state new queue num 0 bypass +ct state new queue num 0 ``` -Loopback traffic never leaves the machine, so exempting it costs you no -outbound coverage. The ruleset installed by the companion -`colony-firewall-nft.service` unit includes this exemption; the caveat -applies mainly if you carry an older copy of the snippet in your own -`/etc/nftables.conf`. +While the daemon runs, it judges them like any other flow: explicit rules +apply, and unmatched local IPC (the stub resolver, CUPS, a local dev +server) is allowed without prompting. While nothing listens on the queue, +`bypass` makes the kernel accept them, so local DNS and IPC keep working +when the daemon is down. Every non-loopback new flow still meets the +fail-closed rule. + +If DNS is slow, flaky, or dead while direct-by-IP connections work, check +for an older copy of the snippet in your own `/etc/nftables.conf`: one +without the loopback rule drops the stub resolver whenever the daemon is +down, and one with an explicit `oifname lo accept` skips the daemon for +loopback entirely. Note the daemon already exempts its *own* reverse-DNS lookups internally (they would otherwise deadlock the queue); the loopback rule is about @@ -267,10 +271,10 @@ everyone else's DNS. What happens to a **new outbound connection** in each state: -| State | Without `bypass` (shipped) | With `bypass` | +| State | Without `bypass` on the final rule (shipped) | With `bypass` | |------------------------------------|--------------------------------|--------------------------------| | Daemon up, nft rule loaded | Filtered: rules, then prompts, then profile fallback | Same | -| Daemon down, nft rule loaded | **Dropped. Total outbound lockout.** | Allowed, unfiltered (silent) | +| Daemon down, nft rule loaded | **Dropped. Outbound lockout** (new loopback flows still allowed) | Allowed, unfiltered (silent) | | Daemon up, nft rule *not* loaded | Allowed, unfiltered (silent - daemon sees nothing) | Same | | Daemon paused (`cfc pause`) | Rules still enforced; only *unmatched* flows pass instead of prompting. Auto-resumes | Same | diff --git a/packaging/rpm/colony-firewall-control.spec b/packaging/rpm/colony-firewall-control.spec index 5acf394..1adcfd8 100644 --- a/packaging/rpm/colony-firewall-control.spec +++ b/packaging/rpm/colony-firewall-control.spec @@ -61,8 +61,9 @@ it - and the daemon adds process attribution, DNS display enrichment, and in-kernel connect(2) denial. Pinned denials survive a daemon crash. Fast Allow was removed; allowed connections go through NFQUEUE. -The ruleset is fail-closed. If the daemon is not running, new outbound -connections are dropped rather than allowed. +The ruleset is fail-closed for everything except new loopback flows. If the +daemon is not running, new non-loopback outbound connections are dropped +rather than allowed; new loopback flows are allowed so local IPC keeps working. %package selinux Summary: SELinux policy module for %{name} diff --git a/packaging/selinux/README.md b/packaging/selinux/README.md index 1b01d1e..a7a1197 100644 --- a/packaging/selinux/README.md +++ b/packaging/selinux/README.md @@ -31,7 +31,7 @@ tells you which kind of trouble you are in. | group | denial costs | |---|---| -| netlink_netfilter, raw sockets | **everything.** The daemon exits before `READY=1`, and the ruleset is fail-closed, so the machine loses outbound network | +| netlink_netfilter, raw sockets | **everything.** The daemon exits before `READY=1`, and the ruleset is fail-closed (except new loopback flows), so the machine loses non-loopback outbound network | | unix socket under `/run` | the CLI, tray and GUI cannot reach the daemon; filtering continues, unattended | | `bpf`, `perf_event`, tracefs, cgroup | the ring-0 layer. Attribution falls back to `sock_diag` + `/proc`, hostnames to PTR lookups. Logged once, then filtering continues | | bpffs (`/sys/fs/bpf`) | in-kernel denials no longer survive the daemon being killed. Silent apart from `enforcement=process` in the startup line | diff --git a/packaging/selinux/TESTING.md b/packaging/selinux/TESTING.md index 42ef5bc..cf51101 100644 --- a/packaging/selinux/TESTING.md +++ b/packaging/selinux/TESTING.md @@ -11,8 +11,9 @@ protocol for whoever has such a host. Run it once, report what you see, and ## What you need - A Rocky 9 or Fedora VM with SELinux enforcing (`getenforce` says - `Enforcing`). A VM, not your workstation: the ruleset is fail-closed, and a - policy gap in the wrong group takes the machine's outbound network down. + `Enforcing`). A VM, not your workstation: the ruleset is fail-closed + (except new loopback flows), and a policy gap in the wrong group takes the + machine's outbound network down. For the same reason, have **console access**, not just SSH. - The audit tooling: `dnf install audit policycoreutils-python-utils`. `semanage` and `audit2allow` live in the second package, and on a minimal @@ -37,8 +38,9 @@ AVC in the audit log **without being enforced** - observed, not suffered. Why that ordering matters here more than for most policies, in the module's own words: a denied `netlink_netfilter` socket is not a degraded feature, it is a daemon that exits before `READY=1` - and because the nftables ruleset is -fail-closed (`ct state new queue num 0`, no `bypass`), a daemon that does not -come up takes the machine's outbound network with it. Running the first pass +fail-closed for everything except new loopback flows, which are allowed while +no daemon listens, a daemon that does not come up takes the machine's +non-loopback outbound network with it. Running the first pass permissive converts that outage into a log line. Dontaudit rules hide denials, and this module carries some diff --git a/packaging/selinux/colony_firewall.te b/packaging/selinux/colony_firewall.te index 3a6c05e..081db82 100644 --- a/packaging/selinux/colony_firewall.te +++ b/packaging/selinux/colony_firewall.te @@ -101,8 +101,9 @@ allow colony_firewalld_t self:unix_dgram_socket create_socket_perms; # # This is the group that must not fail. A denial here is not a degraded # feature, it is a daemon that exits before READY=1 - and because the nftables -# ruleset is fail-closed (`ct state new queue num 0`, no `bypass`), a daemon -# that does not come up takes the machine's outbound network with it. +# ruleset is fail-closed for everything except new loopback flows, which are +# allowed while no daemon listens, a daemon that does not come up takes the +# machine's non-loopback outbound network with it. # ######################################## diff --git a/pkg/README.md b/pkg/README.md index 94c0936..cbc8b65 100644 --- a/pkg/README.md +++ b/pkg/README.md @@ -27,7 +27,7 @@ Key design points: - **`colony-firewall-nft.service`** makes enforcement persistent (`nft -f` the snippet on start, `nft delete table inet colony_firewall` on explicit nft-unit stop). The table survives daemon restarts and stops, - so new flows fail closed while its queue listener is absent. Upgrades reload + so new non-loopback flows fail closed while its queue listener is absent. Upgrades reload active nft units atomically and leave inactive inbound filtering opt-in. The daemon requires this unit before initialization. Enabling either nft unit creates native `Requires` links from NetworkManager and diff --git a/pkg/colony-firewall-control.install b/pkg/colony-firewall-control.install index b69ae22..ffc1231 100644 --- a/pkg/colony-firewall-control.install +++ b/pkg/colony-firewall-control.install @@ -7,7 +7,8 @@ post_install() { 1. Enable the daemon and the persistent nftables rules: systemctl enable --now colony-firewalld colony-firewall-nft (colony-firewall-nft loads 'table inet colony_firewall'; it is - fail-closed while loaded and survives daemon restarts/stops. + fail-closed while loaded, except new loopback flows, and survives + daemon restarts/stops. Stop colony-firewall-nft explicitly to remove filtering.) 2. Let your desktop user talk to the daemon socket: @@ -51,7 +52,7 @@ EOF pre_remove() { # Stop enforcement BEFORE the binaries and the snippet disappear, so # the fail-closed NFQUEUE table can never outlive the daemon and - # blackhole all new outbound traffic. + # blackhole all new non-loopback outbound traffic. # # The inbound pair belongs here too, and used not to be. Removing the # package on a host where the inbound chain had been enabled left diff --git a/scripts/armed-e2e.sh b/scripts/armed-e2e.sh index e963acc..aca44da 100755 --- a/scripts/armed-e2e.sh +++ b/scripts/armed-e2e.sh @@ -19,6 +19,7 @@ SRV_IP=10.200.0.2 ALLOW_PORT=8080 # allow rule DENY_PORT=8081 # deny rule UNMATCHED_PORT=8082 # no rule: balanced profile, nobody subscribed -> Deny +LO_PORT=8083 # 127.0.0.1 inside FW, last step only W="$(mktemp -d "${RUNNER_TEMP:-/tmp}/cfc-e2e.XXXXXX")" SOCK="${W}/cfc.sock" @@ -86,7 +87,9 @@ cleanup() { if sudo test -s "${W}/daemon.pid"; then sudo kill -KILL "$(sudo cat "${W}/daemon.pid")" 2>/dev/null || true fi - sudo ip netns pids "${SRV}" 2>/dev/null | xargs -r sudo kill 2>/dev/null || true + for ns in "${SRV}" "${FW}"; do + sudo ip netns pids "${ns}" 2>/dev/null | xargs -r sudo kill 2>/dev/null || true + done sudo ip netns del "${FW}" 2>/dev/null || true sudo ip netns del "${SRV}" 2>/dev/null || true if [[ "${rc}" -ne 0 ]]; then @@ -115,8 +118,9 @@ in_fw ip addr add 10.200.0.1/24 dev fw0 in_fw ip link set fw0 up in_srv ip addr add "${SRV_IP}/24" dev srv0 in_srv ip link set srv0 up -# FW's lo stays down on purpose: no loopback flows (the daemon's own reverse -# DNS to a 127.0.0.53 stub fails fast instead of being queued). +# FW's lo stays down on purpose while a daemon runs: no loopback flows (the +# daemon's own reverse DNS to a 127.0.0.53 stub fails fast instead of being +# queued). The last step brings it up, after the final daemon is gone. mkdir "${W}/www" for p in "${ALLOW_PORT}" "${DENY_PORT}" "${UNMATCHED_PORT}"; do in_srv python3 -m http.server "${p}" --bind "${SRV_IP}" \ @@ -186,4 +190,16 @@ table_loaded || fail "table gone after SIGKILL" queue_bound && fail "NFQUEUE 0 still bound after SIGKILL" expect_drop "${ALLOW_PORT}" +say "No daemon: a new loopback flow still passes (bypass on lo only)" +in_fw ip link set lo up +in_fw python3 -m http.server "${LO_PORT}" --bind 127.0.0.1 \ + --directory "${W}/www" >"${W}/http-lo.log" 2>&1 & +for _ in $(seq 1 50); do + [[ -n "$(in_fw ss -Hltn "sport = :${LO_PORT}")" ]] && break + sleep 0.2 +done +in_fw curl -sS --noproxy '*' -o /dev/null --connect-timeout 3 --max-time 6 \ + "http://127.0.0.1:${LO_PORT}/" \ + || fail "new loopback flow failed with no daemon; the lo bypass rule should accept it" + say "Armed e2e passed" diff --git a/scripts/vm-bench/README.md b/scripts/vm-bench/README.md index d275824..289d241 100644 --- a/scripts/vm-bench/README.md +++ b/scripts/vm-bench/README.md @@ -29,8 +29,14 @@ isolates one cost. | `floor` | nothing: no daemon, no table | the veth link and `connect()` itself | | `queue-N` | the daemon, the table, a lasting Allow | the NFQUEUE round trip, at N flows | | `poll200us-N` | the same, with a daemon built with a shorter `RECV_POLL_INTERVAL` | how much of that round trip is the worker's idle beat | +| `lo-floor` | nothing; a UDP echo server on `127.0.0.1` inside the guest | the loopback round trip itself | +| `lo-queue` | the daemon and the table, same echo server | what the `oifname "lo" ... queue num 0 bypass` rule costs a new loopback flow | -Both directions run in every state and they answer different questions. `out` +The two `lo-*` states run after the sweep. The client opens a new socket per +round trip (1000 of them), so every round trip is a new conntrack flow, +and reports mean, p50, p90, p95, p99 and max under direction `lo`. + +Both directions run in every veth state and they answer different questions. `out` leaves through the host's output chain and meets the queue. `in` is generated inside the network namespace, whose own output chain carries no colony table, so it never meets a queue - but its client sits in the root cgroup and still diff --git a/scripts/vm-bench/plan.sh b/scripts/vm-bench/plan.sh index 618bc67..0bb82e4 100755 --- a/scripts/vm-bench/plan.sh +++ b/scripts/vm-bench/plan.sh @@ -95,17 +95,64 @@ probe_layer() { stop_daemon } +arm() { # $1 label $2 binary + write_cfg + start_daemon "$2" || { echo "FAIL $1"; return 1; } + nft -f "$SNIPPET" || { echo "FAIL $1 nft"; stop_daemon; return 1; } + write_rules + cfc --socket "$SOCK" rules import --replace /tmp/rules.json >/dev/null 2>&1 + sleep 4 +} + +# Loopback round trip: a UDP echo server on 127.0.0.1 and a client that opens +# one new socket (one new conntrack flow, so one queued packet when armed) per +# round trip. Armed, this is the snippet's `oifname "lo" ... bypass` rule. +measure_lo() { # $1 label $2 mode(none|queue) + local q0 q1 + say "state: $1 n=1000 mode=$2 loopback udp echo" + if [ "$2" != none ]; then arm "$1" /usr/bin/colony-firewalld || return 1; fi + q0="$(qseq)" + python3 - "$1" 1000 <<'ECHO' | while read -r line; do echo "RESULT $line"; done +import json, socket, statistics, sys, threading, time +label, n = sys.argv[1], int(sys.argv[2]) +srv = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) +srv.bind(("127.0.0.1", 0)) +def echo(): + while True: + data, peer = srv.recvfrom(64) + srv.sendto(data, peer) +threading.Thread(target=echo, daemon=True).start() +ms, fails = [], 0 +for _ in range(n): + c = socket.socket(socket.AF_INET, socket.SOCK_DGRAM) + c.settimeout(2) + t = time.perf_counter() + try: + c.sendto(b"x", srv.getsockname()) + c.recv(64) + ms.append((time.perf_counter() - t) * 1000) + except OSError: + fails += 1 + c.close() +r = {"label": label, "direction": "lo", "ok": len(ms), "failed": fails, "ms": None} +if len(ms) > 1: + q = statistics.quantiles(ms, n=100) + r["ms"] = {"mean": statistics.fmean(ms), "p50": q[49], "p90": q[89], + "p95": q[94], "p99": q[98], "max": max(ms)} +print(json.dumps(r)) +ECHO + q1="$(qseq)" + ctx "$1 queued_packets=$(( q1 - q0 )) conntrack=$(ctcount)" + [ "$2" != none ] && stop_daemon + return 0 +} + measure() { # $1 label $2 n $3 mode(none|queue) $4 binary local label="$1" n="$2" mode="$3" bin="${4:-/usr/bin/colony-firewalld}" q0 q1 say "state: $label n=$n mode=$mode daemon=$(basename "$bin")" if [ "$mode" != none ]; then [ -x "$bin" ] || { echo "SKIP $label: $bin is not in this image"; return 0; } - write_cfg - start_daemon "$bin" || { echo "FAIL $label"; return 1; } - nft -f "$SNIPPET" || { echo "FAIL $label nft"; stop_daemon; return 1; } - write_rules - cfc --socket "$SOCK" rules import --replace /tmp/rules.json >/dev/null 2>&1 - sleep 4 + arm "$label" "$bin" || return 1 fi ctx "$label before sockets=$(sockets) conntrack=$(ctcount)" q0="$(qseq)" @@ -135,4 +182,6 @@ for n in "$SMALL" "$LARGE"; do drain; measure "poll200us-$n" "$n" queue /usr/bin/colony-firewalld-alt done [ "$SMALL" != "$LARGE" ] && { drain; measure "floor-$LARGE" "$LARGE" none; } +drain; measure_lo lo-floor none +drain; measure_lo lo-queue queue say "done" diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index 2e1ead4..12378d1 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -54,8 +54,9 @@ TimeoutStopSec=15 # without CAP_DAC_READ_SEARCH every connection resolves to `exe= # pid=0`, so no exe-scoped rule can ever match # and a fail-closed ruleset denies the whole -# machine's traffic. No warning at all: from the -# daemon's side, nothing failed. +# machine's non-loopback traffic. No warning +# at all: from the daemon's side, nothing +# failed. # # CAP_DAC_OVERRIDE is deliberately NOT granted: read-and-search is all this # needs, and the write half is what would let the daemon edit files it does not diff --git a/systemd/daemon.toml.sample b/systemd/daemon.toml.sample index 8b5326e..bc34c3c 100644 --- a/systemd/daemon.toml.sample +++ b/systemd/daemon.toml.sample @@ -87,9 +87,9 @@ profile = "balanced" [nfqueue] # NFQUEUE number. Must match the `queue num N` in the nftables/iptables rule # that enqueues packets. If they disagree, packets queue to a number nobody -# consumes - which under the shipped fail-closed rule is a total outbound -# lockout. Bind failure is fatal: the daemon exits non-zero rather than -# running while enforcing nothing. +# consumes - which under the shipped fail-closed rule is an outbound lockout +# for everything except new loopback flows. Bind failure is fatal: the +# daemon exits non-zero rather than running while enforcing nothing. queue_num = 0 # Kernel queue length: how many packets may wait for a verdict before the # queue overflows. Raise it on a busy host that prompts a lot; each queued diff --git a/systemd/nftables-snippet.conf b/systemd/nftables-snippet.conf index 2eb8d1e..23fa81a 100644 --- a/systemd/nftables-snippet.conf +++ b/systemd/nftables-snippet.conf @@ -1,7 +1,9 @@ -# New outbound flows require a verdict from NFQUEUE 0, without bypass. -# Established flows retain their connection-wide authorization, including SSH -# replies. New loopback flows follow explicit application policy; unmatched -# local IPC is allowed without prompting. See docs/HARDENING.md. +# New outbound flows require a verdict from NFQUEUE 0. Fail-closed for +# everything except new loopback flows, which are allowed while no daemon +# listens on the queue. Established flows retain their connection-wide +# authorization, including SSH replies. While the daemon runs, new loopback +# flows follow explicit application policy; unmatched local IPC is allowed +# without prompting. See docs/HARDENING.md. # Enable colony-firewall-nft.service for persistence. Its rules remain loaded # across daemon restarts and stops. Stop that unit explicitly to lift filtering. @@ -28,6 +30,10 @@ table inet colony_firewall { meta skuid 0 meta mark 0xcfc00001 icmp type destination-unreachable accept meta skuid 0 meta mark 0xcfc00001 icmpv6 type destination-unreachable accept + # Loopback is judged by the daemon while it runs. If the daemon is down, + # bypass lets local IPC such as the systemd-resolved stub keep working; + # every other new flow falls through to the fail-closed rule below. + oifname "lo" ct state new queue num 0 bypass ct state new queue num 0 # INVALID and UNTRACKED traffic drops, including explicit notrack flows. } From 07d761394f1edb726d03fdafbfa514273b3aecdf Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:29:47 +0200 Subject: [PATCH 009/125] docs: say exactly what the loopback bypass lets through With the daemon down, the systemd-resolved stub still answers from its cache, but its upstream queries are new non-loopback flows and stay blocked. While no daemon listens, explicit loopback Deny rules are not enforced, nothing is logged, and a loopback connection opened then keeps its authorization afterwards. Say so in HARDENING, TROUBLESHOOTING, the README and TODO, and fix the inbound snippet comment that still described an unconditional outbound loopback accept. --- README.md | 3 ++- TODO.md | 2 +- docs/HARDENING.md | 5 ++++- docs/TROUBLESHOOTING.md | 20 +++++++++++--------- systemd/nftables-inbound.conf | 10 ++++++---- systemd/nftables-snippet.conf | 3 ++- 6 files changed, 26 insertions(+), 17 deletions(-) diff --git a/README.md b/README.md index 504fd5e..01806f1 100644 --- a/README.md +++ b/README.md @@ -278,7 +278,8 @@ A current descriptor holder does not prove which process sent a packet. While the daemon runs, new direct loopback flows follow explicit rules; unmatched local IPC is allowed without prompting. While no daemon listens on the queue, new loopback flows are allowed (`queue ... bypass` on `lo` only), so -the systemd-resolved stub and other local services keep working. An allowed local resolver or proxy can still relay remote +local services keep working; resolving names that are not cached still needs +the daemon. An allowed local resolver or proxy can still relay remote traffic. CFC cannot establish the originating application's identity from remote flows delegated through local brokers, including AF_UNIX and D-Bus. diff --git a/TODO.md b/TODO.md index 7aa0eaf..9b87283 100644 --- a/TODO.md +++ b/TODO.md @@ -243,7 +243,7 @@ What defeats it completely: |---|---| | **Root** | narrower than it was, and still open. `nft delete table` no longer lifts the denials held in the kernel - those need `rm -rf /sys/fs/bpf/colony-firewall` as well, and anything not yet decided still falls through to a ruleset root can flush. CFC *is* root; it cannot confine root. | | **Code inside an allowed process** | a browser extension, a script under an allowed interpreter, `ptrace`/`LD_PRELOAD` injection. Structural to every application firewall. Making Allow persistent (`72964b5`) improved usability and widened this. | -| **Loopback** | `oifname "lo" ct state new queue num 0 bypass`: the daemon judges new loopback flows while it runs (unmatched local IPC is allowed without prompting), and they are allowed unfiltered while no daemon listens, so the systemd-resolved stub survives a dead daemon. Anything that can reach a local service which egresses is attributed to that service. | +| **Loopback** | `oifname "lo" ct state new queue num 0 bypass`: the daemon judges new loopback flows while it runs (unmatched local IPC is allowed without prompting), and they are allowed unfiltered while no daemon listens, so local IPC survives a dead daemon (the systemd-resolved stub answers from its cache; its upstream queries are not loopback). In that window explicit loopback Deny rules are not enforced and nothing is logged. Anything that can reach a local service which egresses is attributed to that service. | | **DNS tunnelling** | the resolver must be allowed for anything to work. CFC *observes* answers; it does not inspect or block queries. | | **Inherited or passed socket descriptors** | Existing connection authorization is not rechecked for each sending executable; socket attribution is ambiguous when ownership is shared. | | **CAP_NET_RAW packet sockets** | Packet-layer egress can bypass the IP OUTPUT hook. Layer-2 confinement is outside the shipped rules. | diff --git a/docs/HARDENING.md b/docs/HARDENING.md index f0c208a..5ed20d0 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -187,7 +187,10 @@ is a separate launch mode. - **Local relays and DNS**: while the daemon runs, explicit rules apply to new direct loopback flows and unmatched local IPC is allowed without prompting. While no daemon listens on the queue, new loopback flows are - allowed unfiltered (`bypass` on the `lo` rule only). An authorized local + allowed unfiltered (`bypass` on the `lo` rule only): an explicit loopback + Deny or Reject rule is not enforced in that window, nothing records those + flows, and a loopback connection opened then keeps its authorization once + the daemon is back. An authorized local resolver or proxy can relay remote traffic, which is attributed to that service. CFC cannot establish the originating application's identity from remote flows delegated through AF_UNIX or D-Bus brokers. Existing local diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 24ff744..ff0f900 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -253,15 +253,17 @@ ct state new queue num 0 While the daemon runs, it judges them like any other flow: explicit rules apply, and unmatched local IPC (the stub resolver, CUPS, a local dev server) is allowed without prompting. While nothing listens on the queue, -`bypass` makes the kernel accept them, so local DNS and IPC keep working -when the daemon is down. Every non-loopback new flow still meets the -fail-closed rule. - -If DNS is slow, flaky, or dead while direct-by-IP connections work, check -for an older copy of the snippet in your own `/etc/nftables.conf`: one -without the loopback rule drops the stub resolver whenever the daemon is -down, and one with an explicit `oifname lo accept` skips the daemon for -loopback entirely. +`bypass` makes the kernel accept them, so local IPC keeps working when the +daemon is down. That includes the stub resolver's socket, but only for names +it can answer from its cache or local records: its queries to the upstream +servers are new non-loopback flows, so resolving anything else still needs +the daemon. Every non-loopback new flow still meets the fail-closed rule. + +If you carry an older copy of the snippet in your own `/etc/nftables.conf`, +compare it with the shipped one: a copy without the loopback rule drops +every new loopback flow whenever the daemon is down, and one with an +explicit `oifname lo accept` skips the daemon for loopback entirely, so +loopback rules never apply. Note the daemon already exempts its *own* reverse-DNS lookups internally (they would otherwise deadlock the queue); the loopback rule is about diff --git a/systemd/nftables-inbound.conf b/systemd/nftables-inbound.conf index dbc3f6c..0ba22b2 100644 --- a/systemd/nftables-inbound.conf +++ b/systemd/nftables-inbound.conf @@ -47,10 +47,12 @@ table inet colony_firewall_inbound { # is the point. type filter hook input priority 0; policy drop; - # Loopback, first and unconditionally. The same reasoning as the - # outbound chain: the systemd-resolved stub, every local IPC over TCP - # and the daemon's own control socket live here, and filtering them - # buys nothing while being an excellent way to wedge the machine. + # Loopback, first and unconditionally. Every loopback flow also leaves + # through the outbound chain, which is where policy applies to it (the + # daemon judges it while it runs). Filtering the inbound copy as well + # buys nothing: the systemd-resolved stub, every local IPC over TCP and + # the daemon's own control socket live here, and a second drop point is + # an excellent way to wedge the machine. iifname "lo" accept # Traffic we asked for. Without this every reply to an outbound diff --git a/systemd/nftables-snippet.conf b/systemd/nftables-snippet.conf index 23fa81a..5b42e2d 100644 --- a/systemd/nftables-snippet.conf +++ b/systemd/nftables-snippet.conf @@ -31,7 +31,8 @@ table inet colony_firewall { meta skuid 0 meta mark 0xcfc00001 icmpv6 type destination-unreachable accept # Loopback is judged by the daemon while it runs. If the daemon is down, - # bypass lets local IPC such as the systemd-resolved stub keep working; + # bypass lets local IPC keep working, including the systemd-resolved + # stub socket (cached names only: its upstream queries are not loopback); # every other new flow falls through to the fail-closed rule below. oifname "lo" ct state new queue num 0 bypass ct state new queue num 0 From 76005486b07db199d6a56a3404a45cc7e8660327 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:35:41 +0200 Subject: [PATCH 010/125] fix(proto): allow clippy's double_must_use in the generated bindings Clippy 1.99, now the stable the CI resolves, flags the #[async_trait] that tonic emits for the server trait. The code is generated, so the lint is allowed on the generated module only; unknown_lints keeps older clippy versions from rejecting the name. --- crates/cfc-proto/src/lib.rs | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/crates/cfc-proto/src/lib.rs b/crates/cfc-proto/src/lib.rs index ba073e0..a7b18a7 100644 --- a/crates/cfc-proto/src/lib.rs +++ b/crates/cfc-proto/src/lib.rs @@ -3,6 +3,10 @@ //! Generated from `proto/cfc.proto`. Speaks daemon <-> UI/CLI over a Unix //! domain socket (typically `/run/colony-firewall/cfc.sock`). +// Generated code. Clippy 1.99's double_must_use fires inside the +// #[async_trait] that tonic emits for the server trait; nothing here can +// change that. unknown_lints keeps older clippy versions quiet about the name. +#[allow(unknown_lints, clippy::double_must_use)] pub mod v1 { tonic::include_proto!("cfc.v1"); } From c7355b5048a400ebf69b81212d5e2575b05352f7 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:36:59 +0200 Subject: [PATCH 011/125] test(provenance): check curl's owning package, not its version The real-machine test pinned curl 8.21.0-1 and failed as soon as the host updated curl, although provenance still verified. Assert the owning package instead. --- crates/cfc-daemon/src/provenance.rs | 6 +++++- 1 file changed, 5 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/provenance.rs b/crates/cfc-daemon/src/provenance.rs index 7b69bcd..4e6c2a4 100644 --- a/crates/cfc-daemon/src/provenance.rs +++ b/crates/cfc-daemon/src/provenance.rs @@ -1806,7 +1806,11 @@ mod tests { (cold, incl. index build: {:?})", started.elapsed() ); - assert_eq!(package.as_deref(), Some("curl 8.21.0-1")); + // The owning package, not a version: curl updates under this test. + assert!( + package.as_deref().is_some_and(|p| p.starts_with("curl ")), + "/usr/bin/curl must belong to the curl package, got {package:?}" + ); assert_eq!( provenance, Provenance::Verified, From b1b384fa1f0c6728d9c3f1976ca5bc534e38e580 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 00:37:44 +0200 Subject: [PATCH 012/125] test(provenance): compare the tampered case with the package found The previous commit left a second pinned curl version in the foreign-digest assertion. Compare with the package the same test just resolved. --- crates/cfc-daemon/src/provenance.rs | 5 +---- 1 file changed, 1 insertion(+), 4 deletions(-) diff --git a/crates/cfc-daemon/src/provenance.rs b/crates/cfc-daemon/src/provenance.rs index 4e6c2a4..8d86a65 100644 --- a/crates/cfc-daemon/src/provenance.rs +++ b/crates/cfc-daemon/src/provenance.rs @@ -1830,10 +1830,7 @@ mod tests { // the file the kernel mapped is not the file the package shipped. let tampered = describe(curl, Some(&"0".repeat(64))); println!("/usr/bin/curl with a foreign digest -> {tampered:?}"); - assert_eq!( - tampered, - (Some("curl 8.21.0-1".to_string()), Provenance::Modified) - ); + assert_eq!(tampered, (package.clone(), Provenance::Modified)); // A byte-identical copy in /tmp is owned by nobody: the dropper case. let tmp = tempfile::tempdir().unwrap(); From f12ad613ff9c4c605cecc4922e6ec508e033310c Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:24:00 +0200 Subject: [PATCH 013/125] fix(nft): accept IPv6 neighbour discovery, MLD and IGMP on the link Conntrack marks neighbour discovery and MLD untracked, so since the output chain became policy drop in 0.7.0 they never reached the queue and were dropped, breaking IPv6 neighbour resolution and multicast membership. The inbound chain dropped MLD the same way and refused every IGMP query. Both chains now accept these in the kernel, limited to the hop limits and sources RFC 4861, RFC 3810 and RFC 3376 require. The armed e2e pings FW over IPv6 with the table loaded, which needs FW's neighbour advertisement. TROUBLESHOOTING explains what is settled in the kernel, including why notrack flows still drop. --- CHANGELOG.md | 10 ++++++++++ docs/TROUBLESHOOTING.md | 18 ++++++++++++++++++ scripts/armed-e2e.sh | 9 +++++++++ systemd/nftables-inbound.conf | 8 ++++++++ systemd/nftables-snippet.conf | 16 +++++++++++++++- 5 files changed, 60 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1e166db..4d73994 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -16,6 +16,16 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). nftables set, disarms the legacy pinned maps and removes the old sendmsg link pins. +### Fixed + +- Since 0.7.0 the outbound table dropped IPv6 neighbour discovery and MLD, + which conntrack marks untracked, so IPv6 stopped working on hosts that load + it. Both tables now accept neighbour discovery, MLD and IGMP membership + traffic in the kernel, limited to the hop limits and sources the RFCs + require, so the inbound table no longer drops MLD or refuses IGMP queries + either. Other untracked traffic, including explicit `notrack` flows, still + drops; TROUBLESHOOTING.md says how to keep it. + ## [0.7.0] - 2026-09-30 ### Added diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index ff0f900..ece4bb8 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -269,6 +269,24 @@ Note the daemon already exempts its *own* reverse-DNS lookups internally (they would otherwise deadlock the queue); the loopback rule is about everyone else's DNS. +## Traffic that never reaches the daemon + +The `output` chain (and the inbound one) only queues `ct state new`. +Two kinds of packet are settled in the kernel instead: + +- **Link control is accepted.** IPv6 neighbour discovery and MLD, which + conntrack itself marks untracked, and IGMP membership traffic. Only the + hop limits (and, for MLD, the sources) the RFCs require match, so these + stay on the link. Without them IPv6 neighbour resolution and multicast group + membership stop working; ND and MLD never reach the queue, so no rule + could restore them. +- **Other INVALID and UNTRACKED packets drop**, including flows an + explicit `notrack` rule touched (a busy DNS or NTP server's tuning, for + instance). An `accept` in another table does not override this chain's + `policy drop`. To keep such flows, load a local copy of the snippet with + an accept for them above the queue rules, and point the unit at it with + a drop-in. + ## Fail-open vs fail-closed matrix What happens to a **new outbound connection** in each state: diff --git a/scripts/armed-e2e.sh b/scripts/armed-e2e.sh index aca44da..93a07e3 100755 --- a/scripts/armed-e2e.sh +++ b/scripts/armed-e2e.sh @@ -144,6 +144,15 @@ queue_bound && fail "something is already bound to NFQUEUE 0" say "Fail-closed before the daemon ever started" expect_drop "${ALLOW_PORT}" +say "IPv6 neighbour discovery passes the fail-closed chain" +in_fw ip -6 addr add fd00:200::1/64 dev fw0 nodad +in_srv ip -6 addr add fd00:200::2/64 dev srv0 nodad +# SRV's echo request is inbound to FW and FW's reply is established, so only +# FW's Neighbour Advertisement, which conntrack leaves untracked, meets the +# output chain's policy here. +in_srv ping -6 -c 1 -W 3 fd00:200::1 >/dev/null \ + || fail "IPv6 ping into FW failed; its neighbour advertisement was dropped" + cat >"${W}/daemon.toml" < Date: Thu, 8 Oct 2026 01:24:06 +0200 Subject: [PATCH 014/125] fix(daemon): count an absent /proc/net/udp6 as an empty table A kernel booted with ipv6.disable=1 has no udp6 table. The UDP search treated the failed read as a partial search, so every IPv4 UDP flow went unattributed and executable-scoped Allows (the DNS, NTP and DHCP bootstrap rules) refused. NotFound now means no sockets of that family; other read errors and deadline expiry still leave attribution unknown. --- CHANGELOG.md | 4 ++++ crates/cfc-daemon/src/process_resolve.rs | 18 +++++++++++++++++- docs/ARCHITECTURE.md | 5 +++-- 3 files changed, 24 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 4d73994..98547ca 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). require, so the inbound table no longer drops MLD or refuses IGMP queries either. Other untracked traffic, including explicit `notrack` flows, still drops; TROUBLESHOOTING.md says how to keep it. +- On kernels booted with `ipv6.disable=1` the missing `/proc/net/udp6` left + every IPv4 UDP flow unattributed, so executable-scoped Allows such as the + DNS, NTP and DHCP bootstrap rules refused. An absent table now counts as + empty. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 2f10a78..cba579d 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -454,6 +454,10 @@ fn proc_net_inode( /// A partial table read cannot establish uniqueness. Failure or expiry means /// unknown, including when the first table already contained one candidate. +/// An absent table is complete and empty: a kernel booted with +/// `ipv6.disable=1` (or built without IPv6) has no `/proc/net/udp6` and so +/// no IPv6 sockets, and reading that as a failure would leave every IPv4 UDP +/// flow unattributed. fn udp_inode_from_tables( tables: &[&str], local: (IpAddr, u16), @@ -466,7 +470,11 @@ fn udp_inode_from_tables( if Instant::now() > deadline { return None; } - let contents = fs::read_to_string(table).ok()?; + let contents = match fs::read_to_string(table) { + Ok(contents) => contents, + Err(e) if e.kind() == std::io::ErrorKind::NotFound => continue, + Err(_) => return None, + }; entries.extend(contents.lines().skip(1).filter_map(parse_table_line)); } if Instant::now() > deadline { @@ -1260,10 +1268,18 @@ mod tests { fs::write(&table4, format!("{HEADER}{}", line(local, remote, "01", 1))).unwrap(); let tables = [table4.to_str().unwrap(), table6.to_str().unwrap()]; let deadline = Instant::now() + Duration::from_secs(1); + // A table that exists but cannot be read leaves the search partial. + fs::create_dir(&table6).unwrap(); assert_eq!( udp_inode_from_tables(&tables, local, remote, Some(1000), deadline), None ); + // An absent table (ipv6.disable=1) holds no sockets at all. + fs::remove_dir(&table6).unwrap(); + assert_eq!( + udp_inode_from_tables(&tables, local, remote, Some(1000), deadline), + Some(1) + ); fs::write(&table6, HEADER).unwrap(); assert_eq!( diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 4da8559..567aa69 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -156,8 +156,9 @@ hundred microseconds before the packet's latency becomes visible. 2. **`/proc/net/{tcp,udp}{,6}` fallback**, silently, whenever the fast path misses. UDP always reads all relevant tables first and requires one unique compatible inode: exact or wildcard local address, with exact or zero - remote address. Missing tables, an exhausted lookup budget or several - compatible inodes leave attribution unknown. The packet's socket UID, + remote address. An unreadable table, an exhausted lookup budget or + several compatible inodes leave attribution unknown; an absent table + (`udp6` under `ipv6.disable=1`) counts as empty. The packet's socket UID, when present, filters candidates. All comparisons run on canonical form, so `::ffff:a.b.c.d` rows in the v6 tables match plain IPv4 flows - which is what dual-stack From fba3e770c138707bcbffee7dba14b916e37d7dbd Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:24:06 +0200 Subject: [PATCH 015/125] test(reject): pin the snippet's only mark-based accepts Any process can set a socket mark. The test fails if the outbound snippet accepts on a mark other than through the root-owned refusal exception, so a Fast Allow style mark accept cannot return unnoticed. --- crates/cfc-daemon/src/reject.rs | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/crates/cfc-daemon/src/reject.rs b/crates/cfc-daemon/src/reject.rs index d228f77..b759112 100644 --- a/crates/cfc-daemon/src/reject.rs +++ b/crates/cfc-daemon/src/reject.rs @@ -900,6 +900,31 @@ mod tests { const APP_PORT: u16 = 5555; const PEER_PORT: u16 = 80; + /// Any process can set a socket mark, so the shipped snippet may accept + /// on a mark only together with root ownership, and only for this + /// module's refusals. A Fast Allow style `meta mark ... accept` must not + /// come back. + #[test] + fn the_snippet_accepts_marks_only_for_root_refusals() { + let snippet = include_str!("../../../systemd/nftables-snippet.conf"); + let exception = format!("meta skuid 0 meta mark {REJECT_MARK:#x} "); + let marked: Vec<&str> = snippet + .lines() + .map(str::trim) + .filter(|l| !l.starts_with('#') && l.contains("mark")) + .filter(|l| !matches!(*l, "set fast_allow {" | "type mark")) + .collect(); + assert_eq!(marked.len(), 3, "{marked:#?}"); + for line in marked { + assert!(line.starts_with(&exception), "{line}"); + assert!( + line.ends_with("tcp flags & rst == rst accept") + || line.ends_with("type destination-unreachable accept"), + "{line}" + ); + } + } + /// Deliberately naive, independent one's-complement checksum used to /// cross-check [`checksum16`]. Written from RFC 1071 directly rather /// than shared with the implementation, so a bug in one shows up as a From f1cf338de84d6e4d50f369786209c5eddd619056 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:24:06 +0200 Subject: [PATCH 016/125] docs: inbound ICMP needs rules, and a stalled worker holds loopback README shows the rule that keeps ICMP monitoring working once inbound filtering is on. TROUBLESHOOTING says that bypass covers only a daemon that is not listening: a listening daemon with a stuck worker holds new loopback flows until the watchdog restarts it. --- README.md | 5 ++++- docs/TROUBLESHOOTING.md | 6 ++++++ 2 files changed, 10 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index 01806f1..605b223 100644 --- a/README.md +++ b/README.md @@ -263,7 +263,10 @@ managers: a failed nft load blocks their startup. A failed daemon start leaves the loaded tables dropping new flows. The daemon also requires the outbound table before initialization. Tables survive daemon stops and restarts; stop the nft unit explicitly to remove its table. Inbound stays opt-in. Its lockout -guard reads saved SQLite rules without a running daemon. +guard reads saved SQLite rules without a running daemon. With inbound +enabled, ping and other ICMP requests need a rule like any other inbound +flow, so allow your monitoring hosts: +`cfc rules add --direction in --action allow --protocol icmp --src-net --name monitoring`. This contract covers systemd-managed NetworkManager and systemd-networkd after enforcement is enabled. It does not cover networking configured in an diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index ece4bb8..9f296ef 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -259,6 +259,12 @@ it can answer from its cache or local records: its queries to the upstream servers are new non-loopback flows, so resolving anything else still needs the daemon. Every non-loopback new flow still meets the fail-closed rule. +`bypass` only covers a daemon that is not listening. A daemon that listens +but whose single worker is stuck (an executable on a hung mount being +hashed, for instance) leaves new loopback flows, local DNS included, +waiting in the same queue; once it fills they drop until the watchdog +restarts the daemon, which takes up to about 90 seconds. + If you carry an older copy of the snippet in your own `/etc/nftables.conf`, compare it with the shipped one: a copy without the loopback rule drops every new loopback flow whenever the daemon is down, and one with an From 89165090c0156068bcef2751d67caab4f06d5a9b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:30:37 +0200 Subject: [PATCH 017/125] fix(daemon): queue refusal audit rows instead of committing on the packet thread Every refused packet was committed to SQLite with an fsync on the single NFQUEUE thread before its verdict. A flood of refused traffic therefore stalled every new flow on the host, and a store mutex held for 250 ms or a failed commit (full disk) ended the worker and took networking down until systemd restarted the daemon, wiping until-restart rules on the way. The worker now verdicts first and pushes the refusal row into the event writer's bounded queue with try_send. Rows lost to a full queue, feeder lag or a failed batch commit share one counter and are logged; the "connection blocked" journal line still names each refusal. With nothing on the packet path waiting for the store, insert_events goes back to a plain lock. Docs describe the new audit guarantees, journald rate limiting and the storage startup failures. --- CHANGELOG.md | 7 ++ crates/cfc-daemon/src/ipc.rs | 172 ++++++++++++++++++++------ crates/cfc-daemon/src/main.rs | 9 +- crates/cfc-daemon/src/nfqueue.rs | 201 ++++++++++++------------------- crates/cfc-daemon/src/storage.rs | 23 +--- docs/ARCHITECTURE.md | 40 +++--- docs/HARDENING.md | 36 ++++-- docs/TROUBLESHOOTING.md | 37 +++++- systemd/daemon.toml.sample | 2 +- 9 files changed, 305 insertions(+), 222 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 98547ca..37a9878 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,6 +29,13 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). every IPv4 UDP flow unattributed, so executable-scoped Allows such as the DNS, NTP and DHCP bootstrap rules refused. An absent table now counts as empty. +- Every refused packet was committed to SQLite with an fsync on the single + packet thread before its verdict, so a flood of refused traffic stalled + every new flow on the machine, and a store mutex held for 250 ms (a long + `cfc log` query, the minute prune) or a full disk ended the daemon and + dropped all new connections until systemd restarted it. Refusals are now + queued after their verdict to the same bounded batch writer as Allow rows. + Rows it cannot take are counted and logged instead of stopping anything. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 3974121..ac62cc7 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -1143,51 +1143,96 @@ fn secure_socket(path: &Path, ipc: &IpcConfig) -> SocketAuth { // Event persistence pipeline // --------------------------------------------------------------------------- -/// Persists Allow observations. Deny/Reject are committed synchronously by -/// the NFQUEUE delivery gate before publication and must not be duplicated. +/// The bounded queue into the event writer. /// -/// Two tasks keep Allow observation writes off the datapath: +/// `push` never waits. The packet worker delivers its verdict first and +/// records it second, so a slow fsync, a long `ListEvents` holding the store +/// mutex or a full disk costs audit rows - counted and logged - and never +/// stalls or ends the datapath. Every refusal is also logged to the journal +/// as "connection blocked" before it is queued. +#[derive(Clone)] +pub struct EventSink { + tx: mpsc::Sender, + dropped: Arc, +} + +impl EventSink { + pub(crate) fn channel(depth: usize) -> (Self, mpsc::Receiver) { + let (tx, rx) = mpsc::channel(depth); + let sink = Self { + tx, + dropped: Arc::new(AtomicU64::new(0)), + }; + (sink, rx) + } + + /// Queues one row for the writer, or counts it as dropped. + pub fn push(&self, row: EventRow) { + if self.tx.try_send(row).is_err() { + count_dropped(&self.dropped, 1, "event log queue full"); + } + } + + #[cfg(test)] + pub(crate) fn dropped(&self) -> u64 { + self.dropped.load(Ordering::Relaxed) + } +} + +/// Adds `n` to the drop counter and warns on the first drop and every +/// `EVENT_DROP_LOG_EVERY` after it, rather than once per row. +fn count_dropped(dropped: &AtomicU64, n: u64, why: &str) { + let before = dropped.fetch_add(n, Ordering::Relaxed); + let total = before + n; + if before == 0 || before / EVENT_DROP_LOG_EVERY != total / EVENT_DROP_LOG_EVERY { + warn!( + dropped = total, + "{why}; events were not persisted (the packet path never waits for persistence)" + ); + } +} + +/// Starts the event persistence pipeline and returns the sink the packet +/// worker queues its refusals into. +/// +/// - Refusals are pushed straight into the bounded queue by the worker, so +/// they never depend on the lossy live feed. +/// - A *feeder* converts Allow observations from the live feed to +/// [`EventRow`]s and pushes them the same way. +/// - A *writer* drains the queue in batches of `EVENT_BATCH_ROWS` or every +/// second, whichever comes first, and trims the table to `max_rows` once a +/// minute. /// -/// - a *feeder* that converts broadcast items to [`EventRow`]s and -/// `try_send`s them into a bounded queue, counting (never awaiting on) -/// drops; -/// - a *writer* that drains the queue in batches of `EVENT_BATCH_ROWS` or -/// every second, whichever comes first, and trims the table to -/// `max_rows` once a minute. +/// Every row lost on the way (queue full, feeder lag, failed batch commit) +/// is counted in one counter and logged. pub fn spawn_event_pipeline( store: RuleStore, observed_tx: &broadcast::Sender, max_rows: u32, -) { - let (tx, rx) = mpsc::channel::(EVENT_QUEUE_DEPTH); +) -> EventSink { + let (sink, rx) = EventSink::channel(EVENT_QUEUE_DEPTH); let mut sub = observed_tx.subscribe(); + let feeder = sink.clone(); tokio::spawn(async move { - let dropped = AtomicU64::new(0); loop { match sub.recv().await { Ok(obs) => { if obs.verdict.action != cfc_core::Action::Allow { continue; } - let row = convert::event_row_from_observed( + feeder.push(convert::event_row_from_observed( &obs.connection, &obs.process, &obs.verdict, - ); - if tx.try_send(row).is_err() { - let n = dropped.fetch_add(1, Ordering::Relaxed) + 1; - if n == 1 || n.is_multiple_of(EVENT_DROP_LOG_EVERY) { - warn!( - dropped = n, - "event log queue full; dropping events (the packet path is \ - never blocked for persistence)" - ); - } - } + )); } Err(broadcast::error::RecvError::Lagged(n)) => { - warn!(missed = n, "event log feeder lagged behind the live feed"); + count_dropped( + &feeder.dropped, + n, + "event log feeder lagged behind the live feed", + ); } Err(broadcast::error::RecvError::Closed) => break, } @@ -1195,10 +1240,16 @@ pub fn spawn_event_pipeline( info!("event log feeder stopped: live feed closed"); }); - tokio::spawn(event_writer_task(store, rx, max_rows)); + tokio::spawn(event_writer_task(store, rx, sink.dropped.clone(), max_rows)); + sink } -async fn event_writer_task(store: RuleStore, mut rx: mpsc::Receiver, max_rows: u32) { +async fn event_writer_task( + store: RuleStore, + mut rx: mpsc::Receiver, + dropped: Arc, + max_rows: u32, +) { let mut batch: Vec = Vec::with_capacity(EVENT_BATCH_ROWS); let mut flush = tokio::time::interval(std::time::Duration::from_secs(EVENT_BATCH_INTERVAL_SECS)); @@ -1213,16 +1264,16 @@ async fn event_writer_task(store: RuleStore, mut rx: mpsc::Receiver, m Some(row) => { batch.push(row); if batch.len() >= EVENT_BATCH_ROWS { - write_batch(&store, &mut batch); + write_batch(&store, &mut batch, &dropped); } } None => { - write_batch(&store, &mut batch); + write_batch(&store, &mut batch, &dropped); info!("event log writer stopped: queue closed"); return; } }, - _ = flush.tick() => write_batch(&store, &mut batch), + _ = flush.tick() => write_batch(&store, &mut batch, &dropped), _ = prune.tick() => match store.prune_events(max_rows) { Ok(n) if n > 0 => tracing::debug!(removed = n, cap = max_rows, "pruned old events"), Ok(_) => {} @@ -1232,12 +1283,17 @@ async fn event_writer_task(store: RuleStore, mut rx: mpsc::Receiver, m } } -fn write_batch(store: &RuleStore, batch: &mut Vec) { +fn write_batch(store: &RuleStore, batch: &mut Vec, dropped: &AtomicU64) { if batch.is_empty() { return; } if let Err(e) = store.insert_events(batch) { - warn!(rows = batch.len(), "event log write failed: {e}"); + let total = dropped.fetch_add(batch.len() as u64, Ordering::Relaxed) + batch.len() as u64; + warn!( + rows = batch.len(), + dropped = total, + "event log write failed: {e:#}" + ); } batch.clear(); } @@ -1716,17 +1772,17 @@ mod tests { async fn event_pipeline_persists_the_live_feed() { let store = RuleStore::open_in_memory().unwrap(); let (tx, _rx) = broadcast::channel(64); - spawn_event_pipeline(store.clone(), &tx, 1000); + let sink = spawn_event_pipeline(store.clone(), &tx, 1000); tx.send(observed(443, cfc_core::Action::Allow)).unwrap(); + // The worker queues a refusal itself and then publishes it; the + // feeder must not record it a second time. let blocked = observed(80, cfc_core::Action::Deny); - store - .insert_events(&[convert::event_row_from_observed( - &blocked.connection, - &blocked.process, - &blocked.verdict, - )]) - .unwrap(); + sink.push(convert::event_row_from_observed( + &blocked.connection, + &blocked.process, + &blocked.verdict, + )); tx.send(blocked).unwrap(); // Well past the batch interval; paused time auto-advances. @@ -1777,6 +1833,42 @@ mod tests { assert_eq!(rows.len(), 2, "table should be trimmed to max_rows"); } + #[tokio::test(start_paused = true)] + async fn a_failing_store_costs_counted_rows_never_a_stall() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("events.db"); + let store = RuleStore::open(&path).unwrap(); + rusqlite::Connection::open(&path) + .unwrap() + .execute_batch( + "CREATE TRIGGER refuse_audit BEFORE INSERT ON events \ + BEGIN SELECT RAISE(ABORT, 'disk full'); END;", + ) + .unwrap(); + let (tx, _rx) = broadcast::channel(64); + let sink = spawn_event_pipeline(store.clone(), &tx, 1000); + let blocked = observed(80, cfc_core::Action::Deny); + let row = convert::event_row_from_observed( + &blocked.connection, + &blocked.process, + &blocked.verdict, + ); + for _ in 0..3 { + sink.push(row.clone()); + } + tokio::time::sleep(std::time::Duration::from_secs( + EVENT_BATCH_INTERVAL_SECS + 1, + )) + .await; + assert_eq!(sink.dropped(), 3); + + // A full queue drops the row at once instead of waiting for room. + let (full, _rx) = EventSink::channel(1); + full.push(row.clone()); + full.push(row); + assert_eq!(full.dropped(), 1); + } + #[test] fn recording_the_same_peer_twice_does_not_grow_the_queue() { let a = PromptAudience::default(); diff --git a/crates/cfc-daemon/src/main.rs b/crates/cfc-daemon/src/main.rs index 7f260f5..2a574af 100644 --- a/crates/cfc-daemon/src/main.rs +++ b/crates/cfc-daemon/src/main.rs @@ -162,9 +162,10 @@ async fn run() -> anyhow::Result<()> { let (verdict_tx, verdict_rx) = std::sync::mpsc::channel(); let router = prompts::PromptRouter::new(policy.clone(), stats.clone(), verdict_tx); - // Allow observations use the bounded async event pipeline. NFQUEUE - // refusals commit synchronously before verdict delivery and publication. - ipc::spawn_event_pipeline(store.clone(), &observed_tx, cfg.events.max_rows); + // Every verdict is persisted by the bounded async event pipeline: the + // worker queues its refusals into `events` after the verdict, and Allow + // rows come off the live feed. No packet waits for the database. + let events = ipc::spawn_event_pipeline(store.clone(), &observed_tx, cfg.events.max_rows); let (mut ipc_handle, prompt_tx) = ipc::spawn( ipc::IpcOptions { @@ -196,7 +197,7 @@ async fn run() -> anyhow::Result<()> { verdict_rx, observed_tx.clone(), stats.clone(), - store.clone(), + events, // Cloned rather than moved: the eBPF consumers write observed DNS // answers into the same cache, and they are started after READY=1 // (see below) so the handle has to outlive this call. diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index f119df1..4fb5769 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -64,11 +64,11 @@ use crate::config::NfqConfig; use crate::decision::{Decision, Engine}; use crate::dns::DnsCache; +use crate::ipc::EventSink; use crate::packet; use crate::process_resolve; use crate::reject::Rejecter; use crate::stats::Stats; -use crate::storage::RuleStore; use anyhow::Context as _; use cfc_core::{Action, Connection, Direction, Process, Protocol, Verdict}; use nfq::{Message, Queue, Verdict as NfqVerdict}; @@ -204,7 +204,8 @@ pub struct ObservedConnection { pub verdict: Verdict, } -/// Record refusals before the bounded, lossy live-feed channel. +/// Logs a refusal to the journal, then publishes to the bounded, lossy live +/// feed. pub fn publish_observation(tx: &broadcast::Sender, obs: ObservedConnection) { if obs.verdict.action != Action::Allow { info!( @@ -350,7 +351,7 @@ pub fn spawn( verdict_rx: VerdictRx, observed_tx: broadcast::Sender, stats: Stats, - store: RuleStore, + events: EventSink, dns_cache: DnsCache, ) -> anyhow::Result { anyhow::ensure!( @@ -430,7 +431,7 @@ pub fn spawn( verdict_rx, observed_tx, stats, - store, + events, dns: Box::new(dns_cache), resolver: Box::new(ProcfsResolver), waiters: HashMap::new(), @@ -559,7 +560,8 @@ struct Worker { verdict_rx: VerdictRx, observed_tx: broadcast::Sender, stats: Stats, - store: RuleStore, + /// Refusal rows go here, after the verdict; see [`Worker::deliver`]. + events: EventSink, /// Reverse-DNS / self-identification seam; [`DnsCache`] in production. dns: Box, /// Process attribution seam; [`ProcfsResolver`] in production. @@ -575,9 +577,9 @@ struct Worker { /// Set by main's shutdown path; observed at the top of every iteration. stop: Arc, /// Watchdog liveness cell shared with main's heartbeat task: the - /// unix-ms at which the loop last turned. The idle wait is bounded; - /// packet processing also includes the durable audit commit. A stale - /// stamp means main must withhold the WATCHDOG=1 heartbeat. + /// unix-ms at which the loop last turned. The idle wait is bounded and + /// nothing on this thread waits for storage. A stale stamp means main + /// must withhold the WATCHDOG=1 heartbeat. last_activity: Arc, } @@ -901,25 +903,24 @@ impl Worker { ) } - /// Commit parsed refusals before releasing the packet or publishing it. - /// Storage failure drops this packet and propagates out of the worker, - /// so a later packet cannot receive Allow after an unaudited refusal. + /// Verdicts the packet, then records it. A refusal row goes straight + /// into the event writer's bounded queue rather than through the lossy + /// live feed. Nothing here waits for storage: an fsync per refusal on + /// this single thread let a deny flood stall every new flow on the + /// machine, and a failed commit used to end the worker, which took all + /// networking down until systemd restarted the daemon. A row the queue + /// cannot take is counted and logged by [`EventSink`]; the journal line + /// from [`publish_observation`] still names the refusal. fn deliver(&mut self, msg: Q::Msg, obs: ObservedConnection) -> anyhow::Result<()> { - if obs.verdict.action != Action::Allow { - let row = crate::convert::event_row_from_observed( - &obs.connection, - &obs.process, - &obs.verdict, - ); - if let Err(e) = self.store.insert_events(&[row]) { - self.send_verdict(msg, NfqVerdict::Drop) - .context("dropping packet after verdict audit failure")?; - return Err(e).context("committing verdict audit before NFQUEUE delivery"); - } - } self.apply_action(msg, obs.verdict.action)?; if obs.verdict.action == Action::Allow { self.dns.enqueue(obs.connection.dst_ip); + } else { + self.events.push(crate::convert::event_row_from_observed( + &obs.connection, + &obs.process, + &obs.verdict, + )); } record(&self.stats, obs.verdict.action); publish_observation(&self.observed_tx, obs); @@ -1057,7 +1058,7 @@ enum PacketOutcome { /// Immediate verdict with nothing to observe (self traffic, packets we /// can't parse). Not counted in stats. Silent(NfqVerdict), - /// Parsed decision, committed and counted by the final delivery gate. + /// Parsed decision, verdicted, recorded and counted by [`Worker::deliver`]. Deliver { connection: Connection, process: Process, @@ -1975,7 +1976,7 @@ mod tests { .unwrap() .handle_message(FakeMsg::new(7, tcp_packet(443))) .unwrap(); - let rows = h.store.query_events(10, 0, Default::default()).unwrap(); + let rows: Vec<_> = std::iter::from_fn(|| h.events.try_recv().ok()).collect(); assert_eq!( rows.len(), 1, @@ -2276,8 +2277,6 @@ mod tests { modes: Vec, /// recv calls, so a test can tell a turning loop from a stuck one. recv_calls: u64, - audit_probe: Option, - audited_at_verdict: Vec, } struct FakeQueue { @@ -2303,15 +2302,7 @@ mod tests { fn verdict(&mut self, msg: FakeMsg) -> std::io::Result<()> { let verdict = msg.verdict.expect("worker verdicted without a verdict"); - let mut log = self.log.lock().unwrap(); - if let Some(store) = &log.audit_probe { - let count = store - .query_events(100, 0, Default::default()) - .unwrap() - .len(); - log.audited_at_verdict.push(count); - } - log.verdicts.push((msg.id, verdict)); + self.log.lock().unwrap().verdicts.push((msg.id, verdict)); Ok(()) } } @@ -2341,7 +2332,8 @@ mod tests { verdict_tx: Option, prompt_rx: mpsc::Receiver, observed_rx: broadcast::Receiver, - store: crate::storage::RuleStore, + /// The event writer's end of the worker's [`EventSink`]. + events: mpsc::Receiver, } impl LoopHarness { @@ -2355,7 +2347,7 @@ mod tests { let (verdict_tx, verdict_rx) = std::sync::mpsc::channel(); let (observed_tx, observed_rx) = broadcast::channel(16); let stats = Stats::new(); - let store = RuleStore::open_in_memory().unwrap(); + let (sink, events) = EventSink::channel(256); let stop = Arc::new(AtomicBool::new(false)); let worker = Worker { queue: FakeQueue { @@ -2369,7 +2361,7 @@ mod tests { verdict_rx, observed_tx, stats: stats.clone(), - store: store.clone(), + events: sink, dns: Box::new(StubDns { self_pid: None, host: None, @@ -2396,17 +2388,22 @@ mod tests { verdict_tx: Some(verdict_tx), prompt_rx, observed_rx, - store, + events, } } - fn with_store(mut self, store: crate::storage::RuleStore) -> Self { - self.log.lock().unwrap().audit_probe = Some(store.clone()); - self.worker().store = store.clone(); - self.store = store; + fn with_event_queue_depth(mut self, depth: usize) -> Self { + let (sink, events) = EventSink::channel(depth); + self.worker().events = sink; + self.events = events; self } + /// Every row the worker has queued for the event writer so far. + fn audited(&mut self) -> Vec { + std::iter::from_fn(|| self.events.try_recv().ok()).collect() + } + fn with_tuning(mut self, tuning: Tuning) -> Self { self.worker().tuning = tuning; self @@ -2487,25 +2484,22 @@ mod tests { } #[test] - fn parsed_refusals_commit_before_the_verdict_and_live_feed() { - let dir = tempfile::tempdir().unwrap(); - let path = dir.path().join("audit.db"); - let store = crate::storage::RuleStore::open(&path).unwrap(); + fn parsed_refusals_are_queued_for_the_event_writer() { let mut reject = deny_port_rule(80); reject.action = Action::Reject; - let mut h = LoopHarness::new(vec![], vec![deny_port_rule(443), reject], dp_deny()) - .with_store(store.clone()); + let mut h = LoopHarness::new(vec![], vec![deny_port_rule(443), reject], dp_deny()); h.worker() .handle_message(FakeMsg::new(1, tcp_packet(443))) .unwrap(); h.worker() .handle_message(FakeMsg::new(2, tcp_packet(80))) .unwrap(); - assert_eq!(h.log.lock().unwrap().audited_at_verdict, vec![1, 2]); assert_eq!( h.verdicts(), vec![(1, NfqVerdict::Drop), (2, NfqVerdict::Drop)] ); + let actions: Vec<_> = h.audited().into_iter().map(|r| r.action).collect(); + assert_eq!(actions, vec!["Deny", "Reject"]); assert_eq!( h.observed_rx.try_recv().unwrap().verdict.action, Action::Deny @@ -2514,65 +2508,40 @@ mod tests { h.observed_rx.try_recv().unwrap().verdict.action, Action::Reject ); - let reader = rusqlite::Connection::open(&path).unwrap(); - let count: i64 = reader - .query_row("SELECT COUNT(*) FROM events", [], |r| r.get(0)) - .unwrap(); - assert_eq!( - count, 2, - "commits must be visible to an independent connection" - ); - drop(reader); - drop(h); - drop(store); - assert_eq!( - crate::storage::RuleStore::open(&path) - .unwrap() - .query_events(10, 0, Default::default()) - .unwrap() - .len(), - 2 - ); } #[test] - fn failed_audit_commit_drops_current_packet_and_stops_before_next_allow() { - let dir = tempfile::tempdir().unwrap(); - let path = dir.path().join("audit.db"); - let store = crate::storage::RuleStore::open(&path).unwrap(); - let writer = rusqlite::Connection::open(&path).unwrap(); - writer - .execute_batch( - "CREATE TRIGGER refuse_audit BEFORE INSERT ON events \ - BEGIN SELECT RAISE(ABORT, 'audit unavailable'); END;", - ) - .unwrap(); + fn a_full_event_queue_costs_the_row_not_the_packet_or_the_worker() { let mut h = LoopHarness::new( vec![ Ok(FakeMsg::new(1, tcp_packet(443))), - Ok(FakeMsg::new(2, tcp_packet(80))), + Ok(FakeMsg::new(2, tcp_packet(443))), + Ok(FakeMsg::new(3, tcp_packet(80))), ], vec![deny_port_rule(443), allow_port_rule(80)], dp_deny(), ) - .with_store(store); + .with_event_queue_depth(1); + let sink = h.worker().events.clone(); let done = h.start(); - let result = done.recv_timeout(Duration::from_secs(1)); - if result.is_err() { - h.stop_and_expect_ok(&done); - } - assert!(result - .expect("audit failure must terminate the worker") - .is_err()); - assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); - assert_eq!(h.stats.connections_total(), 0); - assert!(h.observed_rx.try_recv().is_err(), "no unaudited live entry"); + assert!(wait_until(|| h.verdicts().len() == 3)); + h.stop_and_expect_ok(&done); + assert_eq!( + h.verdicts(), + vec![ + (1, NfqVerdict::Drop), + (2, NfqVerdict::Drop), + (3, NfqVerdict::Accept) + ] + ); + assert_eq!(sink.dropped(), 1); + assert_eq!(h.audited().len(), 1); + assert_eq!(h.stats.connections_total(), 3); } #[test] fn prompt_refusals_and_unavailable_router_use_the_same_audit_gate() { - let store = crate::storage::RuleStore::open_in_memory().unwrap(); - let mut h = LoopHarness::new(vec![], vec![], dp_deny()).with_store(store.clone()); + let mut h = LoopHarness::new(vec![], vec![], dp_deny()); h.worker() .handle_message(FakeMsg::new(1, tcp_packet(443))) .unwrap(); @@ -2587,8 +2556,11 @@ mod tests { h.worker() .handle_message(FakeMsg::new(2, tcp_packet(80))) .unwrap(); - assert_eq!(h.log.lock().unwrap().audited_at_verdict, vec![1, 2]); - let rows = store.query_events(10, 0, Default::default()).unwrap(); + assert_eq!( + h.verdicts(), + vec![(1, NfqVerdict::Drop), (2, NfqVerdict::Drop)] + ); + let rows = h.audited(); assert!(rows.iter().any(|r| r.action == "Reject")); assert!(rows.iter().any(|r| r.action == "Deny")); assert_eq!(h.stats.connections_denied(), 2); @@ -2635,23 +2607,22 @@ mod tests { } #[test] - fn loopback_closed_rules_are_audited_before_release_even_when_paused() { + fn loopback_closed_rules_are_audited_even_when_paused() { for action in [Action::Deny, Action::Reject] { let mut scope = RuleScope::any(); scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); let rule = Rule::new("local application refusal", action, scope); - let store = RuleStore::open_in_memory().unwrap(); - let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()).with_store(store); + let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()); h.stats.set_paused(true); let mut payload = tcp_packet(53); // Policy needs only ports. No complete TCP header keeps refusal - // injection inert while testing the real verdict and audit gate. + // injection inert while testing the real verdict and audit path. payload.truncate(24); let mut msg = FakeMsg::new(1, payload); msg.outdev = 1; h.worker().handle_message(msg).unwrap(); assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); - assert_eq!(h.log.lock().unwrap().audited_at_verdict, vec![1]); + assert_eq!(h.audited().len(), 1); assert_eq!(h.observed_rx.try_recv().unwrap().verdict.action, action); assert!(h.prompt_rx.try_recv().is_err()); } @@ -2662,8 +2633,7 @@ mod tests { let mut scope = RuleScope::any(); scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); let rule = Rule::new("local application refusal", Action::Deny, scope); - let store = RuleStore::open_in_memory().unwrap(); - let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()).with_store(store); + let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()); h.worker().resolver = Box::new(StubResolver { pid: None, process: Process::unknown(0), @@ -2674,7 +2644,7 @@ mod tests { msg.outdev = 1; h.worker().handle_message(msg).unwrap(); assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); - assert_eq!(h.log.lock().unwrap().audited_at_verdict, vec![1]); + assert_eq!(h.audited().len(), 1); assert!(h.prompt_rx.try_recv().is_err()); } @@ -2709,9 +2679,7 @@ mod tests { } } let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); - let store = RuleStore::open_in_memory().unwrap(); - let mut h = LoopHarness::new(vec![], vec![allow_port_rule(443)], dp_deny()) - .with_store(store.clone()); + let mut h = LoopHarness::new(vec![], vec![allow_port_rule(443)], dp_deny()); h.worker().dns = Box::new(CountingDns(calls.clone())); h.worker() .handle_message(FakeMsg::new(1, tcp_packet(443))) @@ -2723,17 +2691,12 @@ mod tests { h.observed_rx.try_recv().unwrap().verdict.action, Action::Allow ); - assert!(store - .query_events(10, 0, Default::default()) - .unwrap() - .is_empty()); + assert!(h.audited().is_empty(), "Allow rows come off the live feed"); } #[test] fn live_feed_lag_does_not_lose_parsed_refusal_audits() { - let store = RuleStore::open_in_memory().unwrap(); - let mut h = LoopHarness::new(vec![], vec![deny_port_rule(443)], dp_deny()) - .with_store(store.clone()); + let mut h = LoopHarness::new(vec![], vec![deny_port_rule(443)], dp_deny()); for id in 0..64 { h.worker() .handle_message(FakeMsg::new(id, tcp_packet(443))) @@ -2743,13 +2706,7 @@ mod tests { h.observed_rx.try_recv(), Err(broadcast::error::TryRecvError::Lagged(_)) )); - assert_eq!( - store - .query_events(100, 0, Default::default()) - .unwrap() - .len(), - 64 - ); + assert_eq!(h.audited().len(), 64); assert_eq!(h.verdicts().len(), 64); } diff --git a/crates/cfc-daemon/src/storage.rs b/crates/cfc-daemon/src/storage.rs index 1d9c4fc..7513726 100644 --- a/crates/cfc-daemon/src/storage.rs +++ b/crates/cfc-daemon/src/storage.rs @@ -450,10 +450,7 @@ impl RuleStore { if batch.is_empty() { return Ok(()); } - let conn = self - .conn - .try_lock_for(std::time::Duration::from_millis(250)) - .context("audit storage mutex unavailable within 250ms")?; + let conn = self.conn.lock(); let tx = conn.unchecked_transaction()?; { let mut stmt = tx.prepare_cached( @@ -623,24 +620,6 @@ impl RuleStore { #[cfg(test)] mod tests { - #[test] - fn event_commit_does_not_wait_indefinitely_for_the_store_mutex() { - let store = RuleStore::open_in_memory().unwrap(); - let held = store.conn.lock(); - let writer = store.clone(); - let (tx, rx) = std::sync::mpsc::channel(); - let task = std::thread::spawn(move || { - tx.send(writer.insert_events(&[sample_event(1, "test", "Deny")])) - .unwrap(); - }); - let result = rx.recv_timeout(std::time::Duration::from_secs(1)); - drop(held); - task.join().unwrap(); - assert!(result - .expect("audit lock acquisition must be bounded") - .is_err()); - } - #[test] fn production_storage_requires_a_file_and_bounded_sqlite_contention() { assert!(RuleStore::open(Path::new(":memory:")).is_err()); diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 567aa69..fe7a32d 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -276,22 +276,28 @@ merged at the very last step, when the kernel is told to DROP. ## Event log -Parsed NFQUEUE policy refusals commit to SQLite with WAL/FULL before verdict -delivery, the journald message and live publication. Commit failure drops the -current packet and ends the worker, so later queued packets cannot be allowed -by that worker after an unaudited refusal. +Every verdict is persisted off the packet path. The worker verdicts a parsed +refusal first, logs it to the journal, then queues its row straight into the +event writer's bounded queue with `try_send`. Allow rows reach the same queue +through a feeder on the live feed. The writer commits in batches with WAL and +synchronous=FULL. ``` -parsed Deny/Reject --> durable events commit --> verdict --> journal/live feed +parsed Deny/Reject --> verdict --> journal --> bounded queue --> async writer + \-> live feed Allow --> verdict --> live feed --> bounded queue --> async writer ``` -The live feed and async Allow history can lose observations under load. The -feeder uses `try_send` and logs lag or drops; it skips already committed refusals. -Refusal commits can delay delivery. Database mutex and SQLite busy waits are -each limited to 250 ms; filesystem I/O and fsync are not bounded by these limits. -This gate does not audit malformed packets, nftables drops or kernel-ring -refusals. It is not a universal lossless audit or protection against root +Nothing on the worker thread waits for the database. An fsync per refusal +there let a flood of refused packets stall every new flow on the machine, and +a failed or slow commit used to end the worker, which dropped all new +non-loopback traffic until systemd restarted the daemon. Now a full queue, +feeder lag or a failed batch commit (full disk, I/O error) costs rows instead: +they are counted and logged ("events were not persisted", "event log write +failed"), and the journal line still names each refusal. Rows still in the +writer's batch, at most about a second of them, are lost on a crash, a power +cut or a stop. This does not audit malformed packets, nftables drops or +kernel-ring refusals. It is not a lossless audit or protection against root rewriting the database. The table is pruned to `[events] max_rows` every 60 seconds. `ListEvents` queries it with executable-substring, action and since filters; `cfc log` is @@ -340,14 +346,12 @@ with the calling uid and pid. See [HARDENING.md](HARDENING.md). Prompts go out on a broadcast channel; verdicts come back on a dedicated channel the worker polls. - **ipc server** - tonic gRPC over the Unix socket. -- **event writer** - batches Allow observations into SQLite and prunes on a timer. - Parsed NFQUEUE Deny/Reject decisions commit synchronously before their verdict - and live publication. Audit failure drops the packet and ends the worker. +- **event writer** - batches every verdict into SQLite and prunes on a timer. + The worker queues refusals into it after their verdict and never waits for + it; rows it cannot take are counted and logged. - **storage** - sqlite behind a mutex. Reads are served from the in-memory - `RuleSet`. Production startup requires WAL with synchronous=FULL. Refusal - commits are on the packet path; lock and SQLite busy waits are each limited - to 250 ms. These limits do not bound filesystem I/O or fsync. Kernel/nftables - drops and malformed packets are not covered by this durable delivery gate. + `RuleSet`. Production startup requires WAL with synchronous=FULL. The packet + path never touches it. ## Lifecycle and systemd integration diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 5ed20d0..cb1bb2b 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -328,11 +328,11 @@ log the action, its source, the executable, pid, uid and destination: journalctl -u colony-firewalld -g 'connection blocked' ``` -This line is emitted after the refusal row commits and verdict delivery succeeds. -A failed commit drops the packet and ends the worker before live publication. +This line is emitted once the verdict is delivered, before the row is queued +for the database. -**3. The events table.** Parsed NFQUEUE refusals commit synchronously; -Allow observations use best-effort asynchronous persistence. Query with `cfc log`: +**3. The events table.** Every verdict is written by an asynchronous writer +that commits in batches; no packet waits for it. Query with `cfc log`: ```sh cfc log --since 24h --action deny @@ -340,11 +340,27 @@ cfc log --exe firefox --limit 200 cfc log --json --since 1h | jq -r '.[] | .dst_host // .dst_ip' | sort | uniq -c ``` -Refusal commits can delay a verdict. WAL and synchronous=FULL are required at -startup. Mutex and SQLite busy waits each have a 250 ms limit, which does not -bound filesystem I/O or fsync. Allow rows can be dropped when their queue fills; -that loss is logged. Malformed packets, nftables drops and in-kernel refusals -are not covered by this durable NFQUEUE gate. Retention is a row cap, not a time window: +WAL and synchronous=FULL are required at startup. Rows are lost, never +waited for, when the writer's queue is full or a batch commit fails (a full +disk, an I/O error); the loss is counted and logged: + +```sh +journalctl -u colony-firewalld -g 'events were not persisted|event log write failed' +``` + +Up to about a second of rows still waiting for their batch is lost on a crash, +a power cut or a stop. Malformed packets, nftables drops and in-kernel refusals +are not recorded at all. + +Both journal sources share journald's per-unit rate limit (by default 10000 +messages per 30 seconds). A sustained flood of refused packets can exceed it, +and journald then drops this unit's lines for the rest of that interval, +including the mutating-RPC lines above; it logs how many it suppressed. Raise +`LogRateLimitIntervalSec=`/`LogRateLimitBurst=` in a drop-in for the unit if +that trail matters more than journal volume. The same flood fills the events +table, whose oldest rows the row cap below evicts. + +Retention is a row cap, not a time window: `[events] max_rows` (default 100000), pruned every 60 seconds. Raise it if you want a longer history, and remember the table lives in `/var/lib/colony-firewall/rules.db` - back it up or ship it off the host @@ -434,7 +450,7 @@ without `bypass` - is in Read it before enabling enforcement on a machine you only reach over SSH. `[nfqueue] fail_open` must be `false`; `true` is rejected. Queue overflow -must drop traffic instead of bypassing policy and durable refusal auditing. +must drop traffic instead of bypassing policy and refusal auditing. The nftables `bypass` keyword governs missing listeners; the shipped snippet uses it only on the loopback rule (`oifname "lo"`). diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 9f296ef..d690cca 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -148,7 +148,7 @@ improvement working, and the reason is in the journal: journalctl -u colony-firewalld -b --no-pager | tail -40 ``` -The daemon prints hint lines next to the failure. Three causes: +The daemon prints hint lines next to the failure. Four causes: **Missing capability.** `failed to open NFQUEUE socket: ...` followed by a `CAP_NET_ADMIN` hint. Run it via the bundled unit rather than by hand; @@ -175,6 +175,24 @@ Either stop the other consumer, or move this daemon to a free number in nftables rule. The two must agree or you get the same lockout as a dead daemon. +**The rule database cannot be opened.** Startup opens +`/var/lib/colony-firewall/rules.db` (`[storage] path`) before anything +else, so these fail on every start: + +```sh +journalctl -u colony-firewalld -b -g 'opening rule store|durable storage requires|newer than this daemon supports' +``` + +- `durable storage requires WAL` or `synchronous=FULL`: the path is on a + filesystem that cannot hold a WAL journal (a network share, for one), or + outside the unit's `ReadWritePaths`. Keep `[storage] path` on local disk + under `/var/lib/colony-firewall`. +- `newer than this daemon supports`: the package was downgraded. Reinstall + the newer one, or restore a backup of `rules.db` that the older version + wrote. +- Any other error under `opening rule store` or `purging transient rules`: + usually a full `/var`. Free space and start the daemon again. + Once it starts cleanly the unit reports ready only after both the queue and the control socket are bound, so `systemctl is-active` genuinely means "filtering". @@ -462,8 +480,8 @@ Watchdog timeout (limit 30s)! ``` in the journal means the worker stopped responding, not that the machine -was idle - a worker parked in a blocking `recv` with nothing to do is -explicitly treated as healthy, so an idle system is never killed. Look +was idle - an idle worker still wakes every few milliseconds to check +for work, and that counts as progress, so an idle system is never killed. Look for the daemon's own complaint just before the restart: ```sh @@ -480,8 +498,17 @@ for a stalled one, and under the fail-closed nftables rule a stalled daemon is a dead network. Restarts *without* a watchdog message are ordinary failures - -`Restart=on-failure` retrying a bind that keeps failing. See "The daemon -exits immediately" above. +`Restart=on-failure` retrying a start that keeps failing, such as a queue +bind or the rule database. See "The daemon exits immediately" above. A +full disk at runtime does not restart the daemon: it costs event-log rows, +which the journal reports as `event log write failed`. + +If manual restarts pile on top of the automatic ones, systemd can give up +with `start request repeated too quickly`. Fix the cause, then: + +```sh +sudo systemctl reset-failed colony-firewalld && sudo systemctl start colony-firewalld +``` ## Some rules are not being enforced diff --git a/systemd/daemon.toml.sample b/systemd/daemon.toml.sample index bc34c3c..4801e2d 100644 --- a/systemd/daemon.toml.sample +++ b/systemd/daemon.toml.sample @@ -98,7 +98,7 @@ queue_max_len = 4096 # What the KERNEL does with packets when this queue overflows: false drops # them (fail-closed, default), true lets them through unfiltered # (fail-open). This setting must be false: true is rejected so overflow -# cannot bypass policy or durable refusal auditing. +# cannot bypass policy or refusal auditing. # # Not to be confused with the `bypass` keyword on the nftables rule, which # decides what happens when NO daemon is attached to the queue at all. This From a080d9d6295a6cfea6de37712598d95961fc976c Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:32:26 +0200 Subject: [PATCH 018/125] fix(daemon): keep disabled and quarantined rules reachable after a restart The startup load skipped disabled rows, although the engine keeps disabled rules at runtime and every lookup ignores them. After a restart a paused rule vanished from ListRules, so no client could re-enable or remove it, and import --replace deleted it unseen. All rows now load. Quarantined rows were only named in the journal: skipped_rules now counts them, so cfc status and the GUI warn, and cfc rules remove accepts a full id the daemon does not list, which is the remedy the journal line gives. --- CHANGELOG.md | 6 ++++++ crates/cfc-cli/src/rules.rs | 30 +++++++++++++++------------- crates/cfc-cli/tests/cli_e2e.rs | 27 +++++++++++++++++++++++++ crates/cfc-daemon/src/storage.rs | 34 ++++++++++++++++++++------------ crates/cfc-proto/proto/cfc.proto | 6 ++++-- docs/TROUBLESHOOTING.md | 16 ++++++++++++++- 6 files changed, 89 insertions(+), 30 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 37a9878..8402ecf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -36,6 +36,12 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). dropped all new connections until systemd restarted it. Refusals are now queued after their verdict to the same bounded batch writer as Allow rows. Rows it cannot take are counted and logged instead of stopping anything. +- A disabled rule vanished from `cfc rules list` and the GUI after a daemon + restart, so it could not be re-enabled or removed. Disabled rules now load + at startup; lookups already skip them. +- Rules quarantined at load were reported only in the journal. `cfc status` + and the GUI now count them with the rows that could not be loaded, and + `cfc rules remove ` with the full id deletes such an unlisted row. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 11e0e57..96b1535 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -232,26 +232,28 @@ pub async fn show(client: &mut Client, needle: &str, format: OutputFormat) -> Cl } pub async fn remove(client: &mut Client, needle: &str, format: OutputFormat) -> CliResult { - let rule = resolve_via_daemon(client, needle).await?; - let deleted = client.delete_rule(&rule.id).await?; + let (id, name) = match resolve_via_daemon(client, needle).await { + Ok(rule) => (rule.id, rule.name), + // A quarantined or unreadable row is never listed, so nothing + // resolves to it, but the daemon deletes it by the full id the + // journal names. + Err(CliError::NotFound(_)) if uuid::Uuid::parse_str(needle).is_ok() => { + (needle.to_string(), String::new()) + } + Err(e) => return Err(e), + }; + let deleted = client.delete_rule(&id).await?; if !deleted { - // The rule was listed a moment ago, so this is a race with another - // client rather than a typo - still "not found" for the caller. - return Err(CliError::not_found(format!( - "rule {} disappeared before it could be deleted", - rule.id - ))); + // Either a race with another client or an id that was never there: + // "not found" for the caller either way. + return Err(CliError::not_found(format!("no rule with id {id}"))); } if format.is_json() { return output::print_json(&serde_json::json!({ - "deleted": true, "id": rule.id, "name": rule.name, + "deleted": true, "id": id, "name": name, })); } - println!( - "deleted {} ({})", - rule.id, - output::terminal_safe(&rule.name) - ); + println!("deleted {} ({})", id, output::terminal_safe(&name)); Ok(()) } diff --git a/crates/cfc-cli/tests/cli_e2e.rs b/crates/cfc-cli/tests/cli_e2e.rs index 60b027f..a9ca0c4 100644 --- a/crates/cfc-cli/tests/cli_e2e.rs +++ b/crates/cfc-cli/tests/cli_e2e.rs @@ -1075,6 +1075,33 @@ async fn removing_a_bundle_preserves_a_manual_rule_with_the_same_name() { std::fs::remove_dir_all(dir).unwrap(); } +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn remove_by_full_id_reaches_a_rule_the_daemon_does_not_list() { + // A quarantined row is not in ListRules; the journal names its id. + let socket = socket_path("remove-unlisted"); + let fake = FakeDaemon::default(); + let calls = fake.calls.clone(); + let server = serve(socket.clone(), fake).await; + let socket_arg = socket.to_string_lossy().into_owned(); + let id = "33333333-3333-4333-8333-333333333333"; + let out = tokio::task::spawn_blocking(move || { + run_cli( + &["--socket", &socket_arg, "rules", "remove", id], + Duration::from_secs(5), + ) + }) + .await + .unwrap(); + assert!( + out.status.success(), + "{}", + String::from_utf8_lossy(&out.stderr) + ); + assert_eq!(*calls.lock().unwrap(), vec![Call::Delete(id.into())]); + server.abort(); + let _ = std::fs::remove_file(socket); +} + #[tokio::test] async fn a_partial_opensnitch_replace_changes_nothing() { let dir = std::env::temp_dir().join(format!("cfc-test-{}", uuid::Uuid::new_v4())); diff --git a/crates/cfc-daemon/src/storage.rs b/crates/cfc-daemon/src/storage.rs index 7513726..8fdec52 100644 --- a/crates/cfc-daemon/src/storage.rs +++ b/crates/cfc-daemon/src/storage.rs @@ -102,9 +102,9 @@ fn heal_legacy_protocol(rule: &mut Rule) -> bool { /// /// A rule caught here is quarantined, not deleted: the row is the operator's, /// and `cfc rules list` is not where a firewall should silently lose things. -/// It is also not loaded disabled-but-present - the engine snapshot holds -/// only enabled rules - so "not applied, named in the log, preserved on -/// disk" is the whole contract. Applying it is the bug, and most of these +/// It is also not loaded disabled-but-present - a quarantined row is kept +/// out of the engine altogether - so "not applied, named in the log, +/// preserved on disk" is the whole contract. Applying it is the bug, and most of these /// shapes cannot even be edited away: every client edit is a /// read-modify-write that sends the refused scope straight back to a daemon /// that now rejects it. @@ -212,11 +212,14 @@ impl RuleStore { pub fn snapshot(&self) -> anyhow::Result { let conn = self.conn.lock(); + // Disabled rules load too: every engine lookup skips them, and the + // engine keeps them after a runtime disable, so a restart must not + // hide a paused rule from `cfc rules list` and from re-enabling. + // // Deterministic load order. `created_at` lives inside the JSON blob // (the table has no timestamp column), so order by `id`: stable // across restarts, and the in-memory sort handles priority ordering. - let mut stmt = - conn.prepare("SELECT id, data FROM rules WHERE enabled = 1 ORDER BY id ASC")?; + let mut stmt = conn.prepare("SELECT id, data FROM rules ORDER BY id ASC")?; let rows = stmt.query_map([], |row| { let id: String = row.get(0)?; let json: String = row.get(1)?; @@ -286,14 +289,16 @@ impl RuleStore { healed_ids.len() ); } - self.skipped.store(skipped_ids.len(), Ordering::Relaxed); + self.skipped + .store(skipped_ids.len() + quarantined, Ordering::Relaxed); Ok(RuleSet { rules }) } /// Number of rule rows the most recent [`snapshot`](Self::snapshot) call - /// skipped because their JSON failed to deserialize. Rows are never - /// deleted for failing to parse; this count lets callers surface the - /// problem (e.g. in `status`) instead of losing data silently. + /// did not load: their JSON failed to deserialize, or they were + /// quarantined. Such rows are never deleted; this count lets callers + /// surface the problem (e.g. in `status`) instead of losing data + /// silently. pub fn skipped_rules(&self) -> usize { self.skipped.load(Ordering::Relaxed) } @@ -816,6 +821,7 @@ mod tests { "a parent_exe rule matches every process and must not load: {names:?}" ); assert!(names.contains(&"curl".to_string()), "{names:?}"); + assert_eq!(reopened.skipped_rules(), 1, "status must report it"); // Quarantined, never deleted: the row is the operator's. let rows: i64 = reopened @@ -987,12 +993,14 @@ mod tests { } #[test] - fn disabled_rules_excluded_from_snapshot() { - let store = RuleStore::open_in_memory().unwrap(); + fn disabled_rules_survive_a_restart() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("rules.db"); let mut rule = sample_rule("curl"); rule.enabled = false; - store.upsert(&rule).unwrap(); - assert!(store.snapshot().unwrap().rules.is_empty()); + RuleStore::open(&path).unwrap().upsert(&rule).unwrap(); + let reopened = RuleStore::open(&path).unwrap().snapshot().unwrap(); + assert_eq!(reopened.rules, vec![rule]); } #[test] diff --git a/crates/cfc-proto/proto/cfc.proto b/crates/cfc-proto/proto/cfc.proto index f69ad7e..3d5cf92 100644 --- a/crates/cfc-proto/proto/cfc.proto +++ b/crates/cfc-proto/proto/cfc.proto @@ -227,8 +227,10 @@ message StatusResponse { Action no_ui_action = 11; uint32 prompt_timeout_secs = 12; - // Rule rows the last storage load could not deserialize. Non-zero means - // rules exist on disk that are NOT being enforced. + // Rule rows the last storage load did not load: they could not be + // deserialized, or they fail the API's own rule checks and were + // quarantined. Non-zero means rules exist on disk that are NOT being + // enforced. uint64 skipped_rules = 13; // Best-effort "packets are actually reaching us" signal. False means the diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index d690cca..0818f21 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -528,7 +528,21 @@ journalctl -u colony-firewalld -g 'failed to deserialize' ``` If you need the rule back now and cannot upgrade, delete the offending -row by id and re-create it with `cfc rules add`. +row with `cfc rules remove `, giving the full id from the journal, and +re-create it with `cfc rules add`. + +The same warning counts **quarantined** rows: rules an older version +accepted that the daemon now refuses (for example one scoped only on a +parent executable, which would match every process). They are not +applied, not listed, and preserved on disk. The journal names each one +and why: + +```sh +journalctl -u colony-firewalld -g 'fails the API boundary' +``` + +Remove it with `cfc rules remove ` (the full id) and re-create it in +a form the daemon accepts. `cfc rules import --replace` also deletes such rows. ## Where things live From 0a0cca22f3f921312ea899c859988e3ff9fcd62d Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:33:12 +0200 Subject: [PATCH 019/125] fix(ipc): log ApplyRules with its caller like every other mutation ApplyRules, the RPC behind rules import and import --replace, discarded the authorized peer and logged nothing, so the most destructive mutation was the only one missing from the journal audit trail. It now logs the peer uid and pid, replace, the applied and removed counts and the ids as "rules applied". The docs list ApplyRules among the mutating RPCs and add the message to the audit grep. --- CHANGELOG.md | 4 ++++ crates/cfc-daemon/src/ipc.rs | 15 +++++++++++++-- docs/ARCHITECTURE.md | 2 +- docs/HARDENING.md | 4 ++-- systemd/daemon.toml.sample | 2 +- 5 files changed, 21 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8402ecf..a69f6d1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -42,6 +42,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - Rules quarantined at load were reported only in the journal. `cfc status` and the GUI now count them with the rows that could not be loaded, and `cfc rules remove ` with the full id deletes such an unlisted row. +- `ApplyRules`, behind `cfc rules import` and `import --replace`, was the + only mutating RPC that left no journal line. It now logs the caller's uid + and pid, whether it replaced the rule set, and the applied and removed + counts as "rules applied". ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index ac62cc7..1774001 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -727,7 +727,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - self.authorize(&req, Access::Mutate)?; + let peer = self.authorize(&req, Access::Mutate)?; let req = req.into_inner(); if req.replace && req.rules.is_empty() { return Err(Status::invalid_argument("refusing an empty replacement")); @@ -762,7 +762,18 @@ impl Firewall for FirewallService { .store .apply_rules(&pending, req.replace) .map_err(|e| Status::internal(format!("storage: {e}")))?; - let assigned = pending.iter().map(|rule| rule.id.to_string()).collect(); + let assigned: Vec = pending.iter().map(|rule| rule.id.to_string()).collect(); + info!( + rpc = "ApplyRules", + peer_uid = peer.uid, + peer_pid = ?peer.pid, + replace = req.replace, + applied = assigned.len(), + removed, + rule_ids = ?assigned, + outcome = "ok", + "rules applied" + ); let mut final_rules = if req.replace { Vec::new() } else { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index fe7a32d..29dc4aa 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -314,7 +314,7 @@ the entire attack surface. Two layers: exist the daemon does not refuse to start: it warns with the exact fix and leaves the socket 0600, root-only. 2. **Peer credentials.** Every connection carries `SO_PEERCRED`. Mutating - RPCs (`UpsertRule`, `DeleteRule`, `SetPaused`, `SubmitVerdict`) require + RPCs (`UpsertRule`, `ApplyRules`, `DeleteRule`, `SetPaused`, `SubmitVerdict`) require uid 0 or a socket that is genuinely group-gated. Read-only RPCs (`ListRules`, `GetStatus`, `ListEvents`, `StreamConnections`, `StreamPrompts`) are open to any peer that got past layer 1. diff --git a/docs/HARDENING.md b/docs/HARDENING.md index cb1bb2b..9121b78 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -239,7 +239,7 @@ and the daemon checks the caller per RPC: | RPC class | RPCs | Requires | |-----------|-----------------------------|-----------------------------| -| Mutating | `UpsertRule`, `DeleteRule`, `SetPaused`, `SubmitVerdict` | uid 0, **or** a socket that is genuinely group-gated | +| Mutating | `UpsertRule`, `ApplyRules`, `DeleteRule`, `SetPaused`, `SubmitVerdict` | uid 0, **or** a socket that is genuinely group-gated | | Read-only | `ListRules`, `GetStatus`, `ListEvents`, `StreamConnections`, `StreamPrompts` | Only layer 1 | `require_group = false` in `[ipc]` turns the mutating check off. Leave it @@ -315,7 +315,7 @@ Three places record what the firewall did: logged with the calling uid and pid, the target, and the outcome: ```sh -journalctl -u colony-firewalld -g 'rule upserted|rule delete|verdict submitted|paused' +journalctl -u colony-firewalld -g 'rule upserted|rules applied|rule delete|verdict submitted|paused' ``` so "who deleted the rule blocking that telemetry endpoint" is answerable diff --git a/systemd/daemon.toml.sample b/systemd/daemon.toml.sample index 4801e2d..1a789b2 100644 --- a/systemd/daemon.toml.sample +++ b/systemd/daemon.toml.sample @@ -255,7 +255,7 @@ enabled = true # authentication, no per-user identity, no password. # 2. Peer credentials. Every connection carries SO_PEERCRED, and the # daemon logs the calling uid/pid for every mutating RPC. Mutating -# RPCs (UpsertRule, DeleteRule, SetPaused, SubmitVerdict) need uid 0 +# RPCs (UpsertRule, ApplyRules, DeleteRule, SetPaused, SubmitVerdict) need uid 0 # or a socket that is genuinely group-gated. Read-only RPCs # (ListRules, GetStatus, ListEvents, StreamConnections, StreamPrompts) # are open to any peer that got through layer 1. From ff4dcfbf2934d8f8d762d9397a306f8c9d33351b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:34:07 +0200 Subject: [PATCH 020/125] fix(nfqueue): let a rule Allow beat a parked prompt's timeout fallback Answering one prompt with "Allow always" adds the rule but leaves other prompts for the same program parked. When those timed out, resolve_prompt let only a current refusal override the fallback, so their packets were denied, audited and logged as blocked although a rule now allowed them. A rule-sourced Allow now overrides a timeout or no-UI fallback; an explicit user answer still stands. --- CHANGELOG.md | 4 +++ crates/cfc-daemon/src/nfqueue.rs | 44 +++++++++++++++++++++++++++++++- docs/ARCHITECTURE.md | 4 ++- 3 files changed, 50 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a69f6d1..835ea9a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -46,6 +46,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). only mutating RPC that left no journal line. It now logs the caller's uid and pid, whether it replaced the rule set, and the applied and removed counts as "rules applied". +- Packets parked on a prompt that timed out were refused even when an + "Allow always" given meanwhile for the same program now allowed them. A + rule's Allow now takes precedence over the timeout or no-UI fallback; an + explicit user answer still stands. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index 4fb5769..838941f 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -70,7 +70,7 @@ use crate::process_resolve; use crate::reject::Rejecter; use crate::stats::Stats; use anyhow::Context as _; -use cfc_core::{Action, Connection, Direction, Process, Protocol, Verdict}; +use cfc_core::{Action, Connection, Direction, Process, Protocol, Verdict, VerdictSource}; use nfq::{Message, Queue, Verdict as NfqVerdict}; use std::collections::HashMap; use std::net::IpAddr; @@ -716,7 +716,21 @@ impl Worker { // the individual segment (its sequence numbers, its source // port), and parallel connections share one prompt. let verdict = match self.engine.evaluate(&packet.connection, &packet.process) { + // A refusal decided since the prompt opened always wins. Decision::Resolved(current) if current.action != Action::Allow => current, + // So does a rule's Allow over a fallback nobody chose: a rule + // added while this prompt waited (another prompt's "Allow + // always" for the same program) must not lose to the timeout + // Deny. A user's own answer stands. + Decision::Resolved(current) + if matches!(current.source, VerdictSource::Rule(_)) + && matches!( + pv.verdict.source, + VerdictSource::DefaultPolicy | VerdictSource::Timeout + ) => + { + current + } _ => pv.verdict, }; self.deliver( @@ -2566,6 +2580,34 @@ mod tests { assert_eq!(h.stats.connections_denied(), 2); } + #[test] + fn a_rule_allow_added_while_parked_beats_the_fallback_but_not_the_user() { + for (answer, expected) in [ + (Verdict::from_policy(Action::Deny), NfqVerdict::Accept), + ( + Verdict { + action: Action::Deny, + source: VerdictSource::UserPrompt, + }, + NfqVerdict::Drop, + ), + ] { + let mut h = LoopHarness::new(vec![], vec![], dp_deny()); + h.worker() + .handle_message(FakeMsg::new(1, tcp_packet(443))) + .unwrap(); + let prompt = h.prompt_rx.try_recv().unwrap(); + h.worker().engine.upsert_rule(allow_port_rule(443)); + h.worker() + .resolve_prompt(PromptVerdict { + prompt_id: prompt.prompt_id, + verdict: answer, + }) + .unwrap(); + assert_eq!(h.verdicts(), vec![(1, expected)], "{answer:?}"); + } + } + #[test] fn loopback_unmatched_flows_keep_the_nonprompting_local_default() { // The output interface, including local host addresses, defines this diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 29dc4aa..e6e633a 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -95,7 +95,9 @@ The worker keeps two maps that are created and destroyed together: - `waiters: HashMap` holds the fallback and each parked packet's connection and process snapshot. A prompt answer is - checked against current policy for every packet; a new refusal takes precedence. + checked against current policy for every packet; a new refusal takes + precedence, and a rule's Allow takes precedence over a timeout or no-UI + fallback, though not over a user's own answer. - `pending_flows: HashMap` - the deduplication index. Verdicts arrive asynchronously on a separate channel and are applied out of From 9600f51a952d299b2723e4e3f6015e9530dd6d09 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:41:34 +0200 Subject: [PATCH 021/125] fix(daemon): cache settled image digests instead of rehashing per flow Only root-sealed images reused their digest, so every new flow from any other executable (home directories, programs still running after an upgrade) reread and hashed up to 64 MiB on the single packet thread. One program in a connect loop stalled new flows for the whole machine. The digest cache now covers every image, keyed by dev, inode, size, mtime and ctime with nanoseconds, and is filled only when ctime is at least two seconds old as hashing starts. Userspace cannot set ctime and any change moves it, so a modified image misses the cache and an unchanged one is never rehashed. --- CHANGELOG.md | 7 ++ TODO.md | 5 +- crates/cfc-daemon/src/process_resolve.rs | 97 +++++++++++++++++++----- docs/ARCHITECTURE.md | 14 ++-- 4 files changed, 97 insertions(+), 26 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 835ea9a..9229fe9 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -50,6 +50,13 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). "Allow always" given meanwhile for the same program now allowed them. A rule's Allow now takes precedence over the timeout or no-UI fallback; an explicit user answer still stands. +- Every new flow from an executable that was not root-sealed (anything under + a home directory, and any program still running after its package was + upgraded) reread and rehashed up to 64 MiB on the single packet thread, so + one such program opening connections in a loop stalled new flows for the + whole machine. Digests are cached again by device, inode, size, mtime and + ctime, and only once ctime is two seconds old, so a changed file is always + rehashed and an unchanged one never is. ## [0.7.0] - 2026-09-30 diff --git a/TODO.md b/TODO.md index 9b87283..1f80040 100644 --- a/TODO.md +++ b/TODO.md @@ -117,8 +117,9 @@ are worth stating rather than discovering: Process resolution now rereads policy identity for every packet lookup; pid and start time do not identify an executable across exec. Its path and digest come from one opened mapped image, with metadata and link consistency checks. -Mutable images bypass the digest cache. A raw exec-event filename is retained -for diagnostics only; once `/proc` is gone, the policy executable is unknown. +Digests are cached by full image key, ctime included, once ctime has settled. +A raw exec-event filename is retained for diagnostics only; once `/proc` is +gone, the policy executable is unknown. Shared or passed socket descriptors remain outside sender attribution, and the mapped image is still a read-time snapshot rather than packet-time proof. diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index cba579d..dd3a468 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -54,7 +54,7 @@ use std::net::{IpAddr, Ipv4Addr, Ipv6Addr}; use std::os::unix::fs::MetadataExt; use std::path::{Path, PathBuf}; use std::sync::LazyLock; -use std::time::{Duration, Instant}; +use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; use tracing::trace; /// Per-lookup budget for the /proc slow path. @@ -65,10 +65,30 @@ const RESOLVE_BUDGET: Duration = Duration::from_millis(50); /// produces (SYN, first payload, retransmits). const INODE_CACHE_TTL: Duration = Duration::from_secs(2); -/// Only root-sealed images reuse digests. Mutable images are read afresh: -/// filesystem timestamps are change hints, not a content identity. +/// Digests live until their image key changes or they are evicted. +/// +/// The key is the image's dev, inode, size, mtime and ctime with nanoseconds. +/// mtime alone would be a change hint, not a content identity: its owner can +/// set it back with `utimes`. ctime cannot be set from userspace; any write, +/// truncate or metadata change moves it to the current time. So once a file's +/// ctime is older than [`DIGEST_SETTLE`] when hashing starts, a later change +/// gets a different key. Only such settled images are cached, which closes the +/// window where a write lands in the same timestamp tick as the hash. +/// +/// Without the cache every new flow from an image that was not root-sealed +/// reread and rehashed up to 64 MiB on the single packet thread, so one +/// program in a connect loop stalled every new flow on the machine. +/// +/// Known limit: a store through a shared writable mapping moves ctime only +/// when the page is first dirtied, so bytes changed that way before writeback +/// can keep their key. That takes write access to the file, which a +/// root-sealed image does not give anyone but root. const SHA_CACHE_TTL: Duration = Duration::from_secs(u64::MAX / 2); +/// How old an image's ctime must be before its digest is cached. Far above +/// any filesystem's timestamp granularity. +const DIGEST_SETTLE: Duration = Duration::from_secs(2); + /// Don't hash executables larger than this. Shared with the CLI's /// `--pin-hash` (`cfc_core::rule`), which must refuse to create what this /// side would refuse to compute. @@ -791,12 +811,10 @@ impl MappedImage { if image_key(&meta) != self.key { return None; } - let cacheable = cfc_core::exe_path::file_is_sealed(meta.uid(), meta.mode()) - && cfc_core::exe_path::is_root_sealed(&self.path).unwrap_or(false); if meta.len() > SHA256_MAX_LEN { trace!(len = meta.len(), "exe too large to hash; skipping"); } - let sha256 = sha256_open_file(self.file, SHA256_MAX_LEN, cacheable); + let sha256 = sha256_open_file(self.file, SHA256_MAX_LEN); if image_key(&fs::metadata(link).ok()?) != self.key || fs::read_link(link).ok()? != self.path { @@ -818,18 +836,33 @@ fn image_key(meta: &fs::Metadata) -> ImageKey { ) } -fn sha256_open_file(mut f: fs::File, max_len: u64, cacheable: bool) -> Option { +/// Whether `meta`'s ctime is at least [`DIGEST_SETTLE`] before `now`. A +/// ctime in the future (clock stepped back) is not settled. +fn settled(meta: &fs::Metadata, now: SystemTime) -> bool { + let (Ok(secs), Ok(nanos)) = ( + u64::try_from(meta.ctime()), + u32::try_from(meta.ctime_nsec()), + ) else { + return false; + }; + let ctime = UNIX_EPOCH + Duration::new(secs, nanos); + now.duration_since(ctime) + .is_ok_and(|age| age >= DIGEST_SETTLE) +} + +fn sha256_open_file(mut f: fs::File, max_len: u64) -> Option { let meta = f.metadata().ok()?; if !meta.is_file() || meta.len() > max_len { return None; } let key = image_key(&meta); let now = Instant::now(); - if cacheable { - if let Some(cached) = SHA_CACHE.lock().get(&key, now) { - return Some(cached); - } + if let Some(cached) = SHA_CACHE.lock().get(&key, now) { + return Some(cached); } + // Decided before reading: a change during or after the read must land + // at a later ctime than the one in `key`. + let cacheable = settled(&meta, SystemTime::now()); // Read in a loop rather than io::copy, and hex-encode by hand rather than // with `{:x}`. RustCrypto 0.11 drops `io::Write` on the hashers and returns // an `Array` that no longer implements `LowerHex`, so both idioms stop @@ -871,7 +904,7 @@ fn sha256_open_file(mut f: fs::File, max_len: u64, cacheable: bool) -> Option Option { - sha256_open_file(fs::File::open(path).ok()?, max_len, false) + sha256_open_file(fs::File::open(path).ok()?, max_len) } /// Bounded TTL map. `now` is injected so expiry is unit-testable without @@ -1386,22 +1419,48 @@ mod tests { } #[test] - fn mutable_images_ignore_cached_digests() { + fn an_unchanged_image_is_not_rehashed_and_a_changed_one_is() { let dir = tempfile::tempdir().unwrap(); let path = dir.path().join("image"); - fs::write(&path, b"hello world").unwrap(); - let file = fs::File::open(&path).unwrap(); - let key = image_key(&file.metadata().unwrap()); + fs::write(&path, b"first").unwrap(); + let key = image_key(&fs::metadata(&path).unwrap()); SHA_CACHE .lock() .insert(key, "cached-placeholder".into(), Instant::now()); assert_eq!( - sha256_open_file(file, 1024, false).as_deref(), - Some("b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9") + sha256_open_file(fs::File::open(&path).unwrap(), 1024).as_deref(), + Some("cached-placeholder"), + "same key, no reread" + ); + fs::write(&path, b"hello world").unwrap(); + assert_ne!(image_key(&fs::metadata(&path).unwrap()), key); + assert_eq!( + sha256_open_file(fs::File::open(&path).unwrap(), 1024).as_deref(), + Some("b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9"), + "a changed image misses the cache" ); SHA_CACHE.lock().remove(&key); } + #[test] + fn only_settled_images_are_cached() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("image"); + fs::write(&path, b"fresh image").unwrap(); + let meta = fs::metadata(&path).unwrap(); + let ctime = UNIX_EPOCH + Duration::new(meta.ctime() as u64, meta.ctime_nsec() as u32); + assert!(!settled(&meta, ctime)); + assert!(!settled(&meta, ctime - Duration::from_secs(5))); + assert!(settled(&meta, ctime + DIGEST_SETTLE)); + + // Written just now, so hashing it must not fill the cache. + assert!(sha256_open_file(fs::File::open(&path).unwrap(), 1024).is_some()); + assert_eq!( + SHA_CACHE.lock().get(&image_key(&meta), Instant::now()), + None + ); + } + #[test] fn sha256_uses_the_opened_image_after_path_replacement() { let dir = tempfile::tempdir().unwrap(); @@ -1411,7 +1470,7 @@ mod tests { fs::rename(&path, dir.path().join("previous-image")).unwrap(); fs::write(&path, b"different image").unwrap(); assert_eq!( - sha256_open_file(file, 1024, false).as_deref(), + sha256_open_file(file, 1024).as_deref(), Some("b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9") ); } diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index e6e633a..f81a53f 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -170,12 +170,12 @@ hundred microseconds before the packet's latency becomes visible. walking `/proc/*/fd` for a `socket:[inode]` link. A shared or passed socket descriptor still does not identify which holder sent a packet. -Two bounded caches avoid repeated socket walks and sealed-image hashing: +Two bounded caches avoid repeated socket walks and image hashing: | Cache | Key | Lifetime | |----------------|---------------------------------------------|----------| | inode -> pid | socket inode | 2s | -| sealed exe digest | dev, inode, length, mtime and ctime with nanoseconds | key change or eviction | +| exe digest | dev, inode, length, mtime and ctime with nanoseconds | key change or eviction | A complete process record is read on every resolution: exec changes policy identity without changing pid or start time. A cache hit on the inode cache @@ -186,9 +186,13 @@ The binary's SHA-256 is read through `/proc//exe`, so it hashes the image actually running even if the file on disk was replaced or deleted. The same opened file supplies metadata and bytes. Content changes during hashing are rejected; the mapped link, metadata and process start time must -still agree before publishing executable identity. Mutable images are never -served from the digest cache. Files over 64 MiB retain their path but have no -digest. This remains a read-time snapshot: an exec after the final check can +still agree before publishing executable identity. A digest is cached only +when the image's ctime was at least 2 seconds old as hashing began. Userspace +cannot set ctime and any write moves it, so a changed image misses the cache; +an unchanged one is never rehashed per packet on the single worker. The +exception is a store through a shared writable mapping, which moves ctime only +when the page is first dirtied; it needs write access to the file. Files over +64 MiB retain their path but have no digest. This remains a read-time snapshot: an exec after the final check can change the process before the queued packet receives its verdict. The kernel also reports the originating uid and gid with each queued packet From 2427cd50f6af272351d922355695c1976c487177 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:42:39 +0200 Subject: [PATCH 022/125] fix(daemon): treat an exe path naming another host file as unknown /proc//exe is rendered in the process's own mount namespace. A user with an unprivileged user and mount namespace could mount their own bytes at /usr/bin/curl, present that path, and match every path-only rule for the host's curl; prompt binding also took the host file's root-sealed status for it and skipped the hash. The resolver now stats the reported path in the daemon's view and leaves the executable unknown when it names a file other than the mapped image. Paths the daemon cannot see are kept as reported; HARDENING.md says what that leaves open. --- CHANGELOG.md | 7 +++++ crates/cfc-daemon/src/process_resolve.rs | 35 ++++++++++++++++++++++++ crates/cfc-daemon/src/prompts.rs | 2 ++ docs/ARCHITECTURE.md | 4 ++- docs/HARDENING.md | 10 +++++++ 5 files changed, 57 insertions(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9229fe9..898596c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -57,6 +57,13 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). whole machine. Digests are cached again by device, inode, size, mtime and ctime, and only once ctime is two seconds old, so a changed file is always rehashed and an unchanged one never is. +- A process in its own mount namespace (`unshare -rm`, a container) could + mount its own bytes at a host path such as `/usr/bin/curl` and match every + path-only rule for the host's program, including prompt-created Allows that + skip hash binding for root-sealed paths. An executable path that names a + different file in the daemon's view is now reported as unknown. Container + and Flatpak runtime binaries at such paths therefore lose their executable + identity instead of borrowing the host's. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index dd3a468..2132ac3 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -811,6 +811,10 @@ impl MappedImage { if image_key(&meta) != self.key { return None; } + if !path_names_image(&self.path, &self.key) { + trace!(path = %self.path.display(), "exe path names another file here"); + return None; + } if meta.len() > SHA256_MAX_LEN { trace!(len = meta.len(), "exe too large to hash; skipping"); } @@ -824,6 +828,21 @@ impl MappedImage { } } +/// Whether `path`, read in the daemon's own mount namespace, can stand for +/// the mapped image `key` describes. +/// +/// The kernel renders `/proc//exe` relative to the process's own mount +/// namespace. Any user who can create one (`unshare -rm`, a container) can +/// mount their own bytes at `/usr/bin/curl` and present that path, which then +/// matched every path-only rule for the host's curl and passed as root-sealed +/// in prompt binding. So a path that names a different file here is not this +/// image's identity. A path the daemon cannot stat at all (a Flatpak `/app` +/// path, a home directory behind `ProtectHome`, a replaced image's +/// " (deleted)" name) names nothing here and is left as is. +fn path_names_image(path: &Path, key: &ImageKey) -> bool { + fs::metadata(path).map_or(true, |m| (m.dev(), m.ino()) == (key.0, key.1)) +} + fn image_key(meta: &fs::Metadata) -> ImageKey { ( meta.dev(), @@ -1499,6 +1518,22 @@ mod tests { assert_eq!(image.finish(&link), None, "mixed image identity is unknown"); } + #[test] + fn a_path_naming_another_file_here_is_not_the_image() { + let dir = tempfile::tempdir().unwrap(); + let host = dir.path().join("curl"); + let mounted = dir.path().join("mounted-over-curl"); + fs::write(&host, b"host image").unwrap(); + fs::write(&mounted, b"namespace image").unwrap(); + let mounted_key = image_key(&fs::metadata(&mounted).unwrap()); + assert!(!path_names_image(&host, &mounted_key)); + assert!(path_names_image(&mounted, &mounted_key)); + assert!( + path_names_image(&dir.path().join("absent"), &mounted_key), + "a path invisible here cannot be judged" + ); + } + #[test] fn digest_keys_track_size_when_modification_time_is_restored() { let dir = tempfile::tempdir().unwrap(); diff --git a/crates/cfc-daemon/src/prompts.rs b/crates/cfc-daemon/src/prompts.rs index 476886e..b217886 100644 --- a/crates/cfc-daemon/src/prompts.rs +++ b/crates/cfc-daemon/src/prompts.rs @@ -342,6 +342,8 @@ fn compute_binding(process: &cfc_core::Process) -> PromptBinding { // certainly not root-sealed. Never re-read a PID here: it may have exec'd // or been recycled since the worker recorded this process. // The sealed check rejects symlinks instead of re-resolving this snapshot. + // The resolver publishes a path only if it names the mapped image here or + // names nothing here, so a sealed path is the host file that was running. if cfc_core::exe_path::is_root_sealed(&exe).unwrap_or(false) { return PromptBinding { exe: Some(exe), diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index f81a53f..6f6c09a 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -186,7 +186,9 @@ The binary's SHA-256 is read through `/proc//exe`, so it hashes the image actually running even if the file on disk was replaced or deleted. The same opened file supplies metadata and bytes. Content changes during hashing are rejected; the mapped link, metadata and process start time must -still agree before publishing executable identity. A digest is cached only +still agree before publishing executable identity. The link is rendered in +the process's own mount namespace, so a path that names a different file in +the daemon's view leaves the executable unknown. A digest is cached only when the image's ctime was at least 2 seconds old as hashing began. Userspace cannot set ctime and any write moves it, so a changed image misses the cache; an unchanged one is never rehashed per packet on the single worker. The diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 9121b78..8826e6b 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -201,6 +201,16 @@ is a separate launch mode. reauthorized for each sending executable. Current descriptor ownership and validated eBPF hints reduce false attribution; neither proves which process sent a packet. +- **Mount namespaces and same-user code**: a process reports its executable + path as its own mount namespace sees it. When that path names a different + file in the daemon's view (a container's or `unshare -rm` user's + `/usr/bin/curl`), the executable is reported as unknown, so path rules for + the host's file do not match it. A path the daemon cannot see at all (a + Flatpak `/app` path, anything under `/home`, hidden by `ProtectHome`) is + taken as reported: a hand-written path-only rule for such a path can be + matched from a mount namespace, so pin its hash. Code already running as a + user can also borrow an allowed program's identity by running it with + chosen arguments or with `LD_PRELOAD`, which a hash pin does not prevent. - **Raw and packet sockets**: applications with `CAP_NET_RAW` can use AF_PACKET outside the shipped `inet OUTPUT` hook. Raw IP packets can coincide with another socket's tuple even when TCP matching is strict. Tuple and inode From e718bf04118727ff777128adeb8a2024151f59b3 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:43:56 +0200 Subject: [PATCH 023/125] fix(nfqueue): let sealed images over 64 MiB share one prompt An image over the digest cap has no sha256, so FlowKey gave each of its packets a fresh random origin. Chromium, Electron apps and VS Code opened a prompt per queued packet, retransmits and parallel connections included, and MAX_PACKETS_PER_PROMPT never applied to them. A digest-less image on a root-sealed path now shares by uid and path, the identity a path-only Allow for it already uses. User-writable paths without a digest still never share. --- CHANGELOG.md | 4 +++ crates/cfc-daemon/src/nfqueue.rs | 57 ++++++++++++++++++++++++++++++-- docs/ARCHITECTURE.md | 6 ++-- 3 files changed, 63 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 898596c..142b9f1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -64,6 +64,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). different file in the daemon's view is now reported as unknown. Container and Flatpak runtime binaries at such paths therefore lose their executable identity instead of borrowing the host's. +- Executables over 64 MiB, such as Chromium, Electron apps and VS Code, have + no digest, so every queued packet from them, each retransmit and parallel + connection, opened its own prompt. On a root-sealed path they now share one + prompt per destination like any other program. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index 838941f..1c42351 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -473,7 +473,11 @@ struct FlowKey { /// Only a known image may share a prompt; uncertain identities never do. #[derive(Debug, Clone, PartialEq, Eq, Hash)] enum FlowOrigin { - Exe { path: PathBuf, sha256: String }, + /// `sha256` is `None` only for a root-sealed path. + Exe { + path: PathBuf, + sha256: Option, + }, Unattributed(uuid::Uuid), } @@ -482,8 +486,22 @@ impl FlowKey { let origin = match (&proc.sha256, proc.exe_is_known(), proc.uid) { (Some(sha256), true, Some(_)) => FlowOrigin::Exe { path: proc.exe.clone(), - sha256: sha256.clone(), + sha256: Some(sha256.clone()), }, + // Images over 64 MiB (Chromium, Electron, VS Code) have no + // digest. A root-sealed path still names one image only root can + // change, and the resolver publishes it only when it names the + // mapped file here, which is why a standing Allow for it is + // path-only. Without this every packet, retransmits included, + // opened its own prompt. + (None, true, Some(_)) + if cfc_core::exe_path::is_root_sealed(&proc.exe).unwrap_or(false) => + { + FlowOrigin::Exe { + path: proc.exe.clone(), + sha256: None, + } + } // Neither a PID nor a path identifies an unknown execution. // Such packets must receive independent authorization. _ => FlowOrigin::Unattributed(uuid::Uuid::new_v4()), @@ -2219,6 +2237,41 @@ mod tests { ); } + #[test] + fn digest_less_images_share_a_prompt_only_on_a_sealed_path() { + let sealed = "/usr/bin/env"; + if !cfc_core::exe_path::is_root_sealed(std::path::Path::new(sealed)).unwrap_or(false) { + return; + } + let conn = conn_to(443, 1111); + let digest_less = |pid, exe: &str| Process { + sha256: None, + ..test_process(pid, exe) + }; + // Over 64 MiB, so no digest: retransmits and siblings still share. + assert_eq!( + FlowKey::for_flow(&conn, &digest_less(1, sealed)), + FlowKey::for_flow(&conn, &digest_less(2, sealed)) + ); + let other_user = Process { + uid: Some(1001), + ..digest_less(2, sealed) + }; + assert_ne!( + FlowKey::for_flow(&conn, &digest_less(1, sealed)), + FlowKey::for_flow(&conn, &other_user) + ); + // A path its user can rewrite is no identity without the digest. + let dir = tempfile::tempdir().unwrap(); + let tool = dir.path().join("tool"); + std::fs::write(&tool, b"user image").unwrap(); + let tool = tool.to_str().unwrap(); + assert_ne!( + FlowKey::for_flow(&conn, &digest_less(1, tool)), + FlowKey::for_flow(&conn, &digest_less(1, tool)) + ); + } + // ---- Worker loop, driven through the PacketQueue seam ---- // // These exercise the parts the pure pipeline tests above cannot reach: diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 6f6c09a..00cc99c 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -122,8 +122,10 @@ Fast Allow was removed, so allowed flows also pay the queue round trip. **Prompt deduplication** requires the same UID, executable path, image digest, destination IP, destination port and protocol. Source address and port are -excluded, so equivalent parallel connections may share a prompt. An incomplete -identity never shares authorization. Persistent prompt Allows use the queued +excluded, so equivalent parallel connections may share a prompt. An image over +64 MiB has no digest; it shares by its path when that path is root-sealed, the +same identity a path-only Allow for it uses. Any other incomplete identity +never shares authorization. Persistent prompt Allows use the queued image digest; they never rehash a later image at a reused PID. A retargeted pathname cannot suppress the required hash binding. From 63c643744601ea2d0acccca6bc31734f4ef1a32e Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:45:58 +0200 Subject: [PATCH 024/125] fix(provenance): never block the packet thread on an index rebuild warm() holds the index write lock for the whole rebuild, and the packet thread took a blocking read lock before checking whether it may build, so the first flow from any newly seen binary waited for the build: about 120 ms with pacman, up to the 10 s rpm query timeout during a dnf transaction. The packet thread now uses try_read and answers NotReady when the index is busy or stale. NotReady is a distinct lookup result that is never cached and reads as Provenance::Unknown; before, the declined lookup was cached for an hour as "not from a package". --- CHANGELOG.md | 6 + crates/cfc-daemon/src/main.rs | 6 +- crates/cfc-daemon/src/nfqueue.rs | 9 +- crates/cfc-daemon/src/provenance.rs | 212 ++++++++++++++++++++-------- 4 files changed, 170 insertions(+), 63 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 142b9f1..a3ebd32 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -68,6 +68,12 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). no digest, so every queued packet from them, each retransmit and parallel connection, opened its own prompt. On a root-sealed path they now share one prompt per destination like any other program. +- The package-index warmer held the index's write lock for a whole rebuild, + so the packet thread blocked behind it on the first flow from any newly + seen binary: about 120 ms with pacman, up to 10 s with rpm during a `dnf` + transaction. The packet thread now skips a busy or stale index, and that + "not ready" answer is no longer cached for an hour as "not from a package"; + it shows as unknown until the index is ready. ## [0.7.0] - 2026-09-30 diff --git a/crates/cfc-daemon/src/main.rs b/crates/cfc-daemon/src/main.rs index 2a574af..683aa6d 100644 --- a/crates/cfc-daemon/src/main.rs +++ b/crates/cfc-daemon/src/main.rs @@ -264,9 +264,9 @@ async fn run() -> anyhow::Result<()> { // is immediate so a fresh daemon has provenance within a second; after // that it is a poll, because the trigger is the package database's mtime // changing under us and there is no cheap way to be told about that. Two - // minutes is chosen against what it costs to be wrong: a package installed - // just now shows as unpackaged for at most that long, in a field that - // decorates an event and decides nothing. + // minutes is chosen against what it costs to be wrong: a binary first seen + // after a package transaction shows provenance unknown for at most that + // long, in a field that decorates an event and decides nothing. tokio::spawn(async { let mut tick = tokio::time::interval(PROVENANCE_WARM_INTERVAL); loop { diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index 1c42351..20d45bf 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -609,10 +609,11 @@ impl Worker { // This thread is the datapath, and it is the only one. Saying so once // here covers everything reached from it, however deep: in particular - // a provenance lookup that finds the package index stale now answers - // "no package" rather than rebuilding it, which reads every installed - // package's file list. Measured cold on the owner's machine, that - // rebuild took 123 ms - and since this loop is a single thread, that + // a provenance lookup that finds the package index stale or being + // rebuilt answers "not ready" rather than rebuilding it or waiting + // for the rebuild, which reads every installed package's file list. + // Measured cold on the owner's machine, that rebuild took 123 ms - + // and since this loop is a single thread, that // is not one slow packet, it is every flow on the machine stopping // together. `provenance::warm` does the build off this thread. crate::provenance::mark_datapath_thread(); diff --git a/crates/cfc-daemon/src/provenance.rs b/crates/cfc-daemon/src/provenance.rs index 8d86a65..1cc416a 100644 --- a/crates/cfc-daemon/src/provenance.rs +++ b/crates/cfc-daemon/src/provenance.rs @@ -81,14 +81,19 @@ pub struct PackageFile { pub sha256: Option, } +/// The packet thread declined to build, or to wait for, the package index. +/// Not an answer: it is never cached and reads as [`Provenance::Unknown`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct NotReady; + /// A distribution package database, viewed as a path -> record lookup. pub trait PackageDb: Send + Sync { /// Human-readable backend name, for logs. fn name(&self) -> &'static str; - /// The package record for an absolute path, or `None` when no installed + /// The package record for an absolute path, `Ok(None)` when no installed /// package owns it. - fn lookup(&self, exe: &Path) -> Option; + fn lookup(&self, exe: &Path) -> Result, NotReady>; } // --------------------------------------------------------------------------- @@ -185,7 +190,9 @@ pub fn describe(exe: &Path, running_sha256: Option<&str>) -> (Option, Pr let Ok(meta) = std::fs::metadata(exe) else { return (None, Provenance::Unknown); }; - let record = cached_lookup(db.as_ref(), exe, &meta); + let Ok(record) = cached_lookup(db.as_ref(), exe, &meta) else { + return (None, Provenance::Unknown); + }; let provenance = decide(record.as_ref(), running_sha256); (record.map(|r| r.package), provenance) } @@ -222,15 +229,23 @@ fn decide(record: Option<&PackageFile>, running_sha256: Option<&str>) -> Provena /// The key is already content-addressed for our purposes: replacing the /// file changes the inode or the mtime, so a swapped binary can never be /// answered from a stale entry. -fn cached_lookup(db: &dyn PackageDb, exe: &Path, meta: &std::fs::Metadata) -> Option { +/// +/// [`NotReady`] is not memoized. Cached as "no package", it showed every +/// binary first seen at boot or after an upgrade as unpackaged for the +/// cache's full hour. +fn cached_lookup( + db: &dyn PackageDb, + exe: &Path, + meta: &std::fs::Metadata, +) -> Result, NotReady> { let key = (meta.dev(), meta.ino(), meta.mtime(), meta.mtime_nsec()); let now = Instant::now(); if let Some(hit) = LOOKUP_CACHE.lock().get(&key, now) { - return hit; + return Ok(hit); } - let record = db.lookup(exe); + let record = db.lookup(exe)?; LOOKUP_CACHE.lock().insert(key, record.clone(), now); - record + Ok(record) } // --------------------------------------------------------------------------- @@ -378,8 +393,8 @@ thread_local! { /// Declares the calling thread to be the packet worker. /// /// Call once, from the worker itself. After this, a provenance lookup on this -/// thread that finds the package index stale answers "no package" instead of -/// building it, and leaves a note for [`warm`]. +/// thread that finds the package index stale or locked for a rebuild answers +/// [`NotReady`] instead of building or waiting, and leaves a note for [`warm`]. pub fn mark_datapath_thread() { ON_DATAPATH.with(|c| c.set(true)); } @@ -476,8 +491,8 @@ impl IndexCache { &self, exe: &Path, build: impl FnOnce(Option) -> Index, - ) -> Option { - self.record_of(exe, build, may_build()).map(|(pkg, _)| pkg) + ) -> Result, NotReady> { + Ok(self.record_of(exe, build, may_build())?.map(|(pkg, _)| pkg)) } /// The package owning `exe` *and* the digest the index carries for it, for @@ -488,15 +503,23 @@ impl IndexCache { exe: &Path, build: impl FnOnce(Option) -> Index, may_build: bool, - ) -> Option<(String, Option)> { + ) -> Result)>, NotReady> { let stamp = self.stamp(); let key = path_hash(exe); - if let Some(idx) = self.index.read().as_ref() { + // Nor does the datapath wait for a build: a builder holds the write + // lock through all of it, up to RPM_QUERY_TIMEOUT for rpm. + let current = if may_build { + Some(self.index.read()) + } else { + self.index.try_read() + }; + if let Some(idx) = current.as_ref().and_then(|guard| guard.as_ref()) { if idx.stamp == stamp { - return idx.get(key).map(|p| (p, idx.digest(key))); + return Ok(idx.get(key).map(|p| (p, idx.digest(key)))); } } + drop(current); // The datapath does not build. Building means reading every installed // package's file list - 123 ms and 66 MB through the allocator on the @@ -505,14 +528,13 @@ impl IndexCache { // a tenth of a second. Measured cold: the second connection after a // restart took 21 ms while the rest took 0.6 ms. // - // So a caller that cannot afford to build says so, gets `None`, and - // leaves a note. `None` here means "no package known", which is - // already the honest answer for anything unpackaged, and provenance is - // not a rule predicate - it decorates prompts and events. Nothing is - // enforced differently while the index is a few seconds late. + // So a caller that cannot afford to build says so, gets `NotReady`, + // and leaves a note. Provenance is not a rule predicate - it decorates + // prompts and events - so nothing is enforced differently while the + // index is a few seconds late; the record shows `Unknown` meanwhile. if !may_build { self.wanted.store(true, Ordering::Relaxed); - return None; + return Err(NotReady); } let mut guard = self.index.write(); @@ -541,8 +563,9 @@ impl IndexCache { release_index_scratch(); self.wanted.store(false, Ordering::Relaxed); } - let idx = guard.as_ref()?; - idx.get(key).map(|p| (p, idx.digest(key))) + Ok(guard + .as_ref() + .and_then(|idx| idx.get(key).map(|p| (p, idx.digest(key))))) } } @@ -594,8 +617,10 @@ impl PackageDb for Pacman { "pacman" } - fn lookup(&self, exe: &Path) -> Option { - let dir = self.cache.owner_of(exe, |s| self.build_index(s))?; + fn lookup(&self, exe: &Path) -> Result, NotReady> { + let Some(dir) = self.cache.owner_of(exe, |s| self.build_index(s))? else { + return Ok(None); + }; let (name, version) = split_pkg_dir(&dir); let mtree = self.cache.root.join(&dir).join("mtree"); let sha256 = match mtree_digest_for(&mtree, exe) { @@ -605,10 +630,10 @@ impl PackageDb for Pacman { None } }; - Some(PackageFile { + Ok(Some(PackageFile { package: format!("{name} {version}"), sha256, - }) + })) } } @@ -788,7 +813,8 @@ fn errno_of(err: &anyhow::Error) -> Option { /// Not a performance knob - a safety one. `dnf` holds the rpmdb open for the /// length of a transaction, and a query that arrives mid-upgrade waits for it. /// Without a bound, a `dnf update` on a slow disk would block whichever thread -/// is building the index, and that thread is on the packet path. Ten seconds +/// is building the index, and every other non-packet lookup waiting on its +/// lock. Ten seconds /// is far past any healthy query on the biggest installs and far short of a /// package transaction. const RPM_QUERY_TIMEOUT: Duration = Duration::from_secs(10); @@ -1031,11 +1057,11 @@ impl PackageDb for Rpm { "rpm" } - fn lookup(&self, exe: &Path) -> Option { - let (package, sha256) = self + fn lookup(&self, exe: &Path) -> Result, NotReady> { + Ok(self .cache - .record_of(exe, |s| self.build_index(s), may_build())?; - Some(PackageFile { package, sha256 }) + .record_of(exe, |s| self.build_index(s), may_build())? + .map(|(package, sha256)| PackageFile { package, sha256 })) } } @@ -1109,13 +1135,15 @@ impl PackageDb for Dpkg { "dpkg" } - fn lookup(&self, exe: &Path) -> Option { - let pkg = self.cache.owner_of(exe, |s| self.build_index(s))?; - Some(PackageFile { - package: pkg, - // Deliberately unverified; see the type docs. - sha256: None, - }) + fn lookup(&self, exe: &Path) -> Result, NotReady> { + Ok(self + .cache + .owner_of(exe, |s| self.build_index(s))? + .map(|pkg| PackageFile { + package: pkg, + // Deliberately unverified; see the type docs. + sha256: None, + })) } } @@ -1374,7 +1402,7 @@ mod tests { fake_pacman(tmp.path(), "abc123"); let db = Pacman::new(tmp.path().to_path_buf()); - let hit = db.lookup(Path::new("/usr/bin/curl")).unwrap(); + let hit = db.lookup(Path::new("/usr/bin/curl")).unwrap().unwrap(); assert_eq!(hit.package, "curl 8.21.0-1"); assert_eq!(hit.sha256.as_deref(), Some("abc123")); assert_eq!(decide(Some(&hit), Some("abc123")), Provenance::Verified); @@ -1382,13 +1410,16 @@ mod tests { // Owned but absent from mtree (a file listed in `files` only): // package known, nothing to verify. - let hit = db.lookup(Path::new("/usr/bin/curl-config")).unwrap(); + let hit = db + .lookup(Path::new("/usr/bin/curl-config")) + .unwrap() + .unwrap(); assert_eq!(hit.package, "curl 8.21.0-1"); assert_eq!(hit.sha256, None); assert_eq!(decide(Some(&hit), Some("abc123")), Provenance::Unknown); // Nobody owns it. - assert_eq!(db.lookup(Path::new("/tmp/curl")), None); + assert_eq!(db.lookup(Path::new("/tmp/curl")), Ok(None)); assert_eq!(decide(None, Some("abc123")), Provenance::Unpackaged); } @@ -1404,10 +1435,11 @@ mod tests { let db = Pacman::new(root.clone()); mark_datapath_thread(); - // The index is cold. A datapath lookup must answer "no package" + // The index is cold. A datapath lookup must answer "not ready" // rather than spending 123 ms reading every package's file list. - assert!( - db.lookup(Path::new("/usr/bin/curl")).is_none(), + assert_eq!( + db.lookup(Path::new("/usr/bin/curl")), + Err(NotReady), "the packet worker built the package index" ); assert!( @@ -1420,7 +1452,7 @@ mod tests { let other = std::thread::spawn(move || Pacman::new(root).lookup(Path::new("/usr/bin/curl"))); assert!( - other.join().unwrap().is_some(), + other.join().unwrap().unwrap().is_some(), "a non-datapath thread must still build" ); }) @@ -1428,6 +1460,59 @@ mod tests { .unwrap(); } + #[test] + fn the_datapath_thread_does_not_wait_for_a_build_in_progress() { + let tmp = tempfile::tempdir().unwrap(); + fake_pacman(tmp.path(), "abc123"); + let db = std::sync::Arc::new(Pacman::new(tmp.path().to_path_buf())); + assert!(db.lookup(Path::new("/usr/bin/curl")).unwrap().is_some()); + + // A builder (warm) holds the write lock through a whole rebuild. + let building = db.cache.index.write(); + let worker = std::sync::Arc::clone(&db); + let (tx, rx) = std::sync::mpsc::channel(); + std::thread::spawn(move || { + mark_datapath_thread(); + let _ = tx.send(worker.lookup(Path::new("/usr/bin/curl"))); + }); + let answer = rx.recv_timeout(Duration::from_secs(5)); + drop(building); + assert_eq!( + answer.expect("the packet worker waited for the build"), + Err(NotReady) + ); + } + + #[test] + fn a_not_ready_answer_is_not_cached() { + struct LateIndex(AtomicBool); + impl PackageDb for LateIndex { + fn name(&self) -> &'static str { + "late" + } + fn lookup(&self, _: &Path) -> Result, NotReady> { + if !self.0.swap(true, Ordering::Relaxed) { + return Err(NotReady); + } + Ok(Some(PackageFile { + package: "curl 8.21.0-1".into(), + sha256: None, + })) + } + } + let tmp = tempfile::tempdir().unwrap(); + let exe = tmp.path().join("curl"); + std::fs::write(&exe, b"image").unwrap(); + let meta = std::fs::metadata(&exe).unwrap(); + let db = LateIndex(AtomicBool::new(false)); + assert_eq!(cached_lookup(&db, &exe, &meta), Err(NotReady)); + assert_eq!( + cached_lookup(&db, &exe, &meta).unwrap().unwrap().package, + "curl 8.21.0-1", + "the index is ready now and must be asked again" + ); + } + /// `warm` is the only thing that rebuilds a stale index once the datapath /// refuses to. If it silently stopped working, provenance would go quiet /// and nothing else would fail. @@ -1444,7 +1529,7 @@ mod tests { let tmp = tempfile::tempdir().unwrap(); fake_pacman(tmp.path(), "abc123"); let db = Pacman::new(tmp.path().to_path_buf()); - assert!(db.lookup(Path::new("/usr/bin/wget")).is_none()); + assert!(db.lookup(Path::new("/usr/bin/wget")).unwrap().is_none()); // Install another package; the root's mtime moves. let pkg = tmp.path().join("wget-1.25.0-2"); @@ -1457,7 +1542,7 @@ mod tests { // Force a visibly different mtime even on a coarse-grained clock. filetime_bump(tmp.path()); - let hit = db.lookup(Path::new("/usr/bin/wget")).unwrap(); + let hit = db.lookup(Path::new("/usr/bin/wget")).unwrap().unwrap(); assert_eq!(hit.package, "wget 1.25.0-2"); assert_eq!(hit.sha256.as_deref(), Some("feed")); } @@ -1489,11 +1574,11 @@ mod tests { std::fs::write(tmp.path().join("curl.md5sums"), "abc usr/bin/curl\n").unwrap(); let db = Dpkg::new(tmp.path().to_path_buf()); - let hit = db.lookup(Path::new("/usr/bin/curl")).unwrap(); + let hit = db.lookup(Path::new("/usr/bin/curl")).unwrap().unwrap(); assert_eq!(hit.package, "curl"); assert_eq!(hit.sha256, None, "dpkg records MD5 only; we do not verify"); assert_eq!(decide(Some(&hit), Some("whatever")), Provenance::Unknown); - assert_eq!(db.lookup(Path::new("/tmp/curl")), None); + assert_eq!(db.lookup(Path::new("/tmp/curl")), Ok(None)); } #[test] @@ -1502,7 +1587,10 @@ mod tests { std::fs::write(tmp.path().join("libc6:amd64.list"), "/usr/bin/ldd\n").unwrap(); let db = Dpkg::new(tmp.path().to_path_buf()); assert_eq!( - db.lookup(Path::new("/usr/bin/ldd")).unwrap().package, + db.lookup(Path::new("/usr/bin/ldd")) + .unwrap() + .unwrap() + .package, "libc6:amd64" ); } @@ -1543,14 +1631,20 @@ mod tests { curl 8.0.1-1.el9\tdrwxr-xr-x\t/usr/share/doc/curl\t\n" ), ); - let rec = db.lookup(Path::new("/usr/bin/curl")).expect("owned"); + let rec = db + .lookup(Path::new("/usr/bin/curl")) + .unwrap() + .expect("owned"); assert_eq!(rec.package, "curl 8.0.1-1.el9"); assert_eq!(rec.sha256.as_deref(), Some(SHA_CURL)); // The directory in that output must not have been indexed: if it had, // /usr/share/doc/curl would answer for its whole subtree. - assert!(db.lookup(Path::new("/usr/share/doc/curl")).is_none()); - assert!(db.lookup(Path::new("/usr/bin/wget")).is_none()); + assert!(db + .lookup(Path::new("/usr/share/doc/curl")) + .unwrap() + .is_none()); + assert!(db.lookup(Path::new("/usr/bin/wget")).unwrap().is_none()); } #[test] @@ -1563,7 +1657,10 @@ mod tests { tmp.path(), &format!("ancient 1.0-1\t-rwxr-xr-x\t/usr/bin/ancient\t{MD5_OLD}\n"), ); - let rec = db.lookup(Path::new("/usr/bin/ancient")).expect("owned"); + let rec = db + .lookup(Path::new("/usr/bin/ancient")) + .unwrap() + .expect("owned"); assert_eq!(rec.package, "ancient 1.0-1"); assert_eq!( rec.sha256, None, @@ -1647,7 +1744,7 @@ mod tests { let db = tmp.path().join("rpmdb"); std::fs::create_dir_all(&db).unwrap(); let rpm = Rpm::with_program(db, tmp.path().join("no-such-rpm")); - assert!(rpm.lookup(Path::new("/usr/bin/curl")).is_none()); + assert!(rpm.lookup(Path::new("/usr/bin/curl")).unwrap().is_none()); } #[test] @@ -1751,7 +1848,10 @@ mod tests { let db = Pacman::new(tmp.path().to_path_buf()); assert_eq!( - db.lookup(Path::new("/usr/bin/curl")).unwrap().package, + db.lookup(Path::new("/usr/bin/curl")) + .unwrap() + .unwrap() + .package, "curl 8.21.0-1" ); } From acd3153373f6fae4bd799a2ff47aae3c7dd279cd Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:47:31 +0200 Subject: [PATCH 025/125] fix(daemon): check the fd walk deadline on failed entries too flatten() over the /proc//fd iterator skipped failed entries, descriptors closing during the walk, inside a single next() call, so a run of them ran past the 50 ms lookup budget on the packet thread. Errors are now skipped after the deadline check. --- crates/cfc-daemon/src/process_resolve.rs | 5 ++++- 1 file changed, 4 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 2132ac3..3f61104 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -398,10 +398,13 @@ fn pid_has_socket_inode(pid: u32, inode: u64, deadline: Instant) -> Option= deadline { return None; } + let Ok(fd) = fd else { continue }; if matches!(fd.target, FDTarget::Socket(i) if i == inode) { if read_starttime(pid) != Some(starttime) { return None; From 4399a517a162bcbfc98455aa4ace87e1b6a57034 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:47:35 +0200 Subject: [PATCH 026/125] test(daemon): pin identity across exec and no reverse DNS for refusals Two guarantees had no test. Process identity must follow an exec at the same pid and start time, so a cache keyed on them cannot come back unnoticed. A refused delivery must not enqueue a reverse-DNS lookup for the address the denied program chose, for Deny and Reject alike. --- crates/cfc-daemon/src/nfqueue.rs | 44 +++++++++++++++++------- crates/cfc-daemon/src/process_resolve.rs | 31 +++++++++++++++++ 2 files changed, 63 insertions(+), 12 deletions(-) diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index 20d45bf..a23dc49 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -2760,20 +2760,40 @@ mod tests { assert_eq!(h.stats.connections_total(), 0); } + /// Counts reverse-DNS lookups `deliver` asks for. + struct CountingDns(Arc); + impl HostCache for CountingDns { + fn is_self(&self, _: u32) -> bool { + false + } + fn cached_host(&self, _: IpAddr) -> Option<(String, bool)> { + None + } + fn enqueue(&self, _: IpAddr) { + self.0.fetch_add(1, Ordering::Relaxed); + } + } + #[test] - fn allowed_delivery_enriches_and_counts_once_without_a_refusal_audit() { - struct CountingDns(Arc); - impl HostCache for CountingDns { - fn is_self(&self, _: u32) -> bool { - false - } - fn cached_host(&self, _: IpAddr) -> Option<(String, bool)> { - None - } - fn enqueue(&self, _: IpAddr) { - self.0.fetch_add(1, Ordering::Relaxed); - } + fn refused_deliveries_ask_for_no_reverse_dns() { + // A refused process must not make the daemon resolve an address it + // chose: that is a DNS side channel out of a denied program. + let mut reject = deny_port_rule(443); + reject.action = Action::Reject; + for rule in [deny_port_rule(443), reject] { + let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); + let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()); + h.worker().dns = Box::new(CountingDns(calls.clone())); + h.worker() + .handle_message(FakeMsg::new(1, tcp_packet(443))) + .unwrap(); + assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); + assert_eq!(calls.load(Ordering::Relaxed), 0); } + } + + #[test] + fn allowed_delivery_enriches_and_counts_once_without_a_refusal_audit() { let calls = Arc::new(std::sync::atomic::AtomicUsize::new(0)); let mut h = LoopHarness::new(vec![], vec![allow_port_rule(443)], dp_deny()); h.worker().dns = Box::new(CountingDns(calls.clone())); diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 3f61104..29345b4 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -1521,6 +1521,37 @@ mod tests { assert_eq!(image.finish(&link), None, "mixed image identity is unknown"); } + #[test] + fn identity_follows_an_exec_at_the_same_pid() { + // Pid and start time survive exec, so nothing keyed on them may + // stand in for the image: each resolve must see the current one. + use std::io::Write as _; + let mut child = std::process::Command::new("sh") + .args(["-c", "read _; exec sleep 30"]) + .stdin(std::process::Stdio::piped()) + .spawn() + .unwrap(); + let pid = child.id(); + let link = format!("/proc/{pid}/exe"); + let before = resolve(pid); + assert_eq!(before.exe, fs::read_link(&link).unwrap()); + + child.stdin.take().unwrap().write_all(b"\n").unwrap(); + let deadline = Instant::now() + Duration::from_secs(5); + while fs::read_link(&link).unwrap() == before.exe && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(10)); + } + let mapped = fs::read_link(&link).unwrap(); + let after = resolve(pid); + let _ = child.kill(); + let _ = child.wait(); + if mapped == before.exe { + return; // sh and sleep are one multi-call binary here + } + assert_eq!(after.exe, mapped); + assert_ne!(after.sha256, before.sha256); + } + #[test] fn a_path_naming_another_file_here_is_not_the_image() { let dir = tempfile::tempdir().unwrap(); From 1e8067f1fb1d3476320770931a54f9a99cb21f24 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:52:40 +0200 Subject: [PATCH 027/125] fix(rules): never echo an unbounded client value in a refusal The exe_path length bound ran after the relative-path branch, which printed the whole path, and parent_exe and the CIDR fields were echoed in full. A 4 MiB value then came back as a gRPC status and a log line. Check the length first, drop the parent_exe echo and quote at most 64 characters of a bad CIDR. --- crates/cfc-core/src/rule.rs | 75 ++++++++++++++++++++------------ crates/cfc-daemon/src/convert.rs | 13 +++++- 2 files changed, 57 insertions(+), 31 deletions(-) diff --git a/crates/cfc-core/src/rule.rs b/crates/cfc-core/src/rule.rs index cc22cf3..a726aae 100644 --- a/crates/cfc-core/src/rule.rs +++ b/crates/cfc-core/src/rule.rs @@ -215,16 +215,16 @@ impl RuleScope { /// proto API and `cfc rules import` both take the field, which is exactly /// how the `` rule got onto a real machine. pub fn reject_unmatchable_parent(&self) -> Result<(), String> { - let Some(parent) = &self.parent_exe else { + // The value is not echoed: it is any length a client sent, and the + // message becomes a gRPC status and a log line. + if self.parent_exe.is_none() { return Ok(()); - }; - Err(format!( - "cannot scope a rule on parent_exe ({}): the predicate is not \ + } + Err("cannot scope a rule on parent_exe: the predicate is not \ evaluated, so the rule would match every process rather than the \ ones launched by it - and it would outrank narrower rules while \ - doing so. Scope on the executable itself.", - parent.display() - )) + doing so. Scope on the executable itself." + .to_string()) } /// Refuse a scope whose `exe_path` is not an absolute path. @@ -243,25 +243,8 @@ impl RuleScope { let Some(exe) = &self.exe_path else { return Ok(()); }; - if exe.as_os_str() == crate::UNKNOWN_EXE { - return Err(format!( - "cannot scope a rule to {}: that is what this program shows \ - when it could not identify the process, not a path. Such a \ - rule would match every flow that cannot be attributed, which \ - is every inbound flow. Scope it to a real executable, or use \ - a port and source instead.", - crate::UNKNOWN_EXE - )); - } - if !exe.is_absolute() { - return Err(format!( - "exe path {} is not absolute; rules match on absolute \ - executable paths, so a relative one can never fire", - exe.display() - )); - } // A path no filesystem can hold is a path no process can be running, - // so such a rule can never fire - the same test as the two above, for + // so such a rule can never fire - the same test as the two below, for // a value that arrives over the wire. // // The bound is here rather than at the wire because this is the gate @@ -272,10 +255,10 @@ impl RuleScope { // two million syscalls for a 4 MiB path, on a blocking pool of // sixteen that the prompt router also depends on. // - // The message deliberately does not print the path. Every other arm - // here formats it into a string that becomes a gRPC status and a log - // line, which for this input is the denial-of-service repeated on the - // way out. + // First, and the message deliberately does not print the path: the + // arms below format it into a string that becomes a gRPC status and a + // log line, which for this input would be the denial-of-service + // repeated on the way out. if exe.as_os_str().len() > MAX_EXE_PATH_LEN { return Err(format!( "exe path is {} bytes; the kernel cannot hold a path longer \ @@ -283,6 +266,23 @@ impl RuleScope { exe.as_os_str().len() )); } + if exe.as_os_str() == crate::UNKNOWN_EXE { + return Err(format!( + "cannot scope a rule to {}: that is what this program shows \ + when it could not identify the process, not a path. Such a \ + rule would match every flow that cannot be attributed, which \ + is every inbound flow. Scope it to a real executable, or use \ + a port and source instead.", + crate::UNKNOWN_EXE + )); + } + if !exe.is_absolute() { + return Err(format!( + "exe path {} is not absolute; rules match on absolute \ + executable paths, so a relative one can never fire", + exe.display() + )); + } Ok(()) } @@ -1640,4 +1640,21 @@ mod parent_exe_tests { fn a_scope_without_a_parent_is_untouched() { assert!(RuleScope::any().reject_unmatchable_parent().is_ok()); } + + /// A refusal becomes a gRPC status and a log line, so a 4 MiB value must + /// not come back out in it, whichever check refuses it first. + #[test] + fn refusals_never_echo_an_oversized_path() { + let huge = "a".repeat(MAX_EXE_PATH_LEN * 4); + for exe in [huge.clone(), format!("/{huge}")] { + let mut scope = RuleScope::any(); + scope.exe_path = Some(PathBuf::from(exe)); + let err = scope.reject_unmatchable_exe().expect_err("refused"); + assert!(err.len() < 512, "{} bytes", err.len()); + } + let mut scope = RuleScope::any(); + scope.parent_exe = Some(PathBuf::from(format!("/{huge}"))); + let err = scope.reject_unmatchable_parent().expect_err("refused"); + assert!(err.len() < 512, "{} bytes", err.len()); + } } diff --git a/crates/cfc-daemon/src/convert.rs b/crates/cfc-daemon/src/convert.rs index e8251f6..d8c7f2d 100644 --- a/crates/cfc-daemon/src/convert.rs +++ b/crates/cfc-daemon/src/convert.rs @@ -252,7 +252,9 @@ pub fn scope_to_pb(s: &RuleScope) -> pb::RuleScope { /// the `has_*` flags carry presence explicitly. pub fn scope_from_pb(s: &pb::RuleScope) -> Result { let dst_net = match empty_to_none(&s.dst_net) { - Some(n) => Some(ipnet::IpNet::from_str(&n).map_err(|e| format!("bad dst_net `{n}`: {e}"))?), + Some(n) => { + Some(ipnet::IpNet::from_str(&n).map_err(|e| format!("bad dst_net `{n:.64}`: {e}"))?) + } None => None, }; let protocol = match s.has_protocol { @@ -271,7 +273,9 @@ pub fn scope_from_pb(s: &pb::RuleScope) -> Result { false => None, }; let src_net = match empty_to_none(&s.src_net) { - Some(n) => Some(ipnet::IpNet::from_str(&n).map_err(|e| format!("bad src_net `{n}`: {e}"))?), + Some(n) => { + Some(ipnet::IpNet::from_str(&n).map_err(|e| format!("bad src_net `{n:.64}`: {e}"))?) + } None => None, }; let src_port = @@ -759,6 +763,11 @@ mod tests { "the message must name the field: {e}" ); assert!(e.contains("10.0.0.0/33"), "and quote the value: {e}"); + // But only so much of it: the message is a gRPC status and a log line. + let mut long = pb.clone(); + long.dst_net = "9".repeat(1 << 20); + let e = scope_from_pb(&long).expect_err("refused"); + assert!(e.len() < 256, "{} bytes", e.len()); // A rule carrying it is refused whole, rather than persisted narrower // than it reads. From 6e1f88b072634e4465bdc7b3d49c04e22872c3e8 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:53:52 +0200 Subject: [PATCH 028/125] fix(ipc): log every refused UpsertRule and ApplyRules Only a successful write and an authorization refusal reached the journal, so a validation or storage refusal looked exactly like a request that was never sent (#46). Each refusal now logs a warning with the RPC, the peer uid and pid, the status code and its message. --- crates/cfc-daemon/src/ipc.rs | 237 ++++++++++++++++++++--------------- 1 file changed, 137 insertions(+), 100 deletions(-) diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 1774001..9b997fd 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -100,6 +100,23 @@ async fn resolve_exe_off_thread(scope: &mut cfc_core::RuleScope) -> Result<(), S Ok(()) } +/// Logs a rule write the daemon refused, with its reason. +/// +/// Without it the journal held only successful writes and authorization +/// refusals, so "the client never sent it" and "the daemon refused it" looked +/// the same (issue #46). Authorization refusals are logged by `authorize`. +fn log_refusal(rpc: &'static str, peer: PeerId, status: &Status) { + warn!( + rpc, + peer_uid = peer.uid, + peer_pid = ?peer.pid, + code = ?status.code(), + reason = status.message(), + outcome = "refused", + "rule write refused" + ); +} + fn bind_prompt_allow( rule: &mut cfc_core::Rule, binding: &crate::prompts::PromptBinding, @@ -435,6 +452,118 @@ impl FirewallService { ))) } + async fn upsert_rule_checked( + &self, + peer: PeerId, + req: UpsertRuleRequest, + ) -> Result { + let proto = req + .rule + .ok_or_else(|| Status::invalid_argument("rule required"))?; + let mut rule = convert::rule_from_pb(&proto).map_err(Status::invalid_argument)?; + convert::reject_unpersistable_duration(rule.duration).map_err(Status::invalid_argument)?; + // Every caller must select the canonical mapped target explicitly. + // Missing targets with unchanged ancestry remain valid for preinstallation. + resolve_exe_off_thread(&mut rule.scope).await?; + // hit_count and created_at belong to the daemon: a client editing a + // rule must not be able to rewrite its history, deliberately or (as + // every read-modify-write client did) by echoing back a count that + // already included an unflushed delta. + let _mutation = self.mutations.lock(); + if rule.duration == cfc_core::Duration::Always + && self.engine.snapshot().rules.iter().any(|old| { + old.id == rule.id && matches!(old.duration, cfc_core::Duration::Seconds(_)) + }) + { + return Err(Status::invalid_argument("a timed rule cannot be changed to Always by an older read-modify-write client; delete and recreate it explicitly")); + } + self.engine.preserve_server_owned(&mut rule); + self.store + .upsert(&rule) + .map_err(|e| Status::internal(format!("storage: {e}")))?; + let id = rule.id.to_string(); + info!( + rpc = "UpsertRule", + peer_uid = peer.uid, + peer_pid = ?peer.pid, + rule_id = %id, + action = ?rule.action, + duration = ?rule.duration, + enabled = rule.enabled, + outcome = "ok", + "rule upserted" + ); + self.engine.upsert_rule(rule); + Ok(UpsertRuleResponse { + id, + error: String::new(), + }) + } + + async fn apply_rules_checked( + &self, + peer: PeerId, + req: ApplyRulesRequest, + ) -> Result { + if req.replace && req.rules.is_empty() { + return Err(Status::invalid_argument("refusing an empty replacement")); + } + let mut pending = Vec::with_capacity(req.rules.len()); + let mut ids = HashSet::new(); + for proto in req.rules { + let mut rule = convert::rule_from_pb(&proto).map_err(Status::invalid_argument)?; + convert::reject_unpersistable_duration(rule.duration) + .map_err(Status::invalid_argument)?; + if !ids.insert(rule.id) { + return Err(Status::invalid_argument("duplicate rule id")); + } + resolve_exe_off_thread(&mut rule.scope).await?; + pending.push(rule); + } + let _mutation = self.mutations.lock(); + let existing = self.engine.snapshot(); + for rule in &mut pending { + if rule.duration == cfc_core::Duration::Always + && existing.rules.iter().any(|old| { + old.id == rule.id && matches!(old.duration, cfc_core::Duration::Seconds(_)) + }) + { + return Err(Status::invalid_argument( + "a timed rule cannot become Always in place; delete it and create a new rule", + )); + } + self.engine.preserve_server_owned(rule); + } + let removed = self + .store + .apply_rules(&pending, req.replace) + .map_err(|e| Status::internal(format!("storage: {e}")))?; + let assigned: Vec = pending.iter().map(|rule| rule.id.to_string()).collect(); + info!( + rpc = "ApplyRules", + peer_uid = peer.uid, + peer_pid = ?peer.pid, + replace = req.replace, + applied = assigned.len(), + removed, + rule_ids = ?assigned, + outcome = "ok", + "rules applied" + ); + let mut final_rules = if req.replace { + Vec::new() + } else { + self.engine.snapshot().rules + }; + final_rules.retain(|rule| !ids.contains(&rule.id)); + final_rules.extend(pending); + self.engine.replace_rules(final_rules); + Ok(ApplyRulesResponse { + ids: assigned, + removed: u32::try_from(removed).unwrap_or(u32::MAX), + }) + } + fn policy(&self) -> crate::config::DefaultPolicy { *self .policy @@ -679,48 +808,10 @@ impl Firewall for FirewallService { req: Request, ) -> Result, Status> { let peer = self.authorize(&req, Access::Mutate)?; - let proto = req - .into_inner() - .rule - .ok_or_else(|| Status::invalid_argument("rule required"))?; - let mut rule = convert::rule_from_pb(&proto).map_err(Status::invalid_argument)?; - convert::reject_unpersistable_duration(rule.duration).map_err(Status::invalid_argument)?; - // Every caller must select the canonical mapped target explicitly. - // Missing targets with unchanged ancestry remain valid for preinstallation. - resolve_exe_off_thread(&mut rule.scope).await?; - // hit_count and created_at belong to the daemon: a client editing a - // rule must not be able to rewrite its history, deliberately or (as - // every read-modify-write client did) by echoing back a count that - // already included an unflushed delta. - let _mutation = self.mutations.lock(); - if rule.duration == cfc_core::Duration::Always - && self.engine.snapshot().rules.iter().any(|old| { - old.id == rule.id && matches!(old.duration, cfc_core::Duration::Seconds(_)) - }) - { - return Err(Status::invalid_argument("a timed rule cannot be changed to Always by an older read-modify-write client; delete and recreate it explicitly")); - } - self.engine.preserve_server_owned(&mut rule); - self.store - .upsert(&rule) - .map_err(|e| Status::internal(format!("storage: {e}")))?; - let id = rule.id.to_string(); - info!( - rpc = "UpsertRule", - peer_uid = peer.uid, - peer_pid = ?peer.pid, - rule_id = %id, - action = ?rule.action, - duration = ?rule.duration, - enabled = rule.enabled, - outcome = "ok", - "rule upserted" - ); - self.engine.upsert_rule(rule); - Ok(Response::new(UpsertRuleResponse { - id, - error: String::new(), - })) + self.upsert_rule_checked(peer, req.into_inner()) + .await + .map(Response::new) + .inspect_err(|status| log_refusal("UpsertRule", peer, status)) } async fn apply_rules( @@ -728,64 +819,10 @@ impl Firewall for FirewallService { req: Request, ) -> Result, Status> { let peer = self.authorize(&req, Access::Mutate)?; - let req = req.into_inner(); - if req.replace && req.rules.is_empty() { - return Err(Status::invalid_argument("refusing an empty replacement")); - } - let mut pending = Vec::with_capacity(req.rules.len()); - let mut ids = HashSet::new(); - for proto in req.rules { - let mut rule = convert::rule_from_pb(&proto).map_err(Status::invalid_argument)?; - convert::reject_unpersistable_duration(rule.duration) - .map_err(Status::invalid_argument)?; - if !ids.insert(rule.id) { - return Err(Status::invalid_argument("duplicate rule id")); - } - resolve_exe_off_thread(&mut rule.scope).await?; - pending.push(rule); - } - let _mutation = self.mutations.lock(); - let existing = self.engine.snapshot(); - for rule in &mut pending { - if rule.duration == cfc_core::Duration::Always - && existing.rules.iter().any(|old| { - old.id == rule.id && matches!(old.duration, cfc_core::Duration::Seconds(_)) - }) - { - return Err(Status::invalid_argument( - "a timed rule cannot become Always in place; delete it and create a new rule", - )); - } - self.engine.preserve_server_owned(rule); - } - let removed = self - .store - .apply_rules(&pending, req.replace) - .map_err(|e| Status::internal(format!("storage: {e}")))?; - let assigned: Vec = pending.iter().map(|rule| rule.id.to_string()).collect(); - info!( - rpc = "ApplyRules", - peer_uid = peer.uid, - peer_pid = ?peer.pid, - replace = req.replace, - applied = assigned.len(), - removed, - rule_ids = ?assigned, - outcome = "ok", - "rules applied" - ); - let mut final_rules = if req.replace { - Vec::new() - } else { - self.engine.snapshot().rules - }; - final_rules.retain(|rule| !ids.contains(&rule.id)); - final_rules.extend(pending); - self.engine.replace_rules(final_rules); - Ok(Response::new(ApplyRulesResponse { - ids: assigned, - removed: u32::try_from(removed).unwrap_or(u32::MAX), - })) + self.apply_rules_checked(peer, req.into_inner()) + .await + .map(Response::new) + .inspect_err(|status| log_refusal("ApplyRules", peer, status)) } async fn delete_rule( From 0032edff937ac348c8dff76752f8cdf5474eefdb Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:54:29 +0200 Subject: [PATCH 029/125] docs(ipc): prompt ids start from a random seed each session The SubmitVerdict comment still said ids restart at 1 on every daemon start, which stopped being true when the worker began seeding them at random. Correct it and pin the seeding with a test. --- crates/cfc-daemon/src/ipc.rs | 5 +++-- crates/cfc-daemon/src/nfqueue.rs | 7 +++++++ 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 9b997fd..2926b15 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -665,8 +665,9 @@ impl Firewall for FirewallService { // prompt was gone - so a click on a card whose prompt had already timed // out created a permanent rule while every client said "too late". For // "Allow always" that is standing network access granted by a click the - // user was told did nothing, and prompt ids restart at 1 on every daemon - // start, so a stale card can carry a live id. A verdict that reached + // user was told did nothing. Prompt ids start from a random seed each + // session (`nfqueue::prompt_session_seed`), which makes a stale card + // naming a live id unlikely, not impossible. A verdict that reached // nothing should leave nothing behind. let binding = self.router.submit(&req.prompt_id, verdict); let accepted = binding.is_some(); diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index a23dc49..b7b2afa 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -1288,6 +1288,13 @@ mod tests { const IPPROTO_TCP: u8 = 6; + /// A card left over from an earlier session must not name the id this + /// session hands out first, so ids do not restart at a fixed value. + #[test] + fn each_session_starts_prompt_ids_from_a_fresh_seed() { + assert_ne!(prompt_session_seed(), prompt_session_seed()); + } + /// Minimal IPv4/TCP packet: 1.2.3.4:5555 -> 5.6.7.8:`dst_port`. fn tcp_packet(dst_port: u16) -> Vec { let mut pkt = vec![0u8; 40]; From 03b51fc318c648602bb558fa2cfc6bcc9a0ffb66 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:55:57 +0200 Subject: [PATCH 030/125] fix(ipc): accept a stored executable path sent back unchanged Every upsert re-ran the alias check, so a legacy rule such as /bin/curl, or one whose target a package later turned into a symlink, could not be disabled, renamed or re-imported; only delete worked. The daemon now validates only a new rule or a changed path, and cfc rules disable no longer re-checks the path it just read back. TODO.md still described the pre-validation contract; it now says aliases are refused at entry, and that the daemon's check cannot see paths its ProtectHome and PrivateTmp sandbox hides. --- TODO.md | 35 ++++++++++++++----------- crates/cfc-cli/src/rules.rs | 11 +++----- crates/cfc-daemon/src/ipc.rs | 50 ++++++++++++++++++++++++++++++++++-- 3 files changed, 71 insertions(+), 25 deletions(-) diff --git a/TODO.md b/TODO.md index 1f80040..0d869f2 100644 --- a/TODO.md +++ b/TODO.md @@ -98,21 +98,26 @@ index - but "should be fine" is not "was observed". ## 3. Executable paths: what resolution does and does not fix -Rules now resolve their `exe_path` to the form `/proc//exe` reports, at -every place a path is entered (`cfc_core::exe_path`). Three properties of that -are worth stating rather than discovering: - -- **Forward-only.** Rules already on disk are never re-resolved. An install - that wrote `/bin/curl` before this existed keeps an inert rule after - upgrading. The repair is one round trip - `cfc rules export > r.json && - cfc rules import --replace r.json` - because upsert resolves. -- **A versioned symlink resolves to a version.** `/usr/bin/python -> - python3.13` stores `python3.13` and stops applying when the symlink moves. - Not a regression (the unresolved rule never matched either), but a new - *time-dependent* failure, and worse for a Deny than an Allow. -- **It follows symlinks the path's owner controls.** A rule for - `/home/bob/tool` pointing at `/usr/bin/curl` becomes a rule about curl. The - CLI prints what it stored and the daemon warns; nothing pins the inode. +Rules must name the form `/proc//exe` reports (`cfc_core::exe_path`): +every place a new or changed path is entered refuses an alias and asks for +the canonical target. Three properties of that are worth stating rather than +discovering: + +- **Stored rules keep their target.** Nothing re-resolves a rule on disk. An + install that wrote `/bin/curl` before validation existed keeps an inert + rule after upgrading, and so does a rule whose target a package update later + turned into a symlink. Sending the stored path back unchanged (enable, + disable, rename, `cfc rules export` then `import --replace`) is accepted; + the repair is to edit the rule to name `/usr/bin/curl`. +- **A versioned target is a version.** `/usr/bin/python -> python3.13` has to + be written as `python3.13`, and stops applying when the symlink moves. Not a + regression (the alias never matched either), but a *time-dependent* failure, + and worse for a Deny than an Allow. +- **The daemon cannot see every alias.** It runs with `ProtectHome=true` and + `PrivateTmp=true`, so `/home/bob/tool -> /usr/bin/curl` looks like a target + that is not installed yet and is accepted. The CLI and GUI check in the + caller's own namespace first; a raw gRPC client is not stopped. Such a rule + names a path the user controls and matches only that path, never curl. Process resolution now rereads policy identity for every packet lookup; pid and start time do not identify an executable across exec. Its path and digest diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 96b1535..e586df7 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -269,15 +269,10 @@ pub async fn set_enabled( let was = rule.enabled; let want = target.unwrap_or(!was); + // No executable validation here: the path is the stored one, sent back + // unchanged, and the daemon validates only new or changed paths. Checking + // it again refused to disable a rule whose target became an alias. if want != was { - if let Some(scope) = rule - .scope - .as_ref() - .filter(|scope| !scope.exe_path.is_empty()) - { - cfc_core::exe_path::resolve_policy(std::path::Path::new(&scope.exe_path)) - .map_err(CliError::runtime)?; - } rule.enabled = want; client.upsert_rule(rule.clone()).await?; } diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 2926b15..465e4d1 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -81,6 +81,12 @@ use tracing::{info, warn}; /// /// Missing canonical targets support preinstallation; aliases, other lookup /// failures and worker failures refuse the policy write. +/// +/// Advisory for paths the unit's sandbox hides: under `ProtectHome` and +/// `PrivateTmp` an alias in `/home` or `/tmp` looks like a target that is not +/// installed yet and is accepted. The CLI and GUI run the same check in the +/// caller's namespace first. The stored path is still matched literally, so +/// such a rule never applies to the alias's target. async fn resolve_exe_off_thread(scope: &mut cfc_core::RuleScope) -> Result<(), Status> { let Some(current) = scope.exe_path.clone() else { return Ok(()); @@ -100,6 +106,22 @@ async fn resolve_exe_off_thread(scope: &mut cfc_core::RuleScope) -> Result<(), S Ok(()) } +/// Whether `rule` sends back the executable path already stored under its id. +/// +/// That path was validated when it was written, or predates validation, and +/// sending it back changes nothing about what the rule matches. Validating it +/// again refused every edit of a rule whose target had since become an alias +/// (a package update turned it into a symlink, or a legacy `/bin/curl`), so +/// disabling, renaming or re-importing it failed and only delete was left. A +/// new rule or a changed path is still validated. +fn keeps_stored_exe(stored: &cfc_core::RuleSet, rule: &cfc_core::Rule) -> bool { + rule.scope.exe_path.is_some() + && stored + .rules + .iter() + .any(|old| old.id == rule.id && old.scope.exe_path == rule.scope.exe_path) +} + /// Logs a rule write the daemon refused, with its reason. /// /// Without it the journal held only successful writes and authorization @@ -464,7 +486,9 @@ impl FirewallService { convert::reject_unpersistable_duration(rule.duration).map_err(Status::invalid_argument)?; // Every caller must select the canonical mapped target explicitly. // Missing targets with unchanged ancestry remain valid for preinstallation. - resolve_exe_off_thread(&mut rule.scope).await?; + if !keeps_stored_exe(&self.engine.snapshot(), &rule) { + resolve_exe_off_thread(&mut rule.scope).await?; + } // hit_count and created_at belong to the daemon: a client editing a // rule must not be able to rewrite its history, deliberately or (as // every read-modify-write client did) by echoing back a count that @@ -510,6 +534,7 @@ impl FirewallService { } let mut pending = Vec::with_capacity(req.rules.len()); let mut ids = HashSet::new(); + let stored = self.engine.snapshot(); for proto in req.rules { let mut rule = convert::rule_from_pb(&proto).map_err(Status::invalid_argument)?; convert::reject_unpersistable_duration(rule.duration) @@ -517,7 +542,9 @@ impl FirewallService { if !ids.insert(rule.id) { return Err(Status::invalid_argument("duplicate rule id")); } - resolve_exe_off_thread(&mut rule.scope).await?; + if !keeps_stored_exe(&stored, &rule) { + resolve_exe_off_thread(&mut rule.scope).await?; + } pending.push(rule); } let _mutation = self.mutations.lock(); @@ -1449,6 +1476,25 @@ mod tests { assert_eq!(scope.exe_path, Some(target)); } + #[test] + fn only_an_unchanged_stored_exe_skips_validation() { + let mut scope = cfc_core::RuleScope::any(); + scope.exe_path = Some(PathBuf::from("/bin/curl")); + let old = cfc_core::Rule::new("legacy", cfc_core::Action::Deny, scope); + let stored = cfc_core::RuleSet { + rules: vec![old.clone()], + }; + let mut toggled = old.clone(); + toggled.enabled = false; + assert!(keeps_stored_exe(&stored, &toggled)); + let mut moved = old.clone(); + moved.scope.exe_path = Some(PathBuf::from("/bin/wget")); + assert!(!keeps_stored_exe(&stored, &moved)); + let mut fresh = old.clone(); + fresh.id = uuid::Uuid::new_v4(); + assert!(!keeps_stored_exe(&stored, &fresh)); + } + // -- authorization ------------------------------------------------------ #[test] From 0d9d81ac3317639350d905ad580f2c56995dd7ac Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:56:32 +0200 Subject: [PATCH 031/125] fix(convert): never accept a rule creation date in the future A timed rule expires at created_at plus its duration, and a new rule took created_at from the client. A date in 2100 made an allow for 90s permanent while every list still showed 90s. Clamp it to now; a past date, as an exported rule carries, is kept. --- crates/cfc-daemon/src/convert.rs | 33 ++++++++++++++++++++++++++------ 1 file changed, 27 insertions(+), 6 deletions(-) diff --git a/crates/cfc-daemon/src/convert.rs b/crates/cfc-daemon/src/convert.rs index d8c7f2d..49e3cd3 100644 --- a/crates/cfc-daemon/src/convert.rs +++ b/crates/cfc-daemon/src/convert.rs @@ -362,12 +362,14 @@ pub fn rule_from_pb(r: &pb::RuleInfo) -> Result { scope.reject_unmatchable_parent()?; scope.reject_inbound_destination_scope()?; scope.reject_unattributable_inbound_scope()?; - let created_at = if r.created_at_unix_ms == 0 { - chrono::Utc::now() - } else { - chrono::DateTime::from_timestamp_millis(r.created_at_unix_ms) - .unwrap_or_else(chrono::Utc::now) - }; + // Never in the future: a timed rule expires at created_at + n, so a + // client-supplied date in 2100 made "allow for 90s" permanent while every + // list still showed 90s. A past date is kept, which is what an import + // of an exported rule needs. + let now = chrono::Utc::now(); + let created_at = chrono::DateTime::from_timestamp_millis(r.created_at_unix_ms) + .filter(|_| r.created_at_unix_ms != 0) + .map_or(now, |at| at.min(now)); Ok(Rule { id, name: r.name.clone(), @@ -839,6 +841,25 @@ mod tests { assert!(rule_from_pb(&pb).is_err()); } + #[test] + fn a_future_creation_date_cannot_postpone_expiry() { + let mut scope = RuleScope::any(); + scope.dst_port = Some(443); + let mut pb = rule_to_pb(&Rule::new("x", Action::Allow, scope)); + pb.duration = cfc_proto::v1::Duration::Seconds as i32; + pb.duration_seconds = 90; + let year_2100 = 4_102_444_800_000; + pb.created_at_unix_ms = year_2100; + let rule = rule_from_pb(&pb).unwrap(); + assert!(rule.created_at <= chrono::Utc::now()); + // A past date, as an export carries, is kept. + pb.created_at_unix_ms = 1_000; + assert_eq!( + rule_from_pb(&pb).unwrap().created_at.timestamp_millis(), + 1_000 + ); + } + #[test] fn connection_to_pb_carries_5tuple() { let conn = cfc_core::Connection::new( From b21d7db1adec56552b81d738ff4c25f5e93d22b1 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:56:32 +0200 Subject: [PATCH 032/125] fix(ipc): say plainly why a timed rule cannot become Always UpsertRule blamed an older read-modify-write client, which misleads a current GUI user who picked Always on purpose. Use the same wording as ApplyRules: delete the rule and create a new one. --- crates/cfc-daemon/src/ipc.rs | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 465e4d1..7a8b9de 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -499,7 +499,9 @@ impl FirewallService { old.id == rule.id && matches!(old.duration, cfc_core::Duration::Seconds(_)) }) { - return Err(Status::invalid_argument("a timed rule cannot be changed to Always by an older read-modify-write client; delete and recreate it explicitly")); + return Err(Status::invalid_argument( + "a timed rule cannot become Always in place; delete it and create a new rule", + )); } self.engine.preserve_server_owned(&mut rule); self.store From 5da95626764a7480ad0bc2e5a608854fee6bf6d0 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:57:59 +0200 Subject: [PATCH 033/125] fix(daemon): stop a rule write from double-counting flushed hits The flush task folds drained hit deltas into the in-memory rules and then adds them to the stored rows. An UpsertRule landing between the two stored the folded count, and the merge added the delta again, inflating the count for good. The IPC mutation lock moves into the engine and the flush task holds it for its whole tick. --- crates/cfc-daemon/src/decision.rs | 19 +++++++++++++++++++ crates/cfc-daemon/src/ipc.rs | 10 ++++------ crates/cfc-daemon/src/main.rs | 3 +++ 3 files changed, 26 insertions(+), 6 deletions(-) diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index d6670f7..fc15b91 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -42,6 +42,9 @@ struct EngineInner { /// it a `Weak` capture on the caller's side is what stops the cycle /// (the observer holds this `Engine`). on_change: RwLock>>, + /// Serializes every write that touches both the store and these rules. + /// See [`Engine::lock_mutations`]. + mutations: Mutex<()>, } pub enum Decision { @@ -63,10 +66,26 @@ impl Engine { default_policy, hits: Mutex::new(HashMap::new()), on_change: RwLock::new(None), + mutations: Mutex::new(()), }), } } + /// Held across any change that writes the rule store and this engine + /// together: the IPC rule writes, and the flush task's hit merge and + /// expiry. + /// + /// The store and the engine are two copies of one rule set, and without a + /// shared lock their writers interleave. The flush drains hit deltas into + /// these rules, then adds them to the stored rows; an UpsertRule landing in + /// between stored the already-folded count, and the merge added the delta + /// a second time, inflating the count for good. + /// + /// Never taken on the packet path. Lock order: this before `rules`. + pub fn lock_mutations(&self) -> parking_lot::MutexGuard<'_, ()> { + self.inner.mutations.lock() + } + /// Registers the callback invoked after every rule-set change. /// /// One observer, last writer wins. The callback must not touch this engine diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 7a8b9de..68e6911 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -447,7 +447,6 @@ struct FirewallService { resume_at_ms: Arc, pause_default_secs: u64, dry_run: bool, - mutations: Mutex<()>, } impl FirewallService { @@ -493,7 +492,7 @@ impl FirewallService { // rule must not be able to rewrite its history, deliberately or (as // every read-modify-write client did) by echoing back a count that // already included an unflushed delta. - let _mutation = self.mutations.lock(); + let _mutation = self.engine.lock_mutations(); if rule.duration == cfc_core::Duration::Always && self.engine.snapshot().rules.iter().any(|old| { old.id == rule.id && matches!(old.duration, cfc_core::Duration::Seconds(_)) @@ -549,7 +548,7 @@ impl FirewallService { } pending.push(rule); } - let _mutation = self.mutations.lock(); + let _mutation = self.engine.lock_mutations(); let existing = self.engine.snapshot(); for rule in &mut pending { if rule.duration == cfc_core::Duration::Always @@ -759,7 +758,7 @@ impl Firewall for FirewallService { if rule.action == cfc_core::Action::Allow && binding.hash_expected { persist_note = "the allow is bound to the prompted binary's sha256; a changed file will prompt again".into(); } - let _mutation = self.mutations.lock(); + let _mutation = self.engine.lock_mutations(); match self.store.upsert(&rule) { Ok(()) => { persisted_rule = Some(rule.id); @@ -863,7 +862,7 @@ impl Firewall for FirewallService { let id_str = req.into_inner().id; let id = uuid::Uuid::parse_str(&id_str) .map_err(|e| Status::invalid_argument(format!("bad uuid: {e}")))?; - let _mutation = self.mutations.lock(); + let _mutation = self.engine.lock_mutations(); let deleted = self .store .delete(id) @@ -1436,7 +1435,6 @@ pub async fn spawn( resume_at_ms: Arc::new(AtomicI64::new(0)), pause_default_secs: opts.pause_default_secs, dry_run: opts.dry_run, - mutations: Mutex::new(()), }; info!(socket = %socket_path.display(), "IPC listening"); diff --git a/crates/cfc-daemon/src/main.rs b/crates/cfc-daemon/src/main.rs index 683aa6d..1c1fd32 100644 --- a/crates/cfc-daemon/src/main.rs +++ b/crates/cfc-daemon/src/main.rs @@ -217,6 +217,9 @@ async fn run() -> anyhow::Result<()> { tick.tick().await; // skip immediate fire loop { tick.tick().await; + // The whole tick: a rule write between the drain and the merge + // would count the drained hits twice (see `lock_mutations`). + let _mutation = flush_engine.lock_mutations(); let deltas = flush_engine.drain_hits(); if !deltas.is_empty() { if let Err(e) = flush_store.merge_hit_counts(&deltas) { From 1b78ed20ded49014513ffbdbc4b8fce296dc77e0 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:58:04 +0200 Subject: [PATCH 034/125] fix(ebpf): serialize verdict resyncs and observe rule changes from the start Two resyncs could overlap (the flush task's expiry runs outside the IPC lock), and the one that read the older rules could write last, clearing a deny the newer rules had just installed. A lock now covers each run. The rule-change observer was registered after the startup resync, while IPC is already serving, so a rule changed in between never reached the kernel tables. Register it first; an extra resync is harmless. --- crates/cfc-daemon/src/ebpf/enforce.rs | 8 ++++++++ crates/cfc-daemon/src/ebpf/loader.rs | 7 ++++++- 2 files changed, 14 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/ebpf/enforce.rs b/crates/cfc-daemon/src/ebpf/enforce.rs index 364e6df..e21018c 100644 --- a/crates/cfc-daemon/src/ebpf/enforce.rs +++ b/crates/cfc-daemon/src/ebpf/enforce.rs @@ -179,6 +179,10 @@ pub(super) struct VerdictSink { /// What was last written to the kernel, so an unchanged recompute costs no /// syscalls. `None` until the first compile. last_compiled: Arc>>>, + /// Held for a whole [`Self::resync`]. Each run decides from the rules it + /// read and writes afterwards, so two overlapping runs could finish in the + /// wrong order and leave the older rule set's answers in the kernel. + resync: Arc>, } impl VerdictSink { @@ -218,6 +222,7 @@ impl VerdictSink { exe_rules, exe_rules_on, last_compiled: Arc::new(Mutex::new(None)), + resync: Arc::new(Mutex::new(())), }) } @@ -249,6 +254,9 @@ impl VerdictSink { /// all. The orphan sweep has said so since it was written; the live loop /// inherited the constraint the moment it started reading /proc too. pub(super) fn resync(&self) { + // One run at a time, so the last run to write is the one that read the + // newest rules. Not the map lock: the ring consumers never wait on this. + let _run = self.resync.lock(); // The kernel's table first: it governs processes that do not exist yet, // and it is what survives this daemon. self.compile_rules(); diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index a88cc90..fdd383b 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -778,13 +778,18 @@ pub(super) fn load_and_attach( // `set_live` has not run yet - so the live loop no-ops, // and it is the orphan sweep that reconciles the denials // the previous daemon left in `VERDICTS`. - sink.resync(); + // + // The observer goes in first. IPC is already serving, and + // a rule changed after this resync read the rules but + // before an observer existed would never reach the kernel. + // An extra resync is harmless; runs are serialized. let weak = std::sync::Arc::downgrade(&sink); engine.set_on_change(Box::new(move || { if let Some(sink) = weak.upgrade() { sink.resync(); } })); + sink.resync(); Some(sink) } Err(e) => { From b54bbfad05e28ab470a5d4d892113d92ffeaf873 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 01:58:04 +0200 Subject: [PATCH 035/125] fix(ipc): create the socket directory 0755 whatever the umask create_dir_all ran before the umask was tightened, so a daemon started by hand under umask 000 made the directory world-writable and any local user could replace the socket with their own. Systemd units were not affected: RuntimeDirectoryMode creates it first. --- crates/cfc-daemon/src/ipc.rs | 12 ++++++++++-- 1 file changed, 10 insertions(+), 2 deletions(-) diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 68e6911..3c7baba 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -55,7 +55,7 @@ use anyhow::Context; use futures::StreamExt; use parking_lot::Mutex; use std::collections::{HashMap, HashSet, VecDeque}; -use std::os::unix::fs::PermissionsExt; +use std::os::unix::fs::{DirBuilderExt, PermissionsExt}; use std::path::{Path, PathBuf}; use std::pin::Pin; use std::sync::atomic::{AtomicI64, AtomicU64, Ordering}; @@ -1401,7 +1401,15 @@ pub async fn spawn( ) -> anyhow::Result<(JoinHandle<()>, PromptTx)> { let socket_path = opts.socket_path; if let Some(parent) = socket_path.parent() { - std::fs::create_dir_all(parent).ok(); + // An explicit mode: the umask is only tightened below, and a daemon + // started by hand under umask 000 made this directory world-writable, + // so any local user could swap the socket for one of their own. + // Systemd's RuntimeDirectoryMode creates it 0755 before we get here. + std::fs::DirBuilder::new() + .recursive(true) + .mode(0o755) + .create(parent) + .ok(); } let _ = std::fs::remove_file(&socket_path); From fa77346d82f6687f586d7048cc71fbaedcb7ed6d Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:00:24 +0200 Subject: [PATCH 036/125] perf(ebpf): evict an eighth of a full process table at once At the cap every new exec pruned and then searched all 10,240 entries for the single oldest, under the write lock that packet-path lookups take. Evicting the oldest eighth in one pass lets the next 1,280 execs insert without scanning. --- crates/cfc-daemon/src/ebpf/proc_table.rs | 19 +++++++++++++++---- 1 file changed, 15 insertions(+), 4 deletions(-) diff --git a/crates/cfc-daemon/src/ebpf/proc_table.rs b/crates/cfc-daemon/src/ebpf/proc_table.rs index 2c71ca7..d3dd409 100644 --- a/crates/cfc-daemon/src/ebpf/proc_table.rs +++ b/crates/cfc-daemon/src/ebpf/proc_table.rs @@ -69,7 +69,8 @@ use std::time::{Duration, Instant}; const ENTRY_TTL: Duration = Duration::from_secs(3600); /// Hard cap on live entries, matching the kernel map's `max_entries`. When -/// full, expired entries are pruned first and then the oldest is evicted. +/// full, expired entries are pruned first and then the oldest eighth is +/// evicted. const MAX_ENTRIES: usize = 10_240; /// One process as the kernel described it at `execve()` time. @@ -200,9 +201,14 @@ impl KernelProcTable { if map.len() >= MAX_ENTRIES && !map.contains_key(&proc.pid) { map.retain(|_, e| now.saturating_duration_since(e.seen_at) <= ENTRY_TTL); if map.len() >= MAX_ENTRIES { - if let Some(oldest) = map.iter().min_by_key(|(_, e)| e.seen_at).map(|(k, _)| *k) { - map.remove(&oldest); - } + // The oldest eighth in one pass, not the single oldest: that + // made every exec at the cap scan the whole table twice under + // the write lock packet-path lookups wait on. Now the next + // MAX_ENTRIES / 8 execs insert without scanning. + let mut ages: Vec = map.values().map(|e| e.seen_at).collect(); + let (_, cutoff, _) = ages.select_nth_unstable(MAX_ENTRIES / 8); + let cutoff = *cutoff; + map.retain(|_, e| e.seen_at > cutoff); } } map.insert( @@ -498,6 +504,11 @@ mod tests { ); } assert!(t.len() <= MAX_ENTRIES, "len {} > cap", t.len()); + assert!( + t.len() <= MAX_ENTRIES - MAX_ENTRIES / 8 + 16, + "one full scan makes room for many execs, len {}", + t.len() + ); let last = MAX_ENTRIES as u32 + 15; assert!( t.get(last, Some(u64::from(last)), now).is_some(), From 50c7cb9c50c7b313fbef89d46e03aa1b6ddfd7f5 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:00:26 +0200 Subject: [PATCH 037/125] fix(ebpf): recheck exit candidates instead of dropping them sched_process_exit fires before the task is reaped, so the exit consumer usually found the group still in /proc and dropped the event for good. On kernels without group_dead the pinned deny then stayed until a rule change or restart, where a recycled pid could inherit it, and on every kernel the identity stayed until its one-hour TTL. A zombie leader with no other thread now counts as gone, and a candidate still running is checked again every two seconds until its group is gone, or dropped once its pid belongs to another process. --- crates/cfc-daemon/src/ebpf/loader.rs | 133 ++++++++++++++++++++++++--- docs/ARCHITECTURE.md | 12 ++- 2 files changed, 128 insertions(+), 17 deletions(-) diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index fdd383b..98162d7 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -852,21 +852,28 @@ pub(super) fn load_and_attach( if report.exec_tracking && report.exit_tracking { let t = table.clone(); let sink_exit = sink.clone(); + let pending = PendingExits::default(); + let queue = pending.clone(); match spawn_ring(&mut bpf, MAP_EXIT, move |bytes| { if let Some(event) = decode::(bytes) { - if !process_group_is_gone(event.pid) { - return; - } - t.observe_exit(event.pid); - // The verdict map is pinned, so an entry the daemon forgets - // outlives the daemon. Evicting here is what stops a recycled - // pid inheriting a dead process's answer. - if let Some(sink) = &sink_exit { - sink.on_exit(event.pid); + // Dated before the check, so a later recheck can tell this + // process from the next owner of its pid. + let born = crate::process_resolve::read_starttime(event.pid); + if process_group_is_gone(event.pid) { + evict_exited(&t, sink_exit.as_deref(), event.pid); + } else { + let mut queue = queue.lock(); + if queue.len() >= MAX_PENDING_EXITS { + queue.remove(0); + } + queue.push((event.pid, born)); } } }) { - Ok(task) => tasks.push(task), + Ok(task) => { + tasks.push(task); + tasks.push(spawn_exit_recheck(pending, table.clone(), sink.clone())); + } Err(e) => { // Same reasoning as above: no eviction stream, no kernel // identity. @@ -1200,10 +1207,92 @@ fn exec_process(event: &ExecEvent) -> Process { } } -/// A leader exit event is only a candidate on compatibility kernels. Any -/// readable task directory or permission failure preserves identity and deny. +/// Whether an exit candidate's whole thread group has finished. +/// +/// `sched_process_exit` fires from `do_exit`, before the task is reaped, so the +/// group is usually still in /proc when its event arrives. An absent +/// `/proc//task` means reaped; a zombie leader that lists only itself has +/// no thread left either, only its parent's wait. A leader that exited before +/// its workers lists them, and any other read failure preserves identity and +/// deny. fn process_group_is_gone(pid: u32) -> bool { - matches!(std::fs::read_dir(format!("/proc/{pid}/task")), Err(e) if e.kind() == std::io::ErrorKind::NotFound) + let tasks = match std::fs::read_dir(format!("/proc/{pid}/task")) { + Ok(tasks) => tasks, + Err(e) => return e.kind() == std::io::ErrorKind::NotFound, + }; + let only_leader = tasks + .map(|task| task.map(|task| task.file_name())) + .collect::, _>>() + .is_ok_and(|names| names.len() == 1 && names[0] == *pid.to_string()); + only_leader + && std::fs::read_to_string(format!("/proc/{pid}/stat")).is_ok_and(|stat| { + let state = stat + .rsplit_once(')') + .and_then(|(_, rest)| rest.split_whitespace().next()); + matches!(state, Some("Z" | "X")) + }) +} + +/// Exit candidates waiting for their group to finish, at most. Each is a +/// leader still running its exit or a leader whose workers outlive it, so the +/// list is short; past the cap the oldest is dropped and its entries wait for +/// the next resync or restart, as every candidate did before. +const MAX_PENDING_EXITS: usize = 1024; + +/// Exit candidates still visible on arrival, oldest first, each with the +/// start time read when its event came in. +type PendingExits = std::sync::Arc)>>>; + +/// How often the pending exit candidates are looked at again. +const EXIT_RECHECK_INTERVAL: std::time::Duration = std::time::Duration::from_secs(2); + +/// Drops an exited process's identity and in-kernel verdict. +/// +/// The verdict map is pinned, so an entry the daemon forgets outlives the +/// daemon. Evicting is what stops a recycled pid inheriting a dead process's +/// answer. +fn evict_exited(table: &KernelProcTable, sink: Option<&enforce::VerdictSink>, pid: u32) { + table.observe_exit(pid); + if let Some(sink) = sink { + sink.on_exit(pid); + } +} + +/// Re-examines exit candidates whose group was still visible on arrival. +/// +/// A one-shot check dropped them, and the pinned deny and the identity then +/// stayed until a rule change or a restart. A candidate is evicted once its +/// group is gone, and forgotten without eviction once its pid belongs to +/// another process: that owner's exec has replaced both entries, and clearing +/// them could erase its fresh deny. +fn spawn_exit_recheck( + pending: PendingExits, + table: KernelProcTable, + sink: Option>, +) -> JoinHandle<()> { + tokio::spawn(async move { + let mut tick = tokio::time::interval(EXIT_RECHECK_INTERVAL); + loop { + tick.tick().await; + let candidates = std::mem::take(&mut *pending.lock()); + let waiting: Vec<_> = candidates + .into_iter() + .filter(|&(pid, born)| { + if process_group_is_gone(pid) { + evict_exited(&table, sink.as_deref(), pid); + return false; + } + // Unreadable for now: keep it, the next tick decides. + crate::process_resolve::read_starttime(pid).is_none_or(|now| Some(now) == born) + }) + .collect(); + if !waiting.is_empty() { + let mut queue = pending.lock(); + queue.splice(0..0, waiting); + queue.truncate(MAX_PENDING_EXITS); + } + } + }) } /// Takes a ring-buffer map out of the object and starts a task that drains it. @@ -1279,6 +1368,24 @@ mod tests { assert!(process_group_is_gone(u32::MAX)); } + /// The exit event arrives before the parent reaps, and the zombie is still + /// listed in /proc. It has no thread left, so it is gone. + #[test] + fn an_unreaped_zombie_is_gone() { + let mut child = std::process::Command::new("true").spawn().unwrap(); + let pid = child.id(); + let deadline = Instant::now() + std::time::Duration::from_secs(10); + while !std::fs::read_to_string(format!("/proc/{pid}/stat")) + .unwrap() + .contains(") Z ") + { + assert!(Instant::now() < deadline, "child never became a zombie"); + std::thread::sleep(std::time::Duration::from_millis(5)); + } + assert!(process_group_is_gone(pid)); + child.wait().unwrap(); + } + use std::net::Ipv4Addr; use std::os::unix::fs::PermissionsExt as _; diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 00cc99c..86d7747 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -428,10 +428,14 @@ relied upon. **Compatibility exit handling.** When `sched_process_exit` exposes `group_dead`, the kernel evicts only on confirmed process death. Without that field, it -preserves identity and deny entries on thread or leader exit. The daemon can -remove a candidate only after `/proc//task` is absent. A leader may exit -before its workers, so this conservative fallback can leave stale denials until -exec or reconciliation; it cannot guarantee immediate cleanup after group death. +preserves identity and deny entries on thread or leader exit and reports the +leader's exit as a candidate. The daemon evicts a candidate once +`/proc//task` is absent or lists only a zombie leader. The event usually +arrives before that, while the leader is still exiting or its workers still +run, so such a candidate is checked again every two seconds until its group is +gone, or dropped once its pid belongs to another process. At most 1024 +candidates wait; past that the oldest leaves its entries to exec or the next +reconciliation. **Loaded from a path, not embedded.** The kernel-side crate needs a dated nightly, `-Z build-std=core` and a matching `bpf-linker`, and is deliberately From b66e386fab680bb2593560cf6d0c7a6bb8953fb0 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:00:26 +0200 Subject: [PATCH 038/125] fix(ebpf): name the refused program only after checking its start time The in-kernel deny log looked the pid up with no start time, so an entry left by a dead process named whatever ran under that pid before. Pass the live process's start time so a recycled pid is not misnamed. --- crates/cfc-daemon/src/ebpf/loader.rs | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index 98162d7..928d24d 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -897,8 +897,15 @@ pub(super) fn load_and_attach( let Some(ev) = decode::(bytes) else { return; }; + // Verified against the live process's start time where there is + // one. Without it an entry left by a dead process names whatever + // ran under this pid before, and the admin chases the wrong program. let who = t - .get(ev.pid, None, Instant::now()) + .get( + ev.pid, + crate::process_resolve::read_starttime(ev.pid), + Instant::now(), + ) .map(|p| p.comm) .unwrap_or_else(|| "?".to_string()); tracing::info!( From 4fa0871663a0750d224690931a537ae1d4d10fc7 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:01:19 +0200 Subject: [PATCH 039/125] fix(nfqueue): credit a rule only when its answer releases the packet Releasing a parked prompt re-evaluated each packet, and evaluation counted a hit for any matching rule even when the user's own answer was the one applied. Hit counts drifted upward after every answered prompt. The release check now peeks without counting and credits the rule only when its verdict is delivered. --- crates/cfc-daemon/src/decision.rs | 23 +++++++++++++++++++++-- crates/cfc-daemon/src/nfqueue.rs | 18 ++++++++++++++++-- 2 files changed, 37 insertions(+), 4 deletions(-) diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index fc15b91..773fe0b 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -136,8 +136,28 @@ impl Engine { } /// Evaluate without blocking. Returns `Resolved` if a rule matches, - /// otherwise `NeedsPrompt`. + /// otherwise `NeedsPrompt`, and counts a hit for the matched rule. pub fn evaluate(&self, conn: &Connection, proc: &Process) -> Decision { + let decision = self.peek(conn, proc); + if let Decision::Resolved(Verdict { + source: cfc_core::VerdictSource::Rule(id), + .. + }) = decision + { + self.count_hit(id); + } + decision + } + + /// Credits one match to a rule. + pub fn count_hit(&self, rule_id: uuid::Uuid) { + *self.inner.hits.lock().entry(rule_id).or_insert(0) += 1; + } + + /// [`Self::evaluate`] without counting a hit, for a caller that may not + /// apply the answer: a parked packet released with the user's verdict + /// must not credit the rule it did not follow. + pub fn peek(&self, conn: &Connection, proc: &Process) -> Decision { let now_unix_ms = chrono::Utc::now().timestamp_millis(); let rule_match = { let rules = self.inner.rules.read(); @@ -157,7 +177,6 @@ impl Engine { } }; if let Some((rule_id, action)) = rule_match { - *self.inner.hits.lock().entry(rule_id).or_insert(0) += 1; // Verbatim: a Reject rule must reach the datapath as Reject so // the refusal is actually injected, not silently downgraded. return Decision::Resolved(Verdict::from_rule(action, rule_id)); diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index b7b2afa..3dba252 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -734,7 +734,7 @@ impl Worker { // Per packet, not per prompt: a Reject response is derived from // the individual segment (its sequence numbers, its source // port), and parallel connections share one prompt. - let verdict = match self.engine.evaluate(&packet.connection, &packet.process) { + let verdict = match self.engine.peek(&packet.connection, &packet.process) { // A refusal decided since the prompt opened always wins. Decision::Resolved(current) if current.action != Action::Allow => current, // So does a rule's Allow over a fallback nobody chose: a rule @@ -752,6 +752,10 @@ impl Worker { } _ => pv.verdict, }; + // Only a rule whose answer was applied gets the hit. + if let VerdictSource::Rule(id) = verdict.source { + self.engine.count_hit(id); + } self.deliver( packet.message, ObservedConnection { @@ -2658,7 +2662,9 @@ mod tests { .handle_message(FakeMsg::new(1, tcp_packet(443))) .unwrap(); let prompt = h.prompt_rx.try_recv().unwrap(); - h.worker().engine.upsert_rule(allow_port_rule(443)); + let rule = allow_port_rule(443); + let id = rule.id; + h.worker().engine.upsert_rule(rule); h.worker() .resolve_prompt(PromptVerdict { prompt_id: prompt.prompt_id, @@ -2666,6 +2672,14 @@ mod tests { }) .unwrap(); assert_eq!(h.verdicts(), vec![(1, expected)], "{answer:?}"); + // The rule is credited only when its answer was the one applied. + let hits = h.worker().engine.snapshot().rules[0].hit_count; + assert_eq!( + hits, + u64::from(expected == NfqVerdict::Accept), + "{answer:?}" + ); + assert_eq!(h.worker().engine.snapshot().rules[0].id, id); } } From 957696106d4de255ae2887208d67100b2c800062 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:02:14 +0200 Subject: [PATCH 040/125] docs(daemon): describe observed DNS answers and the resolver bound as built observe_answer still called observed answers first-hand evidence that outranks any PTR result, though they only reach the display cache, and the runtime comment still said dns.rs could queue thousands of lookups, though eight permits bound it. --- crates/cfc-daemon/src/dns.rs | 8 +++++--- crates/cfc-daemon/src/main.rs | 9 ++++----- 2 files changed, 9 insertions(+), 8 deletions(-) diff --git a/crates/cfc-daemon/src/dns.rs b/crates/cfc-daemon/src/dns.rs index 2556475..063e316 100644 --- a/crates/cfc-daemon/src/dns.rs +++ b/crates/cfc-daemon/src/dns.rs @@ -148,9 +148,11 @@ impl DnsCache { /// Records an `A`/`AAAA` record lifted out of a DNS response this host /// received, with `ttl` in seconds as the record carried it. /// - /// This is first-hand evidence (see the module docs) and outranks any PTR - /// result for the same address, present or future. Called from the - /// `DNS_PACKETS` ring-buffer consumer, never from the packet path. + /// Display only, never policy identity: the record is not tied to a + /// resolver transaction (see the module docs). It fills the diagnostic + /// cache alone; the policy cache keeps only forward-confirmed PTR names. + /// Called from the `DNS_PACKETS` ring-buffer consumer, never from the + /// packet path. pub fn observe_answer(&self, ip: IpAddr, name: &str, ttl: u32) { self.observe_answer_at(ip, name, ttl, Instant::now()); } diff --git a/crates/cfc-daemon/src/main.rs b/crates/cfc-daemon/src/main.rs index 1c1fd32..bb0a686 100644 --- a/crates/cfc-daemon/src/main.rs +++ b/crates/cfc-daemon/src/main.rs @@ -103,11 +103,10 @@ fn main() -> anyhow::Result<()> { // would be a much worse bug than the one this fixes. .worker_threads(4) // And a ceiling on the blocking pool, which tokio leaves at 512. - // `dns.rs` hands it one `getaddrinfo` per new destination address, and - // a stalled resolver blocks each for the resolv.conf default of two - // five-second attempts. At the worker's flow rate that queues - // thousands of lookups and spawns threads to match: 512 x 2 MiB of - // stack reservation, plus an arena apiece. Sixteen is plenty - the + // `dns.rs` runs its blocking resolver calls here, at most eight at + // once (`LOOKUP_MAX_IN_FLIGHT`), and rule validation and provenance + // use the same pool. Nothing should ever need 512 threads at 2 MiB of + // stack reservation plus an arena apiece. Sixteen is plenty - the // packet worker permanently occupies one of them - and excess calls // then queue instead of spawning. .max_blocking_threads(16) From a71257c65dd621266b0e23ac12c8219ef45fdad4 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:02:14 +0200 Subject: [PATCH 041/125] fix(storage): name every enabled legacy hostname rule at startup A legacy dst_host rule refuses the flows its other predicates match, and those refusals are recorded as the default policy, so nothing pointed an admin at the rule. Warn once per such rule at load, and say in HARDENING.md that it cannot be disabled, only edited or deleted. The unscoped-rule hint no longer suggests dst_host, which is refused. --- crates/cfc-daemon/src/convert.rs | 2 +- crates/cfc-daemon/src/storage.rs | 14 ++++++++++++++ docs/HARDENING.md | 4 ++++ 3 files changed, 19 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/convert.rs b/crates/cfc-daemon/src/convert.rs index 49e3cd3..a5db97b 100644 --- a/crates/cfc-daemon/src/convert.rs +++ b/crates/cfc-daemon/src/convert.rs @@ -340,7 +340,7 @@ pub fn reject_unscoped(scope: &RuleScope) -> Result<(), String> { return Err( "rule scope constrains nothing, so it would match every process and \ every destination; scope it to at least one of exe_path, uid, \ - dst_host, dst_net, dst_port or protocol" + dst_net, dst_port or protocol" .to_string(), ); } diff --git a/crates/cfc-daemon/src/storage.rs b/crates/cfc-daemon/src/storage.rs index 8fdec52..8a9637d 100644 --- a/crates/cfc-daemon/src/storage.rs +++ b/crates/cfc-daemon/src/storage.rs @@ -256,6 +256,20 @@ impl RuleStore { quarantined += 1; continue; } + if rule.enabled && rule.scope.dst_host.is_some() { + // Its refusals are recorded as the default policy, + // not as this rule, so without this line nothing + // points at the rule behind them. + tracing::warn!( + rule_id = %id, + rule_name = %rule.name, + "legacy hostname rule: flows its other predicates \ + match are refused and logged as the default \ + policy; it cannot be disabled, so replace it with \ + an executable or numeric scope, or delete it with \ + `cfc rules remove {id}`" + ); + } rules.push(rule); } Err(e) => { diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 8826e6b..7a2a668 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -110,6 +110,10 @@ predicates are compatible, their uncertainty refuses the flow before a lower Allow, pause or prompt can admit it. Replace these rules explicitly with executable or numeric scopes; a legacy hostname Allow no longer grants access. The editor requires the old hostname to be removed before saving a replacement. +Such a rule cannot be disabled either, since a toggle sends the hostname back +and the daemon refuses it: edit or delete it (`cfc rules remove `). Its +refusals are logged as the default policy, so the daemon names every enabled +legacy hostname rule in a warning at startup. CLI and GUI destination presets use the observed numeric endpoint, as `/32` for IPv4 or `/128` for IPv6, and label it as an IP. They do not turn a domain From aaa56a844df3c645685603d4625b54f33a648c22 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:03:05 +0200 Subject: [PATCH 042/125] docs(ebpf): describe the legacy Fast Allow cleanup as it runs The changelog and ARCHITECTURE.md said startup always disarms the legacy pinned maps and removes the old sendmsg pins; that only happens when the eBPF layer loads. Comments said the nft unit is ordered after the daemon, though it loads before it, and that a dead armed daemon left a mark the hooks keep setting, though past its deadline they only strip it. --- CHANGELOG.md | 5 +++-- crates/cfc-daemon/src/ebpf.rs | 4 ++-- crates/cfc-daemon/src/ebpf/enforce.rs | 8 +++++--- crates/cfc-daemon/src/ebpf/loader.rs | 7 +++++-- crates/cfc-daemon/src/ebpf/nft_set.rs | 5 +++-- docs/ARCHITECTURE.md | 9 ++++++--- 6 files changed, 24 insertions(+), 14 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a3ebd32..28ba63e 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -13,8 +13,9 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). no longer has a `fast_allow` key, `StatusResponse` field 16 is reserved, and the `[ebpf] fast_allow` and `fast_allow_mark` keys are ignored with a warning. For hosts upgrading from 0.4-0.6, startup still flushes the legacy - nftables set, disarms the legacy pinned maps and removes the old sendmsg - link pins. + nftables set. When the eBPF layer loads, it also disarms the legacy pinned + maps and removes the old sendmsg link pins; with the layer off, without the + object or after a failed load, those stay until reboot. ### Fixed diff --git a/crates/cfc-daemon/src/ebpf.rs b/crates/cfc-daemon/src/ebpf.rs index e7a50e9..a0041fc 100644 --- a/crates/cfc-daemon/src/ebpf.rs +++ b/crates/cfc-daemon/src/ebpf.rs @@ -597,8 +597,8 @@ pub fn nft_table_loaded() -> anyhow::Result { /// crashed while armed can have left its mark in a set that a ruleset not yet /// reloaded still accepts. Called from `main` in every build, whatever the /// layer's mode, and never under `--dry-run`, which touches nothing. A table -/// or set that is not loaded is nothing to flush: at boot the nft unit is -/// ordered after the daemon. +/// or set that is not loaded is nothing to flush. At boot the nft unit is +/// ordered before the daemon, so the flush normally finds the table loaded. pub fn flush_legacy_fast_allow_set() { if let Err(e) = nft_set::flush() { tracing::error!("could not flush the legacy fast_allow nftables set: {e:#}; a mark left by an older daemon may still bypass filtering; run systemctl reload colony-firewall-nft and inspect the journal before relying on filtering"); diff --git a/crates/cfc-daemon/src/ebpf/enforce.rs b/crates/cfc-daemon/src/ebpf/enforce.rs index e21018c..7e499c3 100644 --- a/crates/cfc-daemon/src/ebpf/enforce.rs +++ b/crates/cfc-daemon/src/ebpf/enforce.rs @@ -936,9 +936,11 @@ fn remove_legacy_pins(dir: &Path) { /// /// The daemon no longer grants, but the kernel object still carries the maps /// (ABI v4) and the connect programs still consult them. They are pinned, so -/// a 0.4-0.6 daemon that died while armed left a mark the hooks would go on -/// setting across any number of restarts - a mark that can collide with the -/// fwmark selectors of kube-proxy, Tailscale or wg-quick. With `UNARMED` +/// a 0.4-0.6 daemon that died while armed left its mark armed across any +/// number of restarts. The hooks stop setting it once the deadline passes, but +/// they go on stripping that value from every socket carrying it, and a random +/// mark can collide with the fwmark selectors of kube-proxy, Tailscale or +/// wg-quick. With `UNARMED` /// written, `mark_decision` returns at its first array read. Best effort: a /// map that is missing or will not take the write is logged and skipped. pub(super) fn disarm_legacy_fast_allow(bpf: &mut Ebpf) { diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index 928d24d..5b952cb 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -593,8 +593,11 @@ pub(super) fn load_and_attach( // Fast Allow is gone from the daemon, but the kernel object still carries // its maps (ABI v4) and they are pinned, so a 0.4-0.6 daemon that died - // while armed left a mark the connect hooks would go on setting, past any - // restart. Disarm before anything attaches, whatever else comes up. + // while armed left its mark armed. Past the deadline the connect hooks no + // longer set it, but they go on stripping that value from any socket that + // carries it, past any restart. Disarm before anything attaches, whatever + // else comes up. Only on this path: a daemon that does not load the object + // leaves the old pins alone until reboot. enforce::disarm_legacy_fast_allow(&mut bpf); // --- attach, each independently ------------------------------------ diff --git a/crates/cfc-daemon/src/ebpf/nft_set.rs b/crates/cfc-daemon/src/ebpf/nft_set.rs index 98ca840..80c5c37 100644 --- a/crates/cfc-daemon/src/ebpf/nft_set.rs +++ b/crates/cfc-daemon/src/ebpf/nft_set.rs @@ -79,8 +79,9 @@ pub(super) fn table_loaded() -> anyhow::Result { /// there is accepted by the ruleset. /// /// A missing table or a missing set is success: there is nothing in either -/// that could accept a mark. At boot the table is normally not loaded yet, -/// because `colony-firewall-nft.service` is ordered after the daemon. +/// that could accept a mark. At boot `colony-firewall-nft.service` is ordered +/// before the daemon, so the table is normally loaded and this empties the set +/// it still declares. pub(super) fn flush() -> anyhow::Result<()> { match run(Op::FlushSet) { Ok(()) => { diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 86d7747..4ad8f5b 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -419,9 +419,12 @@ prove the current sender's identity, and lifecycle checks do not repair that property, so it opened bypasses; it was disabled in 0.7.0 and its userspace side is gone. The `[ebpf] fast_allow` keys still parse and only log a warning. The kernel object still carries the Fast Allow maps until an ABI bump, so -startup flushes the legacy nft set once, disarms the pinned maps (unarmed mark, -zero deadline, no grants) and removes the old `sendmsg4`/`sendmsg6` link pins, -which detaches those hooks. The nft snippet has no mark-set accept rule, and +startup flushes the legacy nft set once and, when the eBPF layer loads, +disarms the pinned maps (unarmed mark, zero deadline, no grants) and removes +the old `sendmsg4`/`sendmsg6` link pins, which detaches those hooks. With the +layer off, without the object or after a failed load, the old pins stay until +reboot; they can strip a socket mark equal to the old random value, but the +snippet accepts no packet on an application-set mark, so they open nothing. The nft snippet has no mark-set accept rule, and package upgrades reload active nft units with one atomic transaction. A failed flush emits an error and requires operator action before filtering can be relied upon. From ff63357392e50a9d03dab195481ba44c54dad2a5 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:03:39 +0200 Subject: [PATCH 043/125] docs(changelog): record the loopback and attribution changes for 0.8.0 The Unreleased section, which becomes the release notes, did not say that loopback flows now go through the queue and obey explicit rules, that they bypass it while no daemon listens, or that socket attribution became stricter. --- CHANGELOG.md | 21 +++++++++++++++++++++ 1 file changed, 21 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 28ba63e..9119690 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,27 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ## [Unreleased] +### Changed + +- New loopback flows now go through the queue instead of being accepted + outright (`oifname "lo" accept` is gone from the outbound table). While the + daemon runs, explicit rules apply to them, so a loopback Deny that 0.7.0 + never enforced now takes effect, and unmatched local IPC is still allowed + without a prompt. Each new loopback connection pays the queue round trip. +- Socket attribution is stricter. TCP needs an exact connected tuple and never + selects a listening socket, so an outbound flow no longer inherits a + listener's rules. UDP accepts zero-remote and wildcard-local sockets only + when every compatible socket agrees on its owner. The process and + descriptor found must still hold the socket after the executable is read; + otherwise the identity is unknown. + +### Security + +- While no daemon listens on the queue, new loopback flows are allowed + (`oifname "lo" ct state new queue num 0 bypass`), so local services keep + working when the daemon is down. Loopback Deny rules are not enforced then. + Every other new flow stays fail-closed. + ### Removed - The Fast Allow userspace path, disabled since 0.7.0 because a socket mark From c97c3b21dafd60a67872fc63b7ca044bf1bad60d Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:04:33 +0200 Subject: [PATCH 044/125] chore(deps): take the event-listener and memmap2 soundness fixes event-listener 5.4.1 (RUSTSEC-2026-0221) and memmap2 0.9.10 (RUSTSEC-2026-0186) carry unsound advisories fixed in semver-compatible patch releases. Both reach only the GUI and tray. lru 0.16 via cryoglyph needs an iced upgrade and stays. --- Cargo.lock | 9 ++++----- 1 file changed, 4 insertions(+), 5 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 44ed4d8..722ceb2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1412,11 +1412,10 @@ dependencies = [ [[package]] name = "event-listener" -version = "5.4.1" +version = "5.4.2" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "e13b66accf52311f30a0db42147dadea9850cb48cd070028831ae5f5d4b856ab" +checksum = "5a23add41df1562121a9393cb065eab5146a1242410f23a644851e90cfd669d2" dependencies = [ - "concurrent-queue", "parking", "pin-project-lite", ] @@ -2714,9 +2713,9 @@ checksum = "f8ca58f447f06ed17d5fc4043ce1b10dd205e060fb3ce5b979b8ed8e59ff3f79" [[package]] name = "memmap2" -version = "0.9.10" +version = "0.9.11" source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "714098028fe011992e1c3962653c96b2d578c4b4bce9036e15ff220319b1e0e3" +checksum = "d1219ed1b7f229ee7104d281dd01d6802fe28bb6e95d292942c4daacdeb798c0" dependencies = [ "libc", ] From b74c96b7390438ca202aacd2e0f3b907def3bc43 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:04:54 +0200 Subject: [PATCH 045/125] docs(changelog): list the daemon fixes in this bundle Refusal logging for rule writes (#46), hit count drift, alias rules that could not be disabled, future creation dates, legacy hostname warnings, eBPF exit eviction and resync ordering, and the socket directory mode. --- CHANGELOG.md | 25 +++++++++++++++++++++++++ 1 file changed, 25 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9119690..9cc99af 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -96,6 +96,31 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). transaction. The packet thread now skips a busy or stale index, and that "not ready" answer is no longer cached for an hour as "not from a package"; it shows as unknown until the index is ready. +- A refused `UpsertRule` or `ApplyRules` left nothing in the journal unless + authorization refused it, so "never sent" and "sent and refused" looked the + same (#46). Every refusal now logs the RPC, the caller's uid and pid, the + status code and its message. Refusal messages no longer echo a client + value of unbounded length. +- Rule hit counts drifted upward: a rule write between the 30 s flush's drain + and merge stored the drained hits twice, and releasing a prompt credited a + matching rule even when the user's answer was the one applied. +- A rule whose stored executable path later became an alias (a legacy + `/bin/curl`, or a target a package turned into a symlink) could not be + disabled, renamed or re-imported, only deleted. A path sent back unchanged + is accepted; new and changed paths are still checked. +- A new timed rule took its creation date from the client, so a date in the + future kept "allow for 90s" alive indefinitely. Dates are clamped to now. +- Enabled legacy hostname rules refuse flows that are logged as the default + policy. The daemon now names each one in a warning at startup. +- eBPF: exit events usually arrived before the parent reaped the process and + were dropped, so on kernels without `group_dead` an in-kernel deny outlived + its process until a rule change or restart, where a recycled pid could + inherit it. Candidates are now checked again until their group is gone. + Overlapping verdict resyncs could also leave the older rule set's answers + in the kernel, and a rule changed during the startup resync never reached + it. +- A daemon started by hand under umask 000 created its socket directory + world-writable, so a local user could replace the socket. ## [0.7.0] - 2026-09-30 From 248d91ccdc6d942db34e415ee821d85d76f41e01 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:10:57 +0200 Subject: [PATCH 046/125] fix(ui): answer the visible top prompt and arm verdicts after a change A and D answered the newest prompt, which is the bottom card and often below the fold, from any tab and with Ctrl, Alt or Super held. Shift+A on the card the user was reading could write an always-allow rule for a program further down. The keys now answer the top card, which is marked, only on the Prompts tab and only without Ctrl, Alt or Super. The target is disarmed for one second whenever it changes, and a card's verdict buttons stay disabled for one second after it appears, so a key press or click already on its way when the window was raised or a card moved does not answer it. A prompt arriving while others are pending no longer switches the tab. --- crates/cfc-ui/src/main.rs | 186 ++++++++++++++++++++++++----- crates/cfc-ui/src/views/prompts.rs | 56 ++++++--- 2 files changed, 199 insertions(+), 43 deletions(-) diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 567da4b..ff542d0 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -39,6 +39,13 @@ const DELETE_CONFIRM_MS: i64 = 3_000; /// a smooth bar, slow enough to stay off the CPU when nothing is pending. const DEADLINE_TICK_MS: u64 = 400; +/// How long a prompt's verdict controls stay disabled after the card +/// appears, and how long the keyboard target stays disarmed after it +/// changes. A click or key press already on its way when the layout changed +/// (the window was raised, a card arrived, the card above expired) must not +/// land on a verdict the user never saw. +const PROMPT_ARM_MS: i64 = 1_000; + fn main() -> iced::Result { tracing_subscriber::fmt() .with_env_filter( @@ -61,6 +68,9 @@ pub struct App { pub rules: Vec, pub live: VecDeque, pub prompts: Vec, + /// The card the A/D keys answer and when they may: `(prompt_id, + /// armed_at_ms)`. See [`App::sync_key_target`]. + pub key_target: Option<(String, i64)>, pub status: Option, pub log: StatusLog, pub editor: Option, @@ -333,24 +343,29 @@ pub struct PromptCard { /// Wall clock at which the daemon answers this prompt itself. 0 means /// the daemon attached no deadline. pub deadline_unix_ms: i64, + /// Wall clock before which the verdict buttons stay disabled. + pub armed_at_ms: i64, } impl PromptCard { - fn new(event: proto::PromptEvent) -> Self { + fn new(event: proto::PromptEvent, now_ms: i64) -> Self { Self { deadline_unix_ms: event.deadline_unix_ms, event, + armed_at_ms: now_ms.saturating_add(PROMPT_ARM_MS), } } + + pub fn armed(&self, now_ms: i64) -> bool { + now_ms >= self.armed_at_ms + } } /// How loudly a newly arrived prompt announces itself. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Attention { - /// Leave the window alone entirely. + /// Leave the window and the tab alone; the sidebar badge counts it. None, - /// Show the Prompts tab, but do not touch the window. - Tab, /// Show the Prompts tab and pull the window in front of whatever the /// user is looking at. TabAndRaise, @@ -359,20 +374,22 @@ pub enum Attention { /// Decides what the arrival of a prompt does to the window. /// /// `pending_before` is the queue length *before* this prompt was pushed, so -/// only the 0 -> 1 transition raises. A burst of ten prompts must not slam -/// the window into the user's face ten times; a queue that drains and then -/// refills is a genuinely new interruption and may raise again. +/// only the 0 -> 1 transition switches the tab and raises. A burst of ten +/// prompts must not slam the window into the user's face ten times, and a +/// user who left the Prompts tab with prompts pending already knows about +/// them: switching the tab under their cursor again would turn the click +/// they were making on a rule row into a click on a verdict. A queue that +/// drains and then refills is a genuinely new interruption and may raise +/// again. /// /// An open rule editor suppresses all of it. Focus-stealing is disruptive /// at the best of times, and yanking the tab out from under someone /// half-way through a form loses what they had typed. pub fn prompt_attention(pending_before: usize, editor_open: bool) -> Attention { - if editor_open { + if editor_open || pending_before > 0 { Attention::None - } else if pending_before == 0 { - Attention::TabAndRaise } else { - Attention::Tab + Attention::TabAndRaise } } @@ -519,6 +536,7 @@ impl App { rules: Vec::new(), live: VecDeque::with_capacity(LIVE_CAP), prompts: Vec::new(), + key_target: None, status: None, log: StatusLog::default(), editor: None, @@ -625,6 +643,21 @@ impl App { now, ); } + self.sync_key_target(); + } + + /// Points the keyboard at the top card and disarms it for + /// [`PROMPT_ARM_MS`] whenever that card changes. + /// + /// The top card is the oldest one, first on screen and first to expire, + /// and a new arrival never displaces it. Re-arming on every change means + /// a key press already on its way when the card above expired or was + /// answered does not land on the card that just moved up. + fn sync_key_target(&mut self) { + let top = self.prompts.first().map(|card| &card.event.prompt_id); + if self.key_target.as_ref().map(|(id, _)| id) != top { + self.key_target = top.map(|id| (id.clone(), self.now_ms.saturating_add(PROMPT_ARM_MS))); + } } fn update(&mut self, message: Message) -> Task { @@ -765,17 +798,17 @@ impl App { // notifications now - two bubbles per prompt would train // users to ignore them. The GUI shows its card either way. self.stream_trouble = false; + // The arming deadline is measured from now, not from the + // last tick, which can be two seconds old. + self.now_ms = now_ms(); // A prompt is a held-open connection with a deadline on it; // a card the user only sees if they happen to be looking at // the right tab is not an ask, it is a countdown they lose. let attention = prompt_attention(self.prompts.len(), self.editor.is_some()); - self.prompts.push(PromptCard::new(ev)); + self.prompts.push(PromptCard::new(ev, self.now_ms)); + self.sync_key_target(); match attention { Attention::None => Task::none(), - Attention::Tab => { - self.tab = Tab::Prompts; - Task::none() - } Attention::TabAndRaise => { self.tab = Tab::Prompts; raise_window() @@ -918,7 +951,10 @@ impl App { self.log.clear(); Task::none() } - Message::Key(kp) => self.handle_key(kp), + Message::Key(kp) => { + self.now_ms = now_ms(); + self.handle_key(kp) + } Message::TogglePaused => { let current = self.status.as_ref().map(|s| s.paused).unwrap_or(false); let socket = self.socket_path.clone(); @@ -1077,14 +1113,23 @@ impl App { Task::none() } keyboard::Key::Character(ref c) => { + // Only bare keys and Shift are ours. Ctrl+A is "select all" + // out of habit, and Alt and Super chords belong to the + // desktop; none of them may answer a prompt. + if kp.modifiers.control() || kp.modifiers.alt() || kp.modifiers.logo() { + return Task::none(); + } let shift = kp.modifiers.shift(); + let on_prompts = self.tab == Tab::Prompts; match c.to_lowercase().as_str() { - "a" => self.answer_newest(if shift { + // The verdict keys only work where the card they answer + // is on screen. + "a" if on_prompts => self.answer_key_target(if shift { PromptAction::AllowProgram } else { PromptAction::AllowOnce }), - "d" => self.answer_newest(if shift { + "d" if on_prompts => self.answer_key_target(if shift { PromptAction::BlockProgram } else { PromptAction::BlockOnce @@ -1100,16 +1145,29 @@ impl App { } } - /// Answers the most recent prompt with the same verdict the matching - /// button would submit. + /// Answers the top card, the one marked as the keyboard target, with + /// the same verdict the matching button would submit. Does nothing while + /// the target is disarmed (see [`App::sync_key_target`]). + /// + /// It used to answer the newest prompt, which is the bottom card and + /// often below the fold: the user read the card at the top, pressed + /// Shift+A, and wrote an always-allow rule for a program they never saw. /// /// A program-scoped choice on a flow with no executable path has no /// honest verdict (see `verdict_for`), so it degrades to the one-off /// answer of the same action rather than doing nothing: the user /// pressed Shift+D to stop a connection, and stopping it is the part /// that matters. - fn answer_newest(&mut self, choice: PromptAction) -> Task { - let Some(card) = self.prompts.last() else { + fn answer_key_target(&mut self, choice: PromptAction) -> Task { + self.sync_key_target(); + if !self + .key_target + .as_ref() + .is_some_and(|(_, armed_at)| self.now_ms >= *armed_at) + { + return Task::none(); + } + let Some(card) = self.prompts.first() else { return Task::none(); }; let ev = &card.event; @@ -1256,6 +1314,7 @@ impl App { let hints = column![ text("1-4 switch tab").size(9), + text("Keys answer the top prompt:").size(9), text("A allow once").size(9), text("D block for now").size(9), text("Shift+A always allow program").size(9), @@ -1946,7 +2005,7 @@ mod tests { fn customization_waits_for_a_verdict_and_drops_stale_actions() { let (mut app, _) = App::new(); let event = prompt_event(); - app.prompts.push(PromptCard::new(event.clone())); + app.prompts.push(PromptCard::new(event.clone(), 0)); let task = app.update(Message::CustomizePromptRule(event.prompt_id.clone())); assert_eq!(task.units(), 0, "opening the editor issues no verdict RPC"); assert_eq!(app.prompts.len(), 1, "the prompt remains pending"); @@ -1969,7 +2028,7 @@ mod tests { fn an_expired_prompt_closes_its_customization() { let (mut app, _) = App::new(); let event = prompt_event(); - app.prompts.push(PromptCard::new(event.clone())); + app.prompts.push(PromptCard::new(event.clone(), 0)); app.editor = Some(RuleEditor::from_prompt(&event)); app.now_ms = event.deadline_unix_ms + 1; app.housekeeping(); @@ -2036,9 +2095,10 @@ mod tests { fn a_first_prompt_raises_the_window_and_a_burst_does_not() { // 0 -> 1 is the interruption worth stealing focus for. assert_eq!(prompt_attention(0, false), Attention::TabAndRaise); - // 1 -> 2, 2 -> 3: the user is already looking at the queue. - assert_eq!(prompt_attention(1, false), Attention::Tab); - assert_eq!(prompt_attention(9, false), Attention::Tab); + // 1 -> 2, 2 -> 3: the user already knows about the queue, and a tab + // switch under their cursor would turn a click into a verdict. + assert_eq!(prompt_attention(1, false), Attention::None); + assert_eq!(prompt_attention(9, false), Attention::None); // Drained and refilled: a genuinely new interruption. assert_eq!(prompt_attention(0, false), Attention::TabAndRaise); } @@ -2050,6 +2110,74 @@ mod tests { assert_eq!(prompt_attention(3, true), Attention::None); } + fn key(c: &str, modifiers: keyboard::Modifiers) -> KeyPress { + KeyPress { + key: keyboard::Key::Character(c.into()), + modifiers, + } + } + + fn prompt(id: &str, exe: &str) -> proto::PromptEvent { + let mut ev = prompt_event(); + ev.prompt_id = id.into(); + ev.process.as_mut().unwrap().exe = exe.into(); + ev.deadline_unix_ms = 0; + ev + } + + #[test] + fn verdict_keys_answer_the_armed_top_card_only() { + let (mut app, _) = App::new(); + app.now_ms = 1_000_000; + app.prompts.push(PromptCard::new( + prompt("top", "/usr/lib/firefox/firefox"), + 0, + )); + app.prompts.push(PromptCard::new( + prompt("below", "/home/u/.cache/x/updater"), + 0, + )); + let shift = keyboard::Modifiers::SHIFT; + + // The top card just became the target: a press already on its way + // answers nothing. + assert_eq!(app.handle_key(key("A", shift)).units(), 0); + assert_eq!(app.prompts.len(), 2); + + app.now_ms += PROMPT_ARM_MS; + // Chords and other tabs never answer. + for m in [ + keyboard::Modifiers::CTRL, + keyboard::Modifiers::ALT, + keyboard::Modifiers::LOGO, + keyboard::Modifiers::CTRL | shift, + ] { + assert_eq!(app.handle_key(key("a", m)).units(), 0, "{m:?}"); + } + app.tab = Tab::Rules; + assert_eq!(app.handle_key(key("a", shift)).units(), 0); + app.tab = Tab::Prompts; + + assert_eq!(app.handle_key(key("A", shift)).units(), 1); + let ids: Vec<_> = app.prompts.iter().map(|c| &c.event.prompt_id).collect(); + assert_eq!(ids, ["below"], "the top card is answered, not the newest"); + + // The card that moved up is disarmed again. + assert_eq!( + app.handle_key(key("a", keyboard::Modifiers::empty())) + .units(), + 0 + ); + } + + #[test] + fn a_prompt_card_is_disarmed_right_after_it_appears() { + let card = PromptCard::new(prompt("p", "/usr/bin/curl"), 5_000); + assert!(!card.armed(5_000)); + assert!(!card.armed(5_000 + PROMPT_ARM_MS - 1)); + assert!(card.armed(5_000 + PROMPT_ARM_MS)); + } + #[test] fn prompt_card_captures_the_daemon_deadline() { let ev = proto::PromptEvent { @@ -2057,6 +2185,6 @@ mod tests { deadline_unix_ms: 1_700_000_000_000, ..Default::default() }; - assert_eq!(PromptCard::new(ev).deadline_unix_ms, 1_700_000_000_000); + assert_eq!(PromptCard::new(ev, 0).deadline_unix_ms, 1_700_000_000_000); } } diff --git a/crates/cfc-ui/src/views/prompts.rs b/crates/cfc-ui/src/views/prompts.rs index 0c3b721..048587a 100644 --- a/crates/cfc-ui/src/views/prompts.rs +++ b/crates/cfc-ui/src/views/prompts.rs @@ -39,7 +39,7 @@ pub fn view<'a>( column![ text("No pending prompts").size(18), text("Outbound flows without a matching rule will appear here for you to allow or block.").size(12), - text("Keyboard: A allow once / D block for now; Shift+A always allow this program, Shift+D always block it.").size(11), + text("Keyboard, on the top prompt: A allow once / D block for now; Shift+A always allow this program, Shift+D always block it.").size(11), ] .spacing(8), ) @@ -54,9 +54,12 @@ pub fn view<'a>( .unwrap_or(proto::Action::Unspecified as i32); let timeout_secs = status.map(|s| s.prompt_timeout_secs).unwrap_or(0); + // The first card is the one the A/D keys answer (see + // `App::answer_key_target`), so it is the one that carries the marker. let cards: Vec> = prompts .iter() - .map(|c| prompt_card(c, timeout_action, timeout_secs, now_ms)) + .enumerate() + .map(|(i, c)| prompt_card(c, timeout_action, timeout_secs, now_ms, i == 0)) .collect(); container(scrollable(column(cards).spacing(12).padding(8)).height(Length::Fill)) @@ -359,11 +362,28 @@ fn prompt_card( timeout_action: i32, timeout_secs: u32, now_ms: i64, + key_target: bool, ) -> Element<'_, Message> { let ev = &card.event; let program = program_label(ev); + // Disabled for a moment after the card appears, so a click aimed at + // whatever was here before cannot land on a verdict. + let armed = card.armed(now_ms); - let header = text(heading(ev)).size(16); + let marker: Element<'_, Message> = if key_target { + text("A / D answer this prompt") + .size(10) + .color(crate::theme::PARCHMENT_MUTED) + .into() + } else { + Space::new().into() + }; + let header = row![ + text(heading(ev)).size(16), + Space::new().width(Length::Fill), + marker + ] + .align_y(iced::Alignment::Center); let countdown = countdown_row(card, timeout_action, timeout_secs, now_ms); let table = column( @@ -385,28 +405,32 @@ fn prompt_card( ev, &program, iced::widget::button::success, - true + true, + armed ), action_row( PromptAction::BlockProgram, ev, &program, iced::widget::button::danger, - true + true, + armed ), action_row( PromptAction::BlockOnce, ev, &program, iced::widget::button::secondary, - true + true, + armed ), action_row( PromptAction::AllowOnce, ev, &program, crate::theme::action_secondary, - false + false, + armed ), ] .spacing(6); @@ -470,20 +494,24 @@ fn detail_row<'a>(r: DetailRow) -> Element<'a, Message> { /// A full-width choice: what it does on line one, what it persists on line /// two. `prominent` is the difference between one of the three decisions -/// and the subordinate "Allow once" beneath them. +/// and the subordinate "Allow once" beneath them. Rendered disabled until +/// `armed`. fn action_row<'a>( choice: PromptAction, ev: &proto::PromptEvent, program: &str, style: ButtonStyle, prominent: bool, + armed: bool, ) -> Element<'a, Message> { - let press = verdict_for(choice, ev).map(|v| Message::SubmitVerdict { - prompt_id: ev.prompt_id.clone(), - action: v.action, - scope: v.scope, - duration: v.duration, - }); + let press = verdict_for(choice, ev) + .filter(|_| armed) + .map(|v| Message::SubmitVerdict { + prompt_id: ev.prompt_id.clone(), + action: v.action, + scope: v.scope, + duration: v.duration, + }); let (title_size, sub_size) = if prominent { (14, 11) } else { (12, 10) }; let padding: iced::Padding = if prominent { From 74d2b6b1942c1ba0ada400c2a0366a40bc32888b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:12:18 +0200 Subject: [PATCH 047/125] fix(ui): seed "make rule" with the program, port and protocol (#46) "make rule" pinned the destination to the one address seen, so the next connection of the same service to another address stayed denied while the CLI rule (--exe --dst-port --protocol) worked. With an executable that can scope a rule the editor now seeds that scope. The '' placeholder is never seeded: such rows pin the numeric endpoint instead, and an inbound row keeps its direction. A rule the editor refuses to build is now also reported in the footer, not only in a banner that can sit below the fold, and a saved rule logs the scope that was stored. --- crates/cfc-ui/src/main.rs | 208 +++++++++++++++++++++++++++----- crates/cfc-ui/src/views/live.rs | 2 +- 2 files changed, 181 insertions(+), 29 deletions(-) diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index ff542d0..b257915 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -282,16 +282,11 @@ impl RuleEditor { .as_ref() .map(|p| p.exe.as_str()) .unwrap_or_default(); - let (dst_host, dst_ip, dst_port, protocol) = match ev.connection.as_ref() { - Some(c) => ( - c.dst_host.as_str(), - c.dst_ip.as_str(), - c.dst_port, - c.protocol, - ), - None => ("", "", 0, 0), + let (dst_ip, dst_port, protocol, direction) = match ev.connection.as_ref() { + Some(c) => (c.dst_ip.as_str(), c.dst_port, c.protocol, c.direction), + None => ("", 0, 0, 0), }; - let mut editor = Self::from_observed(exe, dst_host, dst_ip, dst_port, protocol); + let mut editor = Self::from_observed(exe, dst_ip, dst_port, protocol, direction); editor.prompt_id = Some(ev.prompt_id.clone()); editor.prompt_hash_required = ev.binds_to_hash; if ev.binds_to_hash { @@ -304,29 +299,58 @@ impl RuleEditor { editor } - /// Seeds the editor from an observed flow (live feed "make rule"). - /// Pins the numeric endpoint IP; DNS names remain diagnostic. + /// Seeds the editor from an observed flow (live feed "make rule", and + /// a prompt's "Customize"). + /// + /// When the executable can scope a rule, the seed is that program on + /// that port and protocol, the same scope `cfc rules add --exe + /// --dst-port --protocol` builds. Pinning the one address seen made a + /// rule the next connection of the same service, to another CDN address + /// or over IPv6, did not match, so the app stayed denied after "make + /// rule" while the CLI rule worked (#46). The user can still type an + /// address into the destination field. + /// + /// Otherwise the placeholder for an unidentified process is never + /// seeded, since a rule on it is refused, and the numeric endpoint is + /// pinned instead so the rule does not cover the whole port. DNS names + /// stay diagnostic. An inbound flow keeps its direction: without it the + /// rule would be an outbound one to our own address. pub fn from_observed( exe: &str, - _dst_host: &str, dst_ip: &str, dst_port: u32, protocol: i32, + direction: i32, ) -> Self { let protocol = proto::Protocol::try_from(protocol) .ok() .filter(|p| !matches!(p, proto::Protocol::Unspecified)); + let scopable = cfc_client::convert::exe_is_rule_scopable(exe); + let inbound = direction == proto::Direction::Inbound as i32; Self { name: String::new(), - exe: exe.to_string(), + exe: if scopable { + exe.to_string() + } else { + String::new() + }, dst_host: String::new(), - dst_net: format::host_cidr(dst_ip), + dst_net: if scopable { + String::new() + } else { + format::host_cidr(dst_ip) + }, dst_port: if dst_port == 0 { String::new() } else { dst_port.to_string() }, protocol, + carried_scope: CarriedScope { + direction: if inbound { direction } else { 0 }, + has_direction: inbound, + ..CarriedScope::default() + }, ..Self::default() } } @@ -466,10 +490,10 @@ pub enum Message { /// Opens the rule editor pre-filled from an observed connection. MakeRuleFromEvent { exe: String, - dst_host: String, dst_ip: String, dst_port: u32, protocol: i32, + direction: i32, }, CopyText(String), DismissLogEntry(usize), @@ -931,13 +955,13 @@ impl App { } Message::MakeRuleFromEvent { exe, - dst_host, dst_ip, dst_port, protocol, + direction, } => { self.editor = Some(RuleEditor::from_observed( - &exe, &dst_host, &dst_ip, dst_port, protocol, + &exe, &dst_ip, dst_port, protocol, direction, )); self.tab = Tab::Rules; Task::none() @@ -1056,6 +1080,11 @@ impl App { } } Err(e) => { + // Also in the footer: the banner sits at the bottom + // of the form and can be below the fold, which made + // Save look like it did nothing. + self.log + .error(format!("saving rule failed: {e}"), self.now_ms); editor.validation = Some(e); Task::none() } @@ -1070,7 +1099,10 @@ impl App { self.update(Message::VerdictSubmitted(Ok((String::new(), false, None)))) } Message::PromptRuleSaved(Err(error)) => self.update(Message::RuleSaved(Err(error))), - Message::RuleSaved(Ok(_)) => { + Message::RuleSaved(Ok(line)) => { + // Say what was stored: closing the editor in silence read + // as accepted whatever scope the rule ended up with. + self.log.info(line, self.now_ms); self.editor = None; let socket = self.socket_path.clone(); Task::perform(fetch_rules(socket), Message::RulesLoaded) @@ -1559,6 +1591,7 @@ async fn fetch_rules(path: PathBuf) -> Result, String> { client.list_rules().await.map_err(|e| e.to_string()) } +/// Saves `rule` and returns the log line describing what was stored. async fn upsert_rule(path: PathBuf, rule: proto::RuleInfo) -> Result { if let Some(scope) = rule .scope @@ -1567,8 +1600,30 @@ async fn upsert_rule(path: PathBuf, rule: proto::RuleInfo) -> Result *:443 tcp, always"`. +fn saved_rule_line(rule: &proto::RuleInfo) -> String { + use cfc_client::convert; + let summary = convert::rule_summary(rule) + .split_whitespace() + .collect::>() + .join(" "); + let protocol = rule + .scope + .as_ref() + .filter(|s| s.has_protocol) + .map(|s| format!(" {}", convert::protocol_label(s.protocol))) + .unwrap_or_default(); + let disabled = if rule.enabled { "" } else { ", disabled" }; + format!( + "rule saved: {summary}{protocol}, {}{disabled}", + convert::rule_duration_label(rule) + ) } fn build_rule_from_editor(ed: &RuleEditor) -> Result { @@ -1804,16 +1859,113 @@ mod tests { } #[test] - fn observed_seed_pins_the_numeric_endpoint_even_with_a_hostname() { - let ed = RuleEditor::from_observed("/bin/x", "example.com", "1.2.3.4", 443, 1); + fn observed_seed_scopes_by_program_or_pins_the_endpoint_without_one() { + let ed = RuleEditor::from_observed("/bin/x", "1.2.3.4", 443, 1, 0); + assert_eq!(ed.exe, "/bin/x"); assert!(ed.dst_host.is_empty()); - assert_eq!(ed.dst_net, "1.2.3.4/32"); + assert!( + ed.dst_net.is_empty(), + "a program rule is not pinned to one address" + ); assert_eq!(ed.dst_port, "443"); - let ed = RuleEditor::from_observed("/bin/x", "", "2001:db8::1", 0, 0); - assert_eq!(ed.dst_net, "2001:db8::1/128"); - assert!(ed.dst_port.is_empty()); - assert!(ed.protocol.is_none()); + for exe in [cfc_client::convert::UNKNOWN_EXE, ""] { + let ed = RuleEditor::from_observed(exe, "2001:db8::1", 0, 0, 0); + assert!(ed.exe.is_empty(), "{exe:?} is never seeded"); + assert_eq!(ed.dst_net, "2001:db8::1/128"); + assert!(ed.dst_port.is_empty()); + assert!(ed.protocol.is_none()); + } + + let inbound = proto::Direction::Inbound as i32; + let ed = + RuleEditor::from_observed(cfc_client::convert::UNKNOWN_EXE, "10.0.0.2", 22, 1, inbound); + assert!(ed.carried_scope.has_direction); + assert_eq!(ed.carried_scope.direction, inbound); + } + + /// Issue #46: "make rule" on a LIVE row must either send the rule or say + /// why not, and the rule it sends for an identified program must have + /// the CLI's scope (program, port, protocol), not a hostname or one IP. + #[test] + fn live_seeded_rule_is_sent_or_reported() { + use proto::Protocol::{Icmp, Tcp, Udp}; + let rows: [(&str, &str, u32, proto::Protocol); 8] = [ + ("/usr/bin/curl", "93.184.216.34", 443, Tcp), + ("/usr/bin/curl", "2001:db8::1", 443, Tcp), + ("/usr/bin/curl", "9.9.9.9", 53, Udp), + ("/usr/bin/curl", "93.184.216.34", 0, Icmp), + (cfc_client::convert::UNKNOWN_EXE, "93.184.216.34", 443, Tcp), + ("", "93.184.216.34", 443, Tcp), + ("relative/path", "93.184.216.34", 443, Tcp), + ( + cfc_client::convert::UNKNOWN_EXE, + "", + 0, + proto::Protocol::Unspecified, + ), + ]; + for (exe, dst_ip, dst_port, protocol) in rows { + let (mut app, _) = App::new(); + let _ = app.update(Message::MakeRuleFromEvent { + exe: exe.into(), + dst_ip: dst_ip.into(), + dst_port, + protocol: protocol as i32, + direction: 0, + }); + let ed = app.editor.as_ref().expect("editor opened"); + if let Ok(rule) = build_rule_from_editor(ed) { + let scope = rule.scope.unwrap(); + assert!(scope.dst_host.is_empty(), "{exe} {dst_ip}"); + if cfc_client::convert::exe_is_rule_scopable(exe) { + assert_eq!(scope.exe_path, exe); + assert!(scope.dst_net.is_empty(), "{exe} {dst_ip}"); + assert_eq!(scope.has_dst_port, dst_port != 0); + assert!(scope.has_protocol); + } else { + assert!(scope.exe_path.is_empty(), "{exe:?} never reaches a rule"); + } + } + let errors_before = app + .log + .iter() + .filter(|e| e.severity == status_log::Severity::Error) + .count(); + let sent = app.update(Message::SaveRule).units(); + let errors = app + .log + .iter() + .filter(|e| e.severity == status_log::Severity::Error) + .count(); + assert!( + sent == 1 || errors > errors_before, + "{exe:?} {dst_ip}: Save neither sent the rule nor said why" + ); + } + } + + #[test] + fn a_saved_rule_is_described_by_its_stored_scope() { + let rule = proto::RuleInfo { + action: proto::Action::Allow as i32, + duration: proto::Duration::Always as i32, + enabled: true, + scope: Some(proto::RuleScope { + exe_path: "/usr/bin/curl".into(), + dst_port: 443, + has_dst_port: true, + protocol: proto::Protocol::Tcp as i32, + has_protocol: true, + ..Default::default() + }), + ..Default::default() + }; + let line = saved_rule_line(&rule); + assert!( + line.starts_with("rule saved: allow /usr/bin/curl -> *:443 tcp"), + "{line}" + ); } /// A rule as the daemon reports it: created a while ago, already hit, @@ -1941,7 +2093,7 @@ mod tests { #[test] fn a_rule_seeded_from_an_observed_flow_is_new() { - let ed = RuleEditor::from_observed("/bin/x", "example.com", "1.2.3.4", 443, 1); + let ed = RuleEditor::from_observed("/bin/x", "1.2.3.4", 443, 1, 0); assert_eq!(ed.created_at_unix_ms, 0); assert_eq!(ed.hit_count, 0); assert!(ed.enabled); @@ -2067,7 +2219,7 @@ mod tests { let ed = RuleEditor::from_prompt(&prompt_event()); assert_eq!(ed.exe, "/usr/bin/curl"); assert!(ed.dst_host.is_empty()); - assert_eq!(ed.dst_net, "93.184.216.34/32"); + assert!(ed.dst_net.is_empty()); assert_eq!(ed.dst_port, "443"); assert_eq!(ed.protocol, Some(proto::Protocol::Tcp)); // It is a new rule, not an edit of an existing one. diff --git a/crates/cfc-ui/src/views/live.rs b/crates/cfc-ui/src/views/live.rs index f396d8f..97a1114 100644 --- a/crates/cfc-ui/src/views/live.rs +++ b/crates/cfc-ui/src/views/live.rs @@ -239,10 +239,10 @@ fn live_row(e: &LiveEntry) -> Element<'_, Message> { .padding([1, 6]) .on_press(Message::MakeRuleFromEvent { exe: proc.map(|p| p.exe.clone()).unwrap_or_default(), - dst_host: c.dst_host.clone(), dst_ip: c.dst_ip.clone(), dst_port: c.dst_port, protocol: c.protocol, + direction: c.direction, }) .style(crate::theme::subtle_icon) .into(), From c34be4dcb8c1774960f6cb4cb8ea56c7a0a2528a Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:14:03 +0200 Subject: [PATCH 048/125] fix(ui): keep a prompt until its verdict is applied A card was dropped before its verdict RPC ran, so a verdict that never reached the daemon left the flow to the timeout default with nothing to retry, and the one error line could be pushed out of the footer by five expiry warnings. The card now stays, disabled, until the daemon answers; a verdict that was not applied makes it answerable again. Sticky footer errors are evicted last. A customization whose prompt expired, or whose save failed after the answer applied, was closed on the next tick with the user's edits and error. It now stays open as a new rule that Save stores with UpsertRule. --- crates/cfc-ui/src/main.rs | 227 +++++++++++++++++++++++------ crates/cfc-ui/src/status_log.rs | 23 ++- crates/cfc-ui/src/views/prompts.rs | 10 +- 3 files changed, 209 insertions(+), 51 deletions(-) diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index b257915..a494947 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -275,7 +275,8 @@ impl RuleEditor { /// creating it"). /// /// The connection remains pending until Save submits an explicit verdict, - /// or the daemon applies its timeout policy. + /// or the daemon applies its timeout policy. In that second case the + /// editor stays open as a new rule (see `App::detach_orphaned_editor`). pub fn from_prompt(ev: &proto::PromptEvent) -> Self { let exe = ev .process @@ -369,6 +370,10 @@ pub struct PromptCard { pub deadline_unix_ms: i64, /// Wall clock before which the verdict buttons stay disabled. pub armed_at_ms: i64, + /// A verdict for it is on its way to the daemon. The card stays until + /// the daemon confirms, so a verdict that never arrived can be given + /// again instead of leaving the flow to the timeout default. + pub submitting: bool, } impl PromptCard { @@ -377,14 +382,26 @@ impl PromptCard { deadline_unix_ms: event.deadline_unix_ms, event, armed_at_ms: now_ms.saturating_add(PROMPT_ARM_MS), + submitting: false, } } + /// Whether its verdict controls accept input. pub fn armed(&self, now_ms: i64) -> bool { - now_ms >= self.armed_at_ms + !self.submitting && now_ms >= self.armed_at_ms } } +/// A verdict that did not do what the user asked. +#[derive(Debug, Clone)] +pub struct VerdictFailure { + pub prompt_id: String, + /// The daemon applied the answer and only the lasting rule failed. The + /// prompt is gone then, so its card must not come back. + pub applied: bool, + pub message: String, +} + /// How loudly a newly arrived prompt announces itself. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum Attention { @@ -477,7 +494,7 @@ pub enum Message { scope: Option, duration: proto::Duration, }, - VerdictSubmitted(Result<(String, bool, Option), String>), + VerdictSubmitted(Result<(String, bool, Option), VerdictFailure>), OpenEditor, EditExistingRule(String), CloseEditor, @@ -512,7 +529,7 @@ pub enum Message { EditorProtocol(Option), SaveRule, RuleSaved(Result), - PromptRuleSaved(Result<(String, bool, Option), String>), + PromptRuleSaved(Result<(String, bool, Option), VerdictFailure>), } /// Raw key press forwarded from the subscription. The decision of what a @@ -650,14 +667,7 @@ impl App { true } }); - if self - .editor - .as_ref() - .and_then(|editor| editor.prompt_id.as_ref()) - .is_some_and(|id| !self.prompts.iter().any(|card| &card.event.prompt_id == id)) - { - self.editor = None; - } + self.detach_orphaned_editor(); for label in expired { self.log.warn( format!( @@ -678,12 +688,57 @@ impl App { /// a key press already on its way when the card above expired or was /// answered does not land on the card that just moved up. fn sync_key_target(&mut self) { - let top = self.prompts.first().map(|card| &card.event.prompt_id); + let top = self.key_target_card().map(|card| &card.event.prompt_id); if self.key_target.as_ref().map(|(id, _)| id) != top { self.key_target = top.map(|id| (id.clone(), self.now_ms.saturating_add(PROMPT_ARM_MS))); } } + /// The top card that is not already being answered. + fn key_target_card(&self) -> Option<&PromptCard> { + self.prompts.iter().find(|card| !card.submitting) + } + + /// Keeps a customization whose prompt is gone (expired, answered + /// elsewhere, stream dropped) open as a plain new rule. + /// + /// It used to be closed, which threw away the rule the user was + /// building and any error banner explaining why a save failed. Save now + /// stores the rule with `UpsertRule`, since there is no prompt left to + /// answer. + fn detach_orphaned_editor(&mut self) { + let Some(editor) = self.editor.as_mut() else { + return; + }; + if editor + .prompt_id + .as_ref() + .is_some_and(|id| !self.prompts.iter().any(|card| &card.event.prompt_id == id)) + { + editor.prompt_id = None; + self.log.warn( + "the prompt being customized is gone; Save now stores a new rule", + self.now_ms, + ); + } + } + + /// Settles the card a verdict was sent for: gone once the daemon has + /// applied an answer, answerable again when nothing was applied. + fn settle_card(&mut self, prompt_id: &str, applied: bool) { + if applied { + self.prompts + .retain(|card| card.event.prompt_id != prompt_id); + } else if let Some(card) = self + .prompts + .iter_mut() + .find(|card| card.event.prompt_id == prompt_id) + { + card.submitting = false; + } + self.sync_key_target(); + } + fn update(&mut self, message: Message) -> Task { match message { Message::TabSelected(t) => { @@ -842,13 +897,7 @@ impl App { Message::PromptStreamEnded(e) => { self.stream_trouble = true; self.prompts.clear(); - if self - .editor - .as_ref() - .is_some_and(|editor| editor.prompt_id.is_some()) - { - self.editor = None; - } + self.detach_orphaned_editor(); info!("prompt stream interrupted: {e}"); Task::none() } @@ -867,14 +916,25 @@ impl App { scope, duration, } => { - self.prompts.retain(|p| p.event.prompt_id != prompt_id); + // One verdict per card at a time. The card stays, disabled, + // until the daemon confirms (see `settle_card`). + let Some(card) = self + .prompts + .iter_mut() + .find(|p| p.event.prompt_id == prompt_id && !p.submitting) + else { + return Task::none(); + }; + card.submitting = true; + self.sync_key_target(); let socket = self.socket_path.clone(); Task::perform( submit_verdict(socket, prompt_id, action, scope, duration, false), Message::VerdictSubmitted, ) } - Message::VerdictSubmitted(Ok((_, true, note))) => { + Message::VerdictSubmitted(Ok((prompt_id, true, note))) => { + self.settle_card(&prompt_id, true); // The daemon telling the user something true about the rule // it stored - today, that a user-writable binary was // hash-bound and a swapped file will prompt again. Shown in @@ -888,7 +948,8 @@ impl App { let socket = self.socket_path.clone(); Task::perform(fetch_rules(socket), Message::RulesLoaded) } - Message::VerdictSubmitted(Ok((_, false, _))) => { + Message::VerdictSubmitted(Ok((prompt_id, false, _))) => { + self.settle_card(&prompt_id, true); // The daemon had already answered this prompt itself. The // old code swallowed this, so the user believed they had // allowed something the timeout had actually decided. @@ -898,8 +959,17 @@ impl App { ); Task::none() } - Message::VerdictSubmitted(Err(e)) => { - self.log.error(format!("verdict failed: {e}"), self.now_ms); + Message::VerdictSubmitted(Err(failure)) => { + self.settle_card(&failure.prompt_id, failure.applied); + let still = if failure.applied { + "" + } else { + " - the prompt is still waiting for an answer" + }; + self.log.error( + format!("verdict failed: {}{still}", failure.message), + self.now_ms, + ); Task::none() } Message::OpenEditor => { @@ -1058,6 +1128,8 @@ impl App { Task::none() } Message::SaveRule => { + // The prompt may have expired since the last tick. + self.detach_orphaned_editor(); let Some(editor) = &mut self.editor else { return Task::none(); }; @@ -1069,8 +1141,18 @@ impl App { let action = editor.action; let duration = editor.duration; let scope = rule.scope; - self.prompts - .retain(|card| card.event.prompt_id != prompt_id); + // Held, not dropped, until the daemon answers: + // a dropped card closed this editor on the next + // tick and took a failed save's banner with it. + let Some(card) = self + .prompts + .iter_mut() + .find(|c| c.event.prompt_id == prompt_id && !c.submitting) + else { + return Task::none(); + }; + card.submitting = true; + self.sync_key_target(); Task::perform( submit_verdict(socket, prompt_id, action, scope, duration, true), Message::PromptRuleSaved, @@ -1094,11 +1176,18 @@ impl App { self.editor = None; self.update(Message::VerdictSubmitted(Ok((id, true, note)))) } - Message::PromptRuleSaved(Ok((_, false, _))) => { + Message::PromptRuleSaved(Ok((id, false, _))) => { self.editor = None; - self.update(Message::VerdictSubmitted(Ok((String::new(), false, None)))) + self.update(Message::VerdictSubmitted(Ok((id, false, None)))) + } + Message::PromptRuleSaved(Err(failure)) => { + // The editor stays open with the error. If the answer was + // applied the card is gone, and the editor then turns into a + // plain new rule the user can save again. + self.settle_card(&failure.prompt_id, failure.applied); + self.detach_orphaned_editor(); + self.update(Message::RuleSaved(Err(failure.message))) } - Message::PromptRuleSaved(Err(error)) => self.update(Message::RuleSaved(Err(error))), Message::RuleSaved(Ok(line)) => { // Say what was stored: closing the editor in silence read // as accepted whatever scope the rule ended up with. @@ -1199,7 +1288,7 @@ impl App { { return Task::none(); } - let Some(card) = self.prompts.first() else { + let Some(card) = self.key_target_card() else { return Task::none(); }; let ev = &card.event; @@ -1769,23 +1858,33 @@ async fn submit_verdict( scope: Option, duration: proto::Duration, require_confirmed_rule: bool, -) -> Result<(String, bool, Option), String> { +) -> Result<(String, bool, Option), VerdictFailure> { let wanted_rule = scope.is_some(); - let mut client = Client::connect(&path).await.map_err(|e| e.to_string())?; + let fail = |applied: bool, message: String| VerdictFailure { + prompt_id: prompt_id.clone(), + applied, + message, + }; + let mut client = Client::connect(&path) + .await + .map_err(|e| fail(false, e.to_string()))?; let outcome = client .submit_verdict(&prompt_id, action, duration, scope) .await - .map_err(|e| e.to_string())?; + .map_err(|e| fail(false, e.to_string()))?; // A verdict that applied but saved no rule is not a success to report // quietly: the user asked for a lasting answer, did not get one, and will // be prompted again by the next connection from the same program. if outcome.accepted && wanted_rule && outcome.rule_persisted == Some(false) { - return Err(outcome - .persist_error - .unwrap_or_else(|| "the answer applied, but no lasting rule was saved".to_string())); + return Err(fail( + true, + outcome + .persist_error + .unwrap_or_else(|| "the answer applied, but no lasting rule was saved".to_string()), + )); } if outcome.accepted && require_confirmed_rule && outcome.rule_persisted != Some(true) { - return Err("the verdict applied, but this daemon did not confirm a standing rule; restart the updated daemon".into()); + return Err(fail(true, "the verdict applied, but this daemon did not confirm a standing rule; restart the updated daemon".into())); } Ok((prompt_id, outcome.accepted, outcome.persist_note)) } @@ -2167,25 +2266,63 @@ mod tests { ); let _ = app.update(Message::PromptStreamEnded("disconnected".into())); assert!(app.prompts.is_empty()); - assert!(app.editor.is_none()); + assert!( + app.editor.as_ref().unwrap().prompt_id.is_none(), + "the edits stay, as a plain rule" + ); assert_eq!( app.update(Message::CustomizePromptRule(event.prompt_id)) .units(), 0 ); - assert!(app.editor.is_none()); + assert!(app.editor.as_ref().unwrap().prompt_id.is_none()); } #[test] - fn an_expired_prompt_closes_its_customization() { + fn an_expired_prompt_keeps_its_customization_as_a_new_rule() { let (mut app, _) = App::new(); let event = prompt_event(); app.prompts.push(PromptCard::new(event.clone(), 0)); - app.editor = Some(RuleEditor::from_prompt(&event)); + let mut editor = RuleEditor::from_prompt(&event); + editor.name = "careful".into(); + app.editor = Some(editor); app.now_ms = event.deadline_unix_ms + 1; app.housekeeping(); assert!(app.prompts.is_empty()); - assert!(app.editor.is_none()); + let editor = app.editor.as_ref().expect("the edits are kept"); + assert!(editor.prompt_id.is_none()); + assert_eq!(editor.name, "careful"); + // Save now stores the rule instead of answering a dead prompt. + assert_eq!(app.update(Message::SaveRule).units(), 1); + } + + #[test] + fn a_verdict_that_was_not_applied_leaves_the_prompt_answerable() { + let (mut app, _) = App::new(); + let event = prompt_event(); + app.prompts.push(PromptCard::new(event.clone(), 0)); + let submit = || Message::SubmitVerdict { + prompt_id: event.prompt_id.clone(), + action: proto::Action::Deny, + scope: None, + duration: proto::Duration::Once, + }; + assert_eq!(app.update(submit()).units(), 1); + assert!(app.prompts[0].submitting, "held until the daemon confirms"); + assert_eq!(app.update(submit()).units(), 0, "one verdict at a time"); + + let failure = |applied| VerdictFailure { + prompt_id: event.prompt_id.clone(), + applied, + message: "connection refused".into(), + }; + let _ = app.update(Message::VerdictSubmitted(Err(failure(false)))); + assert_eq!(app.prompts.len(), 1); + assert!(!app.prompts[0].submitting, "it can be answered again"); + + let _ = app.update(submit()); + let _ = app.update(Message::VerdictSubmitted(Err(failure(true)))); + assert!(app.prompts.is_empty(), "an applied answer retires the card"); } #[test] @@ -2311,8 +2448,8 @@ mod tests { app.tab = Tab::Prompts; assert_eq!(app.handle_key(key("A", shift)).units(), 1); - let ids: Vec<_> = app.prompts.iter().map(|c| &c.event.prompt_id).collect(); - assert_eq!(ids, ["below"], "the top card is answered, not the newest"); + assert!(app.prompts[0].submitting, "the top card is answered"); + assert!(!app.prompts[1].submitting, "not the newest"); // The card that moved up is disarmed again. assert_eq!( diff --git a/crates/cfc-ui/src/status_log.rs b/crates/cfc-ui/src/status_log.rs index 4c05e37..34d531e 100644 --- a/crates/cfc-ui/src/status_log.rs +++ b/crates/cfc-ui/src/status_log.rs @@ -7,7 +7,7 @@ use std::collections::VecDeque; -/// Entries kept at once. Older ones fall off the back. +/// Entries kept at once. The oldest non-sticky entry is dropped first. pub const CAP: usize = 5; /// How long a non-sticky entry stays visible. @@ -80,8 +80,16 @@ impl StatusLog { count: 1, sticky, }); + // Evict the oldest entry that is not sticky: a burst of routine + // lines (prompt expiries) must not push out the failure the user + // has to act on. while self.entries.len() > CAP { - self.entries.pop_back(); + let victim = self + .entries + .iter() + .rposition(|e| !e.sticky) + .unwrap_or(self.entries.len() - 1); + let _ = self.entries.remove(victim); } } @@ -163,6 +171,17 @@ mod tests { assert!(!log.iter().any(|e| e.text == "line 0")); } + #[test] + fn a_burst_of_warnings_does_not_evict_an_error() { + let mut log = StatusLog::default(); + log.error("verdict failed", 0); + for i in 0..(CAP + 3) { + log.warn(format!("prompt {i} expired"), 1 + i as i64); + } + assert_eq!(log.len(), CAP); + assert!(log.iter().any(|e| e.text == "verdict failed")); + } + #[test] fn coalescing_moves_the_entry_back_to_the_front() { let mut log = StatusLog::default(); diff --git a/crates/cfc-ui/src/views/prompts.rs b/crates/cfc-ui/src/views/prompts.rs index 048587a..07f766f 100644 --- a/crates/cfc-ui/src/views/prompts.rs +++ b/crates/cfc-ui/src/views/prompts.rs @@ -54,12 +54,13 @@ pub fn view<'a>( .unwrap_or(proto::Action::Unspecified as i32); let timeout_secs = status.map(|s| s.prompt_timeout_secs).unwrap_or(0); - // The first card is the one the A/D keys answer (see - // `App::answer_key_target`), so it is the one that carries the marker. + // The top card not already being answered is the one the A/D keys + // answer (see `App::answer_key_target`), so it carries the marker. + let target = prompts.iter().position(|c| !c.submitting); let cards: Vec> = prompts .iter() .enumerate() - .map(|(i, c)| prompt_card(c, timeout_action, timeout_secs, now_ms, i == 0)) + .map(|(i, c)| prompt_card(c, timeout_action, timeout_secs, now_ms, Some(i) == target)) .collect(); container(scrollable(column(cards).spacing(12).padding(8)).height(Length::Fill)) @@ -367,7 +368,8 @@ fn prompt_card( let ev = &card.event; let program = program_label(ev); // Disabled for a moment after the card appears, so a click aimed at - // whatever was here before cannot land on a verdict. + // whatever was here before cannot land on a verdict, and while its + // verdict is on the way. let armed = card.armed(now_ms); let marker: Element<'_, Message> = if key_target { From 3ea3811bdf20b1d97a1806e30ab438d382620852 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:14:35 +0200 Subject: [PATCH 049/125] fix(ui): ignore Pause right after a reconnect Pause takes Reconnect's place at the right end of the header as soon as a handshake lands, so the second click of a double-click on Reconnect paused enforcement, with no confirmation, for the daemon's default duration. Pause now stays disabled for one second after connecting. Resume is unaffected. --- crates/cfc-ui/src/main.rs | 40 ++++++++++++++++++++++++++++++++++----- 1 file changed, 35 insertions(+), 5 deletions(-) diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index a494947..a49bcfa 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -40,10 +40,11 @@ const DELETE_CONFIRM_MS: i64 = 3_000; const DEADLINE_TICK_MS: u64 = 400; /// How long a prompt's verdict controls stay disabled after the card -/// appears, and how long the keyboard target stays disarmed after it -/// changes. A click or key press already on its way when the layout changed -/// (the window was raised, a card arrived, the card above expired) must not -/// land on a verdict the user never saw. +/// appears, how long the keyboard target stays disarmed after it changes, +/// and how long Pause stays disabled after a reconnect. A click or key press +/// already on its way when the layout changed (the window was raised, a card +/// arrived, the card above expired, Pause replaced Reconnect) must not land +/// on a control the user never saw. const PROMPT_ARM_MS: i64 = 1_000; fn main() -> iced::Result { @@ -90,6 +91,9 @@ pub struct App { pub status_ticks: u32, /// Failed reconnect attempts, feeding the backoff. pub retry_attempts: u32, + /// When the last handshake succeeded. Pause stays disabled for + /// [`PROMPT_ARM_MS`] after it (see [`App::pause_armed`]). + pub connected_at_ms: i64, pub retry_at_ms: Option, /// Set when a gRPC stream drops; the badge shows "reconnecting" instead /// of the footer being rewritten every two seconds. @@ -593,6 +597,7 @@ impl App { status_ticks: 0, retry_attempts: 0, retry_at_ms: None, + connected_at_ms: 0, stream_trouble: false, now_ms: now_ms(), }; @@ -611,6 +616,14 @@ impl App { matches!(self.daemon, DaemonState::Connected) } + /// Pause takes the place of Reconnect, at the right end of the header, + /// as soon as a handshake lands. The second click of a double-click on + /// Reconnect would otherwise switch enforcement off with no + /// confirmation. + fn pause_armed(&self) -> bool { + self.now_ms >= self.connected_at_ms.saturating_add(PROMPT_ARM_MS) + } + fn connect_task(&mut self) -> Task { self.daemon = DaemonState::Connecting; self.retry_at_ms = None; @@ -750,6 +763,8 @@ impl App { self.connect_task() } Message::HandshakeDone(Ok(data)) => { + self.now_ms = now_ms(); + self.connected_at_ms = self.now_ms; self.daemon = DaemonState::Connected; self.status = Some(data.status); self.rules = data.rules; @@ -1051,6 +1066,9 @@ impl App { } Message::TogglePaused => { let current = self.status.as_ref().map(|s| s.paused).unwrap_or(false); + if !current && !self.pause_armed() { + return Task::none(); + } let socket = self.socket_path.clone(); Task::perform(set_paused(socket, !current), Message::PausedSet) } @@ -1503,7 +1521,7 @@ impl App { } else { button(text("Pause").size(12)) .padding([4, 14]) - .on_press(Message::TogglePaused) + .on_press_maybe(self.pause_armed().then_some(Message::TogglePaused)) .style(iced::widget::button::secondary) .into() } @@ -2459,6 +2477,18 @@ mod tests { ); } + #[test] + fn pause_ignores_a_click_right_after_reconnecting() { + let (mut app, _) = App::new(); + let _ = app.update(Message::HandshakeDone(Ok(HandshakeData { + status: proto::StatusResponse::default(), + rules: Vec::new(), + }))); + assert_eq!(app.update(Message::TogglePaused).units(), 0); + app.now_ms += PROMPT_ARM_MS; + assert_eq!(app.update(Message::TogglePaused).units(), 1); + } + #[test] fn a_prompt_card_is_disarmed_right_after_it_appears() { let card = PromptCard::new(prompt("p", "/usr/bin/curl"), 5_000); From ed08b7c3d23ff6061c90fe8e15be148c2aab5c64 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:16:52 +0200 Subject: [PATCH 050/125] fix(ui,tray): show process and DNS strings without controls or markup The CLI escaped control and bidi characters in untrusted values; the GUI and the tray showed them raw. A U+202E in a directory name reversed a prompt's Path row, a newline in a command line or a DNS name added fake lines, and the tray put the path and host into a notification body that dunst and mako parse as Pango markup, so a path segment could hide the rest of the path above "Always allow app". The CLI's escaping moves to cfc-client as display_safe and now also covers the GUI's prompt rows, live rows, rule rows and the tray bubbles. Tray bodies are markup-escaped. Rules and copied values keep the raw strings. The GUI's Remote row gains the verified or unverified name label the tray already showed. --- crates/cfc-cli/src/output.rs | 15 +------- crates/cfc-client/src/convert.rs | 48 ++++++++++++++++++++++-- crates/cfc-tray/src/main.rs | 5 ++- crates/cfc-tray/src/model.rs | 59 ++++++++++++++++++++++++++---- crates/cfc-ui/src/format.rs | 6 ++- crates/cfc-ui/src/views/prompts.rs | 57 +++++++++++++++++++++++++++-- crates/cfc-ui/src/views/rules.rs | 2 +- 7 files changed, 158 insertions(+), 34 deletions(-) diff --git a/crates/cfc-cli/src/output.rs b/crates/cfc-cli/src/output.rs index 10cea28..99cf72a 100644 --- a/crates/cfc-cli/src/output.rs +++ b/crates/cfc-cli/src/output.rs @@ -67,20 +67,7 @@ pub fn rfc3339(unix_ms: i64) -> Option { } /// Render untrusted values as one terminal-safe line. JSON keeps raw values. -pub fn terminal_safe(value: &str) -> String { - value - .chars() - .flat_map(|character| { - if character.is_control() - || matches!(character, '\u{202a}'..='\u{202e}' | '\u{2066}'..='\u{2069}') - { - character.escape_default().collect::>() - } else { - vec![character] - } - }) - .collect() -} +pub use cfc_client::convert::display_safe as terminal_safe; /// Clips a cell to `width` characters, marking the cut with `~` so a /// truncated path is never mistaken for a real one. diff --git a/crates/cfc-client/src/convert.rs b/crates/cfc-client/src/convert.rs index b1d5750..7bbf96d 100644 --- a/crates/cfc-client/src/convert.rs +++ b/crates/cfc-client/src/convert.rs @@ -2,6 +2,30 @@ use cfc_proto::v1 as pb; +/// Renders an untrusted value (a path, a command line, a DNS name) as one +/// line that cannot rearrange what is around it. +/// +/// Control characters (newlines included) and the bidi embedding, override +/// and isolate characters are written as escapes. Process strings are +/// chosen by the program being judged, or by whoever named its file, and DNS +/// names by whoever answers the query: a U+202E in a directory name reverses +/// the rest of a Path row, and an embedded newline adds a fake line to a +/// prompt. Only for display; rules and copies keep the raw value. +pub fn display_safe(value: &str) -> String { + value + .chars() + .flat_map(|character| { + if character.is_control() + || matches!(character, '\u{202a}'..='\u{202e}' | '\u{2066}'..='\u{2069}') + { + character.escape_default().collect::>() + } else { + vec![character] + } + }) + .collect() +} + pub fn action_label(a: i32) -> &'static str { match pb::Action::try_from(a).unwrap_or(pb::Action::Unspecified) { pb::Action::Allow => "allow", @@ -126,14 +150,15 @@ pub fn uid_label(uid: Option) -> String { } } +/// The program's basename for display, made [`display_safe`]. pub fn process_display(p: &pb::ProcessInfo) -> String { if p.exe.is_empty() { format!("pid:{}", p.pid) } else { - match std::path::Path::new(&p.exe).file_name() { + display_safe(&match std::path::Path::new(&p.exe).file_name() { Some(n) => n.to_string_lossy().into_owned(), None => p.exe.clone(), - } + }) } } @@ -149,7 +174,10 @@ pub fn rule_summary(r: &pb::RuleInfo) -> String { let target = scope .and_then(|s| { if !s.dst_host.is_empty() { - Some(format!("{} [legacy hostname; uncertain]", s.dst_host)) + Some(format!( + "{} [legacy hostname; uncertain]", + display_safe(&s.dst_host) + )) } else if !s.dst_net.is_empty() { Some(s.dst_net.clone()) } else { @@ -166,7 +194,7 @@ pub fn rule_summary(r: &pb::RuleInfo) -> String { if s.exe_path.is_empty() { None } else { - Some(s.exe_path.clone()) + Some(display_safe(&s.exe_path)) } }) .unwrap_or_else(|| "*".into()); @@ -224,6 +252,18 @@ pub fn rule_summary(r: &pb::RuleInfo) -> String { mod tests { use super::*; + #[test] + fn display_safe_escapes_controls_and_bidi_only() { + assert_eq!(display_safe("line\nnext\tcell"), "line\\nnext\\tcell"); + assert!(display_safe("\u{202e}\u{2066}").is_ascii()); + assert_eq!(display_safe("/usr/bin/caf\u{e9}"), "/usr/bin/caf\u{e9}"); + let p = pb::ProcessInfo { + exe: "/tmp/\u{202e}gpj.sh".into(), + ..Default::default() + }; + assert!(process_display(&p).is_ascii()); + } + fn proc(package: &str, provenance: pb::Provenance) -> pb::ProcessInfo { pb::ProcessInfo { package: package.into(), diff --git a/crates/cfc-tray/src/main.rs b/crates/cfc-tray/src/main.rs index f20032b..1a083ec 100644 --- a/crates/cfc-tray/src/main.rs +++ b/crates/cfc-tray/src/main.rs @@ -483,13 +483,14 @@ fn with_note(mut msg: String, note: &Option) -> String { } /// A short, non-actionable follow-up ("rule created", "too late"). 5s, -/// normal urgency. +/// normal urgency. The body is escaped like a prompt's (see +/// [`model::body_markup`]). fn notify_brief(body: String) { on_notification_thread(move || { let mut n = notify_rust::Notification::new(); let _ = brand(&mut n) .summary("Colony Firewall") - .body(&body) + .body(&model::body_markup(&body)) .timeout(notify_rust::Timeout::Milliseconds(5000)) .show(); }); diff --git a/crates/cfc-tray/src/model.rs b/crates/cfc-tray/src/model.rs index d7ccb0b..00971d7 100644 --- a/crates/cfc-tray/src/model.rs +++ b/crates/cfc-tray/src/model.rs @@ -223,7 +223,8 @@ pub struct PromptNotification { /// first line of the bubble reads as a sentence on its own. pub summary: String, /// The destination on the first line and the full exe path on the - /// second, so the path never runs into the prose. + /// second, so the path never runs into the prose. Already escaped with + /// [`body_markup`]. pub body: String, /// Remaining time until the prompt's deadline, clamped to at least /// [`MIN_PROMPT_TIMEOUT_MS`]. When it expires unanswered the daemon @@ -258,7 +259,7 @@ pub fn prompt_notification(ev: &proto::PromptEvent, now_unix_ms: i64) -> PromptN } else { format!( "{} ({ip}; {} hostname)", - c.dst_host, + cfc_client::convert::display_safe(&c.dst_host), if c.dst_host_verified { "verified" } else { @@ -277,7 +278,7 @@ pub fn prompt_notification(ev: &proto::PromptEvent, now_unix_ms: i64) -> PromptN let mut body = target; if !exe.is_empty() { body.push('\n'); - body.push_str(exe); + body.push_str(&cfc_client::convert::display_safe(exe)); } if cfc_client::convert::exe_is_rule_scopable(exe) { body.push_str("\nAlways allow app covers all destinations until the rule is removed."); @@ -300,7 +301,7 @@ pub fn prompt_notification(ev: &proto::PromptEvent, now_unix_ms: i64) -> PromptN let timeout_ms = remaining.clamp(i64::from(MIN_PROMPT_TIMEOUT_MS), i64::from(u32::MAX)) as u32; PromptNotification { summary, - body, + body: body_markup(&body), timeout_ms, offer_block: cfc_client::convert::exe_is_rule_scopable(exe), } @@ -421,9 +422,28 @@ pub fn one_shot_fallback(choice: PromptChoice, exe: &str) -> String { /// The basename, for a message meant to be read at a glance. fn exe_display_name(exe: &str) -> String { - std::path::Path::new(exe) - .file_name() - .map_or_else(|| exe.to_string(), |n| n.to_string_lossy().into_owned()) + cfc_client::convert::display_safe( + &std::path::Path::new(exe) + .file_name() + .map_or_else(|| exe.to_string(), |n| n.to_string_lossy().into_owned()), + ) +} + +/// Escapes a notification body for servers that parse it as markup. +/// +/// The freedesktop spec lets a server that advertises `body-markup` read +/// the body as a subset of HTML, and dunst and mako parse full Pango markup: +/// a path segment like `` hid the rest of the path right +/// above "Always allow app". The summary is plain text by the spec and is +/// left alone. +/// +/// ponytail: escaped whether or not the server advertises `body-markup`, so +/// a server without it shows `&` for a literal `&`. Only names that +/// carry `&`, `<` or `>` pay that; probe the capability if it ever matters. +pub fn body_markup(body: &str) -> String { + body.replace('&', "&") + .replace('<', "<") + .replace('>', ">") } /// How a newly arrived prompt is surfaced. @@ -764,6 +784,31 @@ mod tests { assert_eq!(n.body, "example.com (unknown; unverified hostname):443 (tcp)\n/usr/bin/curl\nAlways allow app covers all destinations until the rule is removed."); } + #[test] + fn prompt_notification_cannot_be_restyled_or_reflowed_by_its_strings() { + // A valid path whose segments are Pango markup, and a DNS name that + // forges a second, "verified" destination line. + let n = prompt_notification( + &prompt_event( + "/home/u/.cache/evil/firefox", + "google.com (1.2.3.4; verified hostname):443 (tcp)\n/usr/lib/firefox/firefox\n\n.x", + "6.6.6.6", + 0, + ), + 0, + ); + assert!(!n.body.contains('<') && !n.body.contains('>'), "{}", n.body); + assert!(n.body.contains("<span alpha='1'>"), "{}", n.body); + assert_eq!( + n.body.lines().count(), + 3, + "only the tray's own line breaks: {}", + n.body + ); + let n = prompt_notification(&prompt_event("/tmp/\u{202e}fdp.sh", "", "1.2.3.4", 0), 0); + assert!(n.summary.is_ascii() && n.body.is_ascii(), "{}", n.body); + } + #[test] fn prompt_notification_without_exe_offers_no_block_and_stays_one_line() { // No exe path: "Block app always" would need RuleScope { exe_path }, diff --git a/crates/cfc-ui/src/format.rs b/crates/cfc-ui/src/format.rs index a55a0d3..e615d1b 100644 --- a/crates/cfc-ui/src/format.rs +++ b/crates/cfc-ui/src/format.rs @@ -3,6 +3,7 @@ //! Deliberately free of iced types so the fiddly arithmetic (deadlines, //! CIDR widths, truncation) can be unit-tested without a renderer. +use cfc_client::convert::display_safe; use cfc_client::proto; /// Prompt lifetime assumed when the daemon does not report one, so the @@ -80,7 +81,7 @@ pub fn dest_display(dst_host: &str, dst_ip: &str, dst_port: u32) -> String { if dst_host.is_empty() { format!("{ip}:{dst_port}") } else { - format!("{dst_host} ({ip}:{dst_port})") + format!("{} ({ip}:{dst_port})", display_safe(dst_host)) } } @@ -94,6 +95,7 @@ pub fn dest_display(dst_host: &str, dst_ip: &str, dst_port: u32) -> String { /// Empty when the daemon named neither a host nor an address, so the caller /// can drop the row rather than render `?:0`. pub fn remote_display(dst_host: &str, dst_ip: &str, dst_port: u32) -> String { + let dst_host = display_safe(dst_host); match (dst_host.is_empty(), dst_ip.is_empty()) { (true, true) => String::new(), (true, false) => format!("{dst_ip}:{dst_port}"), @@ -133,7 +135,7 @@ pub fn dest_key(dst_host: &str, dst_ip: &str) -> String { dst_ip.to_string() } } else { - dst_host.to_string() + display_safe(dst_host) } } diff --git a/crates/cfc-ui/src/views/prompts.rs b/crates/cfc-ui/src/views/prompts.rs index 07f766f..1f52c58 100644 --- a/crates/cfc-ui/src/views/prompts.rs +++ b/crates/cfc-ui/src/views/prompts.rs @@ -106,9 +106,11 @@ pub fn detail_rows(ev: &proto::PromptEvent) -> Vec { if let Some(p) = ev.process.as_ref() { if !p.exe.is_empty() { rows.push(plain(PROGRAM_LABEL, convert::process_display(p))); + // Display values go through `display_safe`; the copy keeps the + // raw string. rows.push(DetailRow { label: "Path", - value: p.exe.clone(), + value: convert::display_safe(&p.exe), note: "", copy: Some(p.exe.clone()), }); @@ -121,7 +123,7 @@ pub fn detail_rows(ev: &proto::PromptEvent) -> Vec { if !cmdline.is_empty() && cmdline != p.exe && cmdline != convert::process_display(p) { rows.push(DetailRow { label: "Command line", - value: format::ellipsize(&cmdline, CMDLINE_MAX_CHARS), + value: format::ellipsize(&convert::display_safe(&cmdline), CMDLINE_MAX_CHARS), note: "", copy: Some(cmdline), }); @@ -138,7 +140,7 @@ pub fn detail_rows(ev: &proto::PromptEvent) -> Vec { } if !p.cwd.is_empty() { - rows.push(plain("Working dir", p.cwd.clone())); + rows.push(plain("Working dir", convert::display_safe(&p.cwd))); } // Before the buttons, because it changes what Allow means here: @@ -189,7 +191,18 @@ pub fn detail_rows(ev: &proto::PromptEvent) -> Vec { let remote = format::remote_display(&c.dst_host, &c.dst_ip, c.dst_port); if !remote.is_empty() { - rows.push(plain("Remote", remote)); + // Same trust label as the tray bubble: a name from an observed + // DNS answer is whatever the answering server said. + rows.push(DetailRow { + label: "Remote", + value: remote, + note: match (c.dst_host.is_empty(), c.dst_host_verified) { + (true, _) => "", + (false, true) => "(verified name)", + (false, false) => "(unverified name)", + }, + copy: None, + }); } if !matches!( @@ -669,6 +682,42 @@ mod tests { assert_eq!(value_of(&ev, "Protocol").unwrap(), "tcp"); } + #[test] + fn untrusted_strings_render_on_one_line_without_bidi() { + let mut ev = event(); + let p = ev.process.as_mut().unwrap(); + p.exe = "/home/u/\u{202e}gpj.sh".into(); + p.cmdline = vec!["x".into(), "\nPath /usr/bin/firefox".into()]; + p.cwd = "/tmp/\u{2066}a".into(); + ev.connection.as_mut().unwrap().dst_host = "evil\n.example".into(); + for row in detail_rows(&ev) { + assert!(!row.value.contains('\n'), "{}: {}", row.label, row.value); + assert!( + !row.value.contains(['\u{202e}', '\u{2066}']), + "{}: {}", + row.label, + row.value + ); + } + let path = detail_rows(&ev) + .into_iter() + .find(|r| r.label == "Path") + .unwrap(); + assert_eq!( + path.copy.unwrap(), + "/home/u/\u{202e}gpj.sh", + "the copy stays raw" + ); + assert_eq!( + detail_rows(&ev) + .into_iter() + .find(|r| r.label == "Remote") + .unwrap() + .note, + "(unverified name)" + ); + } + #[test] fn remote_prefers_the_hostname_the_daemon_resolved() { let ev = event(); diff --git a/crates/cfc-ui/src/views/rules.rs b/crates/cfc-ui/src/views/rules.rs index dc6bd5f..ec1cb5e 100644 --- a/crates/cfc-ui/src/views/rules.rs +++ b/crates/cfc-ui/src/views/rules.rs @@ -263,7 +263,7 @@ fn rule_row<'a>( text(if r.name.is_empty() { "(unnamed)".to_string() } else { - r.name.clone() + convert::display_safe(&r.name) }) .size(12), text(convert::rule_summary(r)).size(10), From f3b5b0e22fcd037e79ca0c6a845f0d5a46e7bbcb Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:18:11 +0200 Subject: [PATCH 051/125] fix(tray): free a prompt bubble's slot once its prompt is over An actionable slot was freed only by the bubble's own ActionInvoked or NotificationClosed. GNOME Shell keeps expired bubbles in its message list and sends neither, so three timed-out prompts held every slot and each later prompt fell into the overflow bubble, which cannot answer it, with one parked notification thread per stale bubble. Each slot now records its prompt's deadline and its bubble's server id. Slots more than two seconds past the deadline are freed on every poll and before a new prompt is placed, and their bubbles are closed with CloseNotification, which also ends the waiting thread. The overflow bubble no longer tells the user to open the window to answer: a window opened now does not receive prompts already pending. --- crates/cfc-tray/src/main.rs | 84 +++++++++++++++++++++++++++++++++--- crates/cfc-tray/src/model.rs | 75 ++++++++++++++++++++++++++++---- 2 files changed, 143 insertions(+), 16 deletions(-) diff --git a/crates/cfc-tray/src/main.rs b/crates/cfc-tray/src/main.rs index 1a083ec..575df04 100644 --- a/crates/cfc-tray/src/main.rs +++ b/crates/cfc-tray/src/main.rs @@ -17,7 +17,7 @@ use cfc_client::{proto, Client, StreamItem}; use ksni::menu::{StandardItem, SubMenu}; use ksni::{MenuItem, TrayMethods as _}; use model::{DaemonView, NotifyGate, PauseControl, PromptChoice, PromptPresentation}; -use std::collections::HashSet; +use std::collections::HashMap; use std::path::{Path, PathBuf}; use std::pin::Pin; use std::time::Duration; @@ -106,6 +106,12 @@ enum Cmd { exe: String, key: String, }, + /// An actionable prompt notification was shown; `id` is the server id + /// needed to close it once its prompt is over. + PromptShown { + prompt_id: String, + id: u32, + }, /// The collapsed overflow notification was shown; `id` is the server /// id needed to update its count in place later. OverflowShown { @@ -511,8 +517,8 @@ enum OverflowBubble { /// Live actionable-notification bookkeeping, owned by the main loop. struct PromptNotifier { tx: mpsc::UnboundedSender, - /// Prompt ids with an actionable notification currently on screen. - active: HashSet, + /// Prompts with an actionable notification currently on screen. + active: HashMap, /// Prompts folded into the overflow bubble since it appeared. overflow_count: u64, bubble: OverflowBubble, @@ -522,7 +528,7 @@ impl PromptNotifier { fn new(tx: mpsc::UnboundedSender) -> Self { Self { tx, - active: HashSet::new(), + active: HashMap::new(), overflow_count: 0, bubble: OverflowBubble::Down, } @@ -531,6 +537,7 @@ impl PromptNotifier { /// One prompt arrived: its own actionable notification while a slot /// is free, otherwise folded into the single overflow bubble. fn on_prompt(&mut self, ev: &proto::PromptEvent) { + self.reclaim_expired(); match model::present_prompt(self.active.len(), self.overflow_count) { PromptPresentation::Actionable => self.show_actionable(ev), PromptPresentation::Overflow { count } => { @@ -543,7 +550,13 @@ impl PromptNotifier { fn show_actionable(&mut self, ev: &proto::PromptEvent) { let n = model::prompt_notification(ev, now_unix_ms()); let prompt_id = ev.prompt_id.clone(); - self.active.insert(prompt_id.clone()); + self.active.insert( + prompt_id.clone(), + model::Slot { + deadline_unix_ms: ev.deadline_unix_ms, + server_id: None, + }, + ); let exe = ev .process .as_ref() @@ -572,6 +585,8 @@ impl PromptNotifier { notification.action(model::KEY_BLOCK, "Block app"); } notification.action(model::KEY_DEFAULT, "Details"); + let shown_id = prompt_id.clone(); + let shown_tx = tx.clone(); let done = move |key: &str| { // Failing only means the main loop is gone; the process // is on its way out. @@ -582,7 +597,13 @@ impl PromptNotifier { }); }; match notification.show() { - Ok(handle) => handle.wait_for_action(done), + Ok(handle) => { + let _ = shown_tx.send(Cmd::PromptShown { + prompt_id: shown_id, + id: handle.id(), + }); + handle.wait_for_action(done); + } Err(e) => { warn!("showing prompt notification: {e}"); // Free the slot; the daemon's timeout_action covers @@ -657,6 +678,53 @@ impl PromptNotifier { self.active.clear(); self.overflow_count = 0; } + + /// Frees the slots of prompts past their deadline and closes their + /// bubbles (see [`model::reclaim_expired`]). + fn reclaim_expired(&mut self) { + close_notifications(model::reclaim_expired(&mut self.active, now_unix_ms())); + } + + /// The bubble for `prompt_id` is on screen as `id`. If its slot was + /// already reclaimed, the prompt is over and the bubble goes too. + fn on_prompt_shown(&mut self, prompt_id: &str, id: u32) { + match self.active.get_mut(prompt_id) { + Some(slot) => slot.server_id = Some(id), + None => close_notifications(vec![id]), + } + } +} + +/// Closes notification bubbles by server id, off the main loop. +/// +/// The bubble's handle is consumed by the thread waiting on it, so this is +/// the D-Bus `CloseNotification` call made directly. The server then emits +/// `NotificationClosed`, which ends that wait. +fn close_notifications(ids: Vec) { + if ids.is_empty() { + return; + } + tokio::spawn(async move { + let close = async { + let conn = zbus::Connection::session().await?; + for id in ids { + conn.call_method( + Some("org.freedesktop.Notifications"), + "/org/freedesktop/Notifications", + Some("org.freedesktop.Notifications"), + "CloseNotification", + &(id,), + ) + .await?; + } + Ok::<_, anyhow::Error>(()) + }; + match tokio::time::timeout(CAPABILITY_PROBE_TIMEOUT, close).await { + Ok(Ok(())) => {} + Ok(Err(e)) => debug!("closing an expired prompt notification: {e}"), + Err(_) => debug!("closing an expired prompt notification timed out"), + } + }); } /// The collapsed "N more connections waiting" notification. Actionable @@ -948,6 +1016,7 @@ async fn run() -> anyhow::Result<()> { loop { tokio::select! { _ = ticker.tick() => { + notifier.reclaim_expired(); if !refresh(&handle, &mut client, &socket, &mut gate, &mut was_reachable, generic).await { break; } @@ -992,7 +1061,7 @@ async fn run() -> anyhow::Result<()> { Some(Cmd::PromptResult { prompt_id, exe, key }) => { // Slot freed regardless of outcome. After a stream // drop the id is already gone; remove is a no-op. - let current = notifier.active.remove(&prompt_id); + let current = notifier.active.remove(&prompt_id).is_some(); if key == model::KEY_DEFAULT { open_gui(); } else if current { @@ -1003,6 +1072,7 @@ async fn run() -> anyhow::Result<()> { // KEY_CLOSED / anything else: dismissed or expired - // the daemon's timeout_action covers it. } + Some(Cmd::PromptShown { prompt_id, id }) => notifier.on_prompt_shown(&prompt_id, id), Some(Cmd::OverflowShown { id }) => notifier.on_overflow_shown(id), Some(Cmd::OverflowResult { key }) => { notifier.on_overflow_result(); diff --git a/crates/cfc-tray/src/model.rs b/crates/cfc-tray/src/model.rs index 00971d7..965d2d0 100644 --- a/crates/cfc-tray/src/model.rs +++ b/crates/cfc-tray/src/model.rs @@ -3,6 +3,7 @@ //! tested without a daemon, a D-Bus session, or a clock. use cfc_client::{proto, ClientError}; +use std::collections::HashMap; /// At most one desktop notification per this many milliseconds, however /// fast prompts arrive. Only used by the generic (non-actionable) @@ -13,6 +14,10 @@ pub const NOTIFY_MIN_INTERVAL_MS: i64 = 30_000; /// prompts beyond the cap fold into one collapsed overflow notification. pub const MAX_ACTIONABLE_NOTIFICATIONS: usize = 3; +/// How long after a prompt's deadline its bubble keeps its actionable slot, +/// so a click made just before the deadline still reaches the daemon. +pub const SLOT_GRACE_MS: i64 = 2_000; + /// Floor for a prompt notification's expire timeout. A deadline that is /// already past still gets a brief, visible bubble rather than a 0ms /// ("never expire") or negative ("server default") timeout. @@ -446,6 +451,37 @@ pub fn body_markup(body: &str) -> String { .replace('>', ">") } +/// One actionable prompt bubble the tray counts against +/// [`MAX_ACTIONABLE_NOTIFICATIONS`]. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +pub struct Slot { + /// The prompt's deadline; 0 when the daemon attached none. + pub deadline_unix_ms: i64, + /// The notification server's id for the bubble, once it is shown. + pub server_id: Option, +} + +/// Frees the slots of prompts the daemon has already decided, and returns +/// the bubbles to close. +/// +/// A slot used to be freed only by the bubble's own `ActionInvoked` or +/// `NotificationClosed`. Servers that keep expired bubbles in a message +/// list (GNOME Shell) send neither, so three expired prompts held every slot +/// and every later prompt fell into the overflow bubble, which cannot answer +/// it. Closing the bubble also ends the thread waiting on it. +pub fn reclaim_expired(active: &mut HashMap, now_unix_ms: i64) -> Vec { + let mut close = Vec::new(); + active.retain(|_, slot| { + let expired = slot.deadline_unix_ms > 0 + && now_unix_ms > slot.deadline_unix_ms.saturating_add(SLOT_GRACE_MS); + if expired { + close.extend(slot.server_id); + } + !expired + }); + close +} + /// How a newly arrived prompt is surfaced. #[derive(Debug, Clone, Copy, PartialEq, Eq)] pub enum PromptPresentation { @@ -476,7 +512,10 @@ pub fn overflow_body(count: u64) -> String { } else { "connections" }; - format!("{count} more {noun} waiting — open Colony Firewall") + // Not "open Colony Firewall to answer": a window opened now does not + // receive prompts that are already pending, only one already running + // has them. + format!("{count} more {noun} waiting. An open Colony Firewall window can answer them; otherwise the default applies when they time out.") } #[cfg(test)] @@ -784,6 +823,30 @@ mod tests { assert_eq!(n.body, "example.com (unknown; unverified hostname):443 (tcp)\n/usr/bin/curl\nAlways allow app covers all destinations until the rule is removed."); } + #[test] + fn expired_slots_are_freed_and_their_bubbles_closed() { + let slot = |deadline_unix_ms, server_id| Slot { + deadline_unix_ms, + server_id, + }; + let mut active = HashMap::from([ + ("expired".to_string(), slot(1_000, Some(7))), + ("expired-unshown".to_string(), slot(1_000, None)), + ("in-grace".to_string(), slot(5_000, Some(8))), + ("no-deadline".to_string(), slot(0, Some(9))), + ]); + let close = reclaim_expired(&mut active, 5_000 + SLOT_GRACE_MS); + assert_eq!(close, vec![7]); + let mut left: Vec<_> = active.keys().map(String::as_str).collect(); + left.sort_unstable(); + assert_eq!(left, ["in-grace", "no-deadline"]); + assert_eq!( + present_prompt(active.len(), 0), + PromptPresentation::Actionable, + "a freed slot takes the next prompt" + ); + } + #[test] fn prompt_notification_cannot_be_restyled_or_reflowed_by_its_strings() { // A valid path whose segments are Pango markup, and a DNS name that @@ -964,14 +1027,8 @@ mod tests { #[test] fn overflow_body_counts_and_pluralizes() { - assert_eq!( - overflow_body(1), - "1 more connection waiting — open Colony Firewall" - ); - assert_eq!( - overflow_body(4), - "4 more connections waiting — open Colony Firewall" - ); + assert!(overflow_body(1).starts_with("1 more connection waiting.")); + assert!(overflow_body(4).starts_with("4 more connections waiting.")); } // --- capability fallback ------------------------------------------------- From 0f13088cdaa787ae2f6de5f7c6960159ca77370d Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:18:51 +0200 Subject: [PATCH 052/125] docs(changelog): list the GUI and tray fixes in this bundle Records the keyboard target and arming, the #46 rule seed, escaped process and DNS strings, kept prompt cards and customizations, the Pause arming and the tray slot reclamation. --- CHANGELOG.md | 35 +++++++++++++++++++++++++++++++++++ 1 file changed, 35 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9cc99af..ec9467d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,9 +19,32 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). when every compatible socket agrees on its owner. The process and descriptor found must still hold the socket after the executable is read; otherwise the identity is unknown. +- GUI: "make rule" on a Live row seeds the program, port and protocol, the + scope `cfc rules add --exe --dst-port --protocol` builds, instead of + pinning the one address seen, which left the app denied on its next + address (#46). A row without an identified program pins the address and + never seeds ``; an inbound row keeps its direction. "Customize" + on a prompt seeds the same way. A saved rule logs the scope it stored, and + a rule the editor refuses is also reported in the footer. +- GUI: a prompt arriving while others are pending no longer switches to the + Prompts tab; only the first one does, and raises the window. ### Security +- GUI: `A`, `D`, `Shift+A` and `Shift+D` answered the newest prompt, the + bottom card and often off-screen, from any tab and with Ctrl, Alt or Super + held, so `Shift+A` on the card being read could write an always-allow rule + for another program. They now answer the marked top card, only on the + Prompts tab and without those modifiers. The keys are disarmed for one + second whenever their target changes, and a card's buttons for one second + after it appears, so input already on its way when the window was raised + or a card moved does not answer it. +- GUI and tray: executable paths, command lines, working directories and DNS + names were shown raw, so bidi and control characters could reorder or add + lines to a prompt, and the tray's notification body was parsed as markup + by dunst and mako, so a path could hide part of itself. They are now + escaped as the CLI already did. The GUI's Remote row says whether the name + is verified. - While no daemon listens on the queue, new loopback flows are allowed (`oifname "lo" ct state new queue num 0 bypass`), so local services keep working when the daemon is down. Loopback Deny rules are not enforced then. @@ -121,6 +144,18 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). it. - A daemon started by hand under umask 000 created its socket directory world-writable, so a local user could replace the socket. +- GUI: a prompt card was dropped before its verdict reached the daemon, so a + failed verdict left the flow to the timeout default with nothing to retry. + The card now stays until the daemon answers. A customization whose prompt + expired was closed with the user's edits; it now stays open as a new rule. + Footer errors are no longer pushed out by a burst of warnings. +- GUI: Pause replaced Reconnect under the cursor as soon as the daemon came + back, so a double-click on Reconnect paused enforcement. Pause now stays + disabled for one second after connecting. +- Tray: on GNOME, three expired prompt bubbles held every actionable slot, + so later prompts only reached the overflow bubble, which cannot answer + them. Slots are freed once their prompt's deadline has passed, and the + stale bubbles are closed. ## [0.7.0] - 2026-09-30 From 1e96139f395d33bf8b93fbc1fe3b740143164b29 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:28:12 +0200 Subject: [PATCH 053/125] fix(confinement): pass the full BPF_PROG_QUERY attribute The gate's Query struct stopped after prog_cnt (32 bytes). Kernels 6.17 to 7.1 write query.revision at offset 56 regardless of the size passed, so every query wrote 8 bytes past a stack object in the root gate. The struct now covers the layout through revision, with a size assertion. --- crates/cfc-cli/src/confinement/native.rs | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/crates/cfc-cli/src/confinement/native.rs b/crates/cfc-cli/src/confinement/native.rs index b97b889..a92fb62 100644 --- a/crates/cfc-cli/src/confinement/native.rs +++ b/crates/cfc-cli/src/confinement/native.rs @@ -573,6 +573,9 @@ mod platform { Ok(file) } + // The whole `bpf_attr.query` layout through `revision`, not only the fields + // read here: kernels 6.17 to 7.1 write `revision` at offset 56 whatever + // size the caller passed, so a shorter struct is overwritten past its end. #[repr(C)] #[derive(Default)] struct Query { @@ -583,7 +586,12 @@ mod platform { prog_ids: u64, prog_cnt: u32, padding: u32, + prog_attach_flags: u64, + link_ids: u64, + link_attach_flags: u64, + revision: u64, } + const _: () = assert!(mem::size_of::() == 64); #[repr(C)] struct Info { fd: u32, From 06a7114e729e07b9f1b589a7317edab7777fd191 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:28:17 +0200 Subject: [PATCH 054/125] fix(confinement): accept ancestor cgroup_skb programs during attestation The gate required the unit's effective ingress and egress lists to be exactly its own two systemd programs. The daemon's DNS observer, on by default, is attached at the cgroup root as a link, so it is effective in every descendant and every confined launch was refused. The direct pair must still be exact; extra effective programs are now accepted, since cgroup_skb verdicts are ANDed and they can only drop more. --- crates/cfc-cli/src/confinement/native.rs | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/crates/cfc-cli/src/confinement/native.rs b/crates/cfc-cli/src/confinement/native.rs index a92fb62..1aa7dba 100644 --- a/crates/cfc-cli/src/confinement/native.rs +++ b/crates/cfc-cli/src/confinement/native.rs @@ -58,6 +58,11 @@ fn ip_prefixes(value: &Value) -> Result> { Ok(result) } +/// The unit's own pair must be attached directly and run. Extra effective +/// programs come from ancestors, such as the daemon's DNS observer on the +/// cgroup root. They are accepted because these attach points only take +/// cgroup_skb programs, whose verdicts the kernel ANDs: another program can +/// drop more traffic, never admit what the native pair refuses. fn program_set(direct: &[u32], effective: &[u32]) -> Result<()> { let expected: BTreeSet<_> = direct.iter().copied().collect(); let actual: BTreeSet<_> = effective.iter().copied().collect(); @@ -66,7 +71,7 @@ fn program_set(direct: &[u32], effective: &[u32]) -> Result<()> { "missing or duplicate native cgroup filters" ); ensure!( - effective.len() == 2 && actual == expected, + actual.len() == effective.len() && !actual.contains(&0) && actual.is_superset(&expected), "unexpected effective cgroup filters" ); Ok(()) @@ -1132,12 +1137,16 @@ mod tests { } #[test] - fn effective_filters_must_exactly_match_direct_pair() { + fn effective_filters_must_run_the_direct_pair() { assert!(program_set(&[7, 9], &[9, 7]).is_ok()); + // An ancestor's program, such as CFC's DNS observer on the root. + assert!(program_set(&[7, 9], &[7, 9, 10]).is_ok()); for (direct, effective) in [ (vec![], vec![]), (vec![7, 9], vec![7]), - (vec![7, 9], vec![7, 9, 10]), + (vec![7, 9], vec![7, 10]), + (vec![7, 9], vec![7, 9, 9]), + (vec![7, 9], vec![7, 9, 0]), (vec![7, 7], vec![7, 7]), (vec![0, 9], vec![0, 9]), ] { From d737a4987d7f6f56d6140f595fbc0b1e533ded0f Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:29:05 +0200 Subject: [PATCH 055/125] fix(confinement): log gate refusals to the journal with their own status The unit sent the gate's stderr to /dev/null and the gate exited 1, so a refused launch looked exactly like the application exiting 1 and its reason was lost. The unit now sends stderr to the journal (bwrap and the application still get /dev/null), the gate exits 125, and the launcher names the journal command when it sees that status. --- README.md | 3 ++- crates/cfc-cli/src/confinement/mod.rs | 11 ++++++++++- crates/cfc-cli/src/confinement/native.rs | 2 +- crates/cfc-cli/src/main.rs | 5 +++-- 4 files changed, 16 insertions(+), 5 deletions(-) diff --git a/README.md b/README.md index 605b223..d2691c2 100644 --- a/README.md +++ b/README.md @@ -326,7 +326,8 @@ The initial supported platform is x86_64 Linux with cgroup v2, systemd 262 or newer, a working system D-Bus, Bubblewrap 0.13.0 or newer, and libbpf-backed interface filtering. CFC verifies actual IP/interface BPF attachments, their policy maps and synthetic decisions before starting the application. Missing -support or failed verification refuses the launch. Local routes through `lo` +support or failed verification refuses the launch: the launcher reports +status 125 and the reason is in `journalctl -u cfc-app-ID.service`. Local routes through `lo` remain blocked even when an approved address later belongs to the host. Prepare an administrator-owned runtime containing the executable and all its diff --git a/crates/cfc-cli/src/confinement/mod.rs b/crates/cfc-cli/src/confinement/mod.rs index c6b2c3f..2d0d682 100644 --- a/crates/cfc-cli/src/confinement/mod.rs +++ b/crates/cfc-cli/src/confinement/mod.rs @@ -18,6 +18,9 @@ use std::path::{Component, Path, PathBuf}; use std::process::{Command, Stdio}; const CONTROL: &str = "/run/colony-firewall-apps"; +/// Exit status of a gate that refused to release the application. Its reason +/// goes to the unit's journal; the application itself never gets that stream. +pub(super) const GATE_REFUSED: i32 = 125; const UNITS: &str = "/run/systemd/system"; #[derive(Debug, Serialize, Deserialize)] @@ -247,7 +250,7 @@ fn unit_text(manifest: &Manifest, launcher: &Path) -> Result { .map(|ip| format!("{ip}/{}", if ip.is_ipv4() { 32 } else { 128 })) .collect::>() .join(" "); - Ok(format!("[Unit]\nDescription=CFC confined application\n[Service]\nType=exec\nSlice=system.slice\nDynamicUser=yes\nUser={}\nExecStart=+:\"{}\" __cfc_application_gate {}\nIPAddressDeny=any\nIPAddressAllow={}\nIPAccounting=no\nRestrictNetworkInterfaces=~lo\nDelegate=no\nStandardInput=null\nStandardOutput=null\nStandardError=null\nKillMode=control-group\nTimeoutStopSec=5s\nNoNewPrivileges=yes\nRestart=no\nFileDescriptorStoreMax=0\nNotifyAccess=none\nUMask=0077\n", user(&manifest.id), launcher, manifest.id, peers)) + Ok(format!("[Unit]\nDescription=CFC confined application\n[Service]\nType=exec\nSlice=system.slice\nDynamicUser=yes\nUser={}\nExecStart=+:\"{}\" __cfc_application_gate {}\nIPAddressDeny=any\nIPAddressAllow={}\nIPAccounting=no\nRestrictNetworkInterfaces=~lo\nDelegate=no\nStandardInput=null\nStandardOutput=null\nStandardError=journal\nKillMode=control-group\nTimeoutStopSec=5s\nNoNewPrivileges=yes\nRestart=no\nFileDescriptorStoreMax=0\nNotifyAccess=none\nUMask=0077\n", user(&manifest.id), launcher, manifest.id, peers)) } #[cfg(target_arch = "x86_64")] @@ -439,6 +442,10 @@ pub(super) async fn run(command: ApplicationsCmd, format: OutputFormat) -> Resul serde_json::json!({"id": manifest.id, "completed": completed, "application_status": status}) ); } + ensure!( + !completed || status != GATE_REFUSED.to_string(), + "confined application exited with status {GATE_REFUSED}, which the gate uses when it refuses a launch; the reason is in `journalctl -u {application}`" + ); ensure!( !completed || status == "0", "confined application failed with status {status}" @@ -668,6 +675,8 @@ mod tests { let text = unit_text(&manifest, Path::new("/usr/bin/cfc%$name")).unwrap(); assert!(text.contains("\nIPAddressDeny=any\nIPAddressAllow=\n")); assert!(text.contains("\nRestrictNetworkInterfaces=~lo\n")); + // The gate's refusal reason reaches the journal; bwrap gets /dev/null. + assert!(text.contains("\nStandardOutput=null\nStandardError=journal\n")); assert!(text.contains("ExecStart=+:\"/usr/bin/cfc%%$name\" __cfc_application_gate ")); manifest.allow = vec![ "203.0.113.7".parse().unwrap(), diff --git a/crates/cfc-cli/src/confinement/native.rs b/crates/cfc-cli/src/confinement/native.rs index 1aa7dba..c7a1d95 100644 --- a/crates/cfc-cli/src/confinement/native.rs +++ b/crates/cfc-cli/src/confinement/native.rs @@ -252,7 +252,7 @@ mod platform { ("NetworkNamespacePath", ""), ("StandardInput", "null"), ("StandardOutput", "null"), - ("StandardError", "null"), + ("StandardError", "journal"), ("KillMode", "control-group"), ("Restart", "no"), ("NotifyAccess", "none"), diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index 7eaf85d..20af694 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -238,10 +238,11 @@ fn main() { (Some(id), None) => confinement::gate(id), _ => Err(anyhow::anyhow!("invalid application gate invocation")), }; + // The unit sends this stream to the journal; the launcher points there. if let Err(error) = result { - eprintln!("cfc: {}", output::terminal_safe(&error.to_string())); + eprintln!("cfc: {}", output::terminal_safe(&format!("{error:#}"))); } - std::process::exit(error::EXIT_RUNTIME); + std::process::exit(confinement::GATE_REFUSED); } cli_main(); } From 46983b0a685b8914e348f2f845686a5aae44a8e9 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:29:41 +0200 Subject: [PATCH 056/125] fix(confinement): stop the tree on SIGINT, SIGHUP and SIGQUIT at any point Only SIGTERM was caught before provisioning; SIGINT was registered once the observation loop started and SIGHUP never, so Ctrl-C during systemctl, a closed terminal or a dropped SSH session killed the launcher and left the tree running with its grants. All four signals are now caught up front and end in stop(), and the tree identity is printed before the start job. --- README.md | 4 +++- crates/cfc-cli/src/confinement/mod.rs | 22 ++++++++++++++++------ 2 files changed, 19 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index d2691c2..4f9ef70 100644 --- a/README.md +++ b/README.md @@ -350,7 +350,9 @@ an external filesystem broker or concealed lower storage. To approve a peer, repeat the launch with `--allow IP` before `--`; repeat the flag for additional peers. The launcher prints the tree identity. Use `sudo cfc applications stop ID` to terminate it from another terminal, or -Ctrl-C in the launching terminal. +Ctrl-C in the launching terminal; closing that terminal or losing its SSH +session also stops the tree. A launcher killed outright (SIGKILL) cannot +clean up: the tree keeps running until `cfc applications stop ID`. Each active tree receives a reserved host UID and private PID, mount, user, IPC, UTS and cgroup namespaces. Its writable state is private and its runtime diff --git a/crates/cfc-cli/src/confinement/mod.rs b/crates/cfc-cli/src/confinement/mod.rs index 2d0d682..706010f 100644 --- a/crates/cfc-cli/src/confinement/mod.rs +++ b/crates/cfc-cli/src/confinement/mod.rs @@ -353,8 +353,14 @@ pub(super) async fn run(command: ApplicationsCmd, format: OutputFormat) -> Resul fs::create_dir_all(CONTROL)?; sealed_directory(Path::new(CONTROL))?; sealed_directory(Path::new(UNITS))?; - let mut terminate = - tokio::signal::unix::signal(tokio::signal::unix::SignalKind::terminate())?; + // Installed before anything is provisioned, so a signal that arrives + // while systemctl runs, or a closed terminal or SSH session, ends in + // stop() instead of the default action leaving the tree running. + use tokio::signal::unix::{signal, SignalKind}; + let mut interrupt = signal(SignalKind::interrupt())?; + let mut terminate = signal(SignalKind::terminate())?; + let mut hangup = signal(SignalKind::hangup())?; + let mut quit = signal(SignalKind::quit())?; let manifest = Manifest { id: uuid::Uuid::new_v4().simple().to_string(), runtime, @@ -380,6 +386,11 @@ pub(super) async fn run(command: ApplicationsCmd, format: OutputFormat) -> Resul "application preparation failed; cleanup: {cleanup:?}" ))); } + // Before the start job, so the identity is known even if this + // process is killed outright while the tree runs. + if matches!(format, OutputFormat::Human) { + eprintln!("Confined application: {}", manifest.id); + } let application = unit(&manifest.id); let started = manager(&["daemon-reload"]).and_then(|_| manager(&["start", &application])); @@ -389,9 +400,6 @@ pub(super) async fn run(command: ApplicationsCmd, format: OutputFormat) -> Resul error.context(format!("application setup failed; cleanup: {cleanup:?}")) ); } - if matches!(format, OutputFormat::Human) { - eprintln!("Confined application: {}", manifest.id); - } let outcome = async { let completed = loop { if !directory(&manifest.id).exists() { @@ -412,8 +420,10 @@ pub(super) async fn run(command: ApplicationsCmd, format: OutputFormat) -> Resul "unexpected application state: {state}" ); tokio::select! { - result = tokio::signal::ctrl_c() => { result?; break false; }, + _ = interrupt.recv() => break false, _ = terminate.recv() => break false, + _ = hangup.recv() => break false, + _ = quit.recv() => break false, _ = tokio::time::sleep(std::time::Duration::from_millis(250)) => {}, } }; From 846b60080335456281360a1a95b93aede47ac83b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:29:52 +0200 Subject: [PATCH 057/125] docs(confinement): state what a confined tree still shares with the host Grants apply in both directions, listeners share the host port space, /proc/net shows the host network namespace, and there is no memory cap. None of this was written down. --- README.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/README.md b/README.md index 4f9ef70..80ff406 100644 --- a/README.md +++ b/README.md @@ -370,6 +370,12 @@ edited. Stop the tree to revoke its permissions. Approving a peer approves that endpoint, including any remote relay it provides. Trusted host root, the operating system and kernel vulnerabilities are outside this boundary. +The tree shares the host network namespace. An approved peer may also +connect in to anything the tree listens on, a listener takes the port from +the whole host, and `/proc/net` shows the host's sockets and connections. +Nor is the tree resource-isolated: memory, including its tmpfs directories, +is not capped, and only systemd's default `TasksMax` bounds its processes. + ## Quick start Open the GUI: From 4939b5f6622075685b7aed48cf403171987ff381 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:30:42 +0200 Subject: [PATCH 058/125] fix(prompts): drop terminal type-ahead before showing a prompt The stdin reader queued every byte for the whole session and nothing drained it, so keys typed after a prompt expired, or while idle, answered the next prompt as soon as it was printed, and an arrow key skipped one prompt and left A or D to answer the next. In raw mode the queue and the terminal input buffer are now flushed before each prompt; line mode still answers prompts in order. Escape at the scope step skips instead of quitting the session. --- crates/cfc-cli/src/prompts.rs | 43 +++++++++++++++++++++++++++++++++-- crates/cfc-cli/src/tty.rs | 10 ++++++++ 2 files changed, 51 insertions(+), 2 deletions(-) diff --git a/crates/cfc-cli/src/prompts.rs b/crates/cfc-cli/src/prompts.rs index 42d8b70..c4f637f 100644 --- a/crates/cfc-cli/src/prompts.rs +++ b/crates/cfc-cli/src/prompts.rs @@ -247,7 +247,7 @@ struct Term { line_mode: bool, /// True when stdout is a terminal and an in-place countdown is useful. countdown: bool, - _raw: Option, + raw: Option, } enum Input { @@ -291,10 +291,23 @@ impl Term { keys: tty::spawn_key_reader(), line_mode: raw.is_none(), countdown: tty::stdout_is_tty(), - _raw: raw, + raw, }) } + /// Drops keys typed while no prompt was on screen, and what is left of + /// an escape sequence, so they cannot answer a prompt nobody has read + /// yet. Line mode keeps them: piped input answers prompts in order. + fn discard_typeahead(&mut self) { + if self.line_mode { + return; + } + if let Some(raw) = &self.raw { + raw.discard_input(); + } + while self.keys.try_recv().is_ok() {} + } + /// Waits for one of `valid` keys, redrawing a countdown until the /// prompt's deadline passes. async fn choose( @@ -506,6 +519,7 @@ async fn handle_prompt( return Ok(false); }; + term.discard_typeahead(); print_prompt(ev); let label = "answer: [a]llow [d]eny [r]eject [s]kip [q]uit"; @@ -587,6 +601,11 @@ async fn handle_prompt( Input::Key('1') => Scope::ExeAndPort, Input::Key('2') => Scope::Exe, Input::Key('3') => Scope::Destination, + // Escape, which an arrow key also starts. + Input::Key('s') => { + println!(" skipped (the daemon will apply {timeout_action} at the deadline)"); + return Ok(false); + } Input::Key(_) | Input::Interrupted | Input::Closed => return Ok(true), Input::TimedOut => { println!(" expired (daemon applied {timeout_action})"); @@ -711,6 +730,26 @@ fn print_prompt(ev: &proto::PromptEvent) { mod tests { use super::*; + // Keys typed while no prompt was shown, such as "a32" after one expired or + // the "[A" an Up arrow leaves, must not answer the next prompt. + #[test] + fn typeahead_is_dropped_before_a_new_prompt_in_raw_mode_only() { + for line_mode in [false, true] { + let (tx, keys) = tokio::sync::mpsc::unbounded_channel(); + for byte in *b"a32[A" { + tx.send(byte).unwrap(); + } + let mut term = Term { + keys, + line_mode, + countdown: false, + raw: None, + }; + term.discard_typeahead(); + assert_eq!(term.keys.try_recv().is_ok(), line_mode); + } + } + fn process() -> proto::ProcessInfo { proto::ProcessInfo { pid: 4242, diff --git a/crates/cfc-cli/src/tty.rs b/crates/cfc-cli/src/tty.rs index 9a15ecd..caab219 100644 --- a/crates/cfc-cli/src/tty.rs +++ b/crates/cfc-cli/src/tty.rs @@ -59,6 +59,16 @@ impl RawMode { } } +impl RawMode { + /// Discards bytes the terminal received that nobody has read yet. + pub fn discard_input(&self) { + // SAFETY: the fd is the live terminal captured in `enable`. + unsafe { + libc::tcflush(self.fd, libc::TCIFLUSH); + } + } +} + impl Drop for RawMode { fn drop(&mut self) { // SAFETY: restoring the exact termios captured in `enable`. From a0b9dcfbcb78a4edeffc91d73b56be5c4b4b0dd6 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:31:22 +0200 Subject: [PATCH 059/125] fix(client): escape the daemon's persist error and note for display SubmitVerdict's persist_error can quote a resolved executable path, which a local user can make contain escape sequences, and the CLI, GUI and tray showed it raw. The client now escapes both strings once for every front end. --- crates/cfc-client/src/lib.rs | 9 +++++++-- 1 file changed, 7 insertions(+), 2 deletions(-) diff --git a/crates/cfc-client/src/lib.rs b/crates/cfc-client/src/lib.rs index 5b54c13..e378d9e 100644 --- a/crates/cfc-client/src/lib.rs +++ b/crates/cfc-client/src/lib.rs @@ -149,6 +149,7 @@ pub struct VerdictOutcome { /// must complain only on `Some(false)`. pub rule_persisted: Option, /// Why the rule could not be stored, when one was asked for and failed. + /// Display-safe, like `persist_note`. pub persist_error: Option, /// Something true about the rule that WAS stored, worth showing: today, /// that it was hash-bound (a replaced binary will prompt again), or that @@ -291,8 +292,12 @@ impl Client { (true, false) => None, (true, true) => Some(!resp.persisted_rule_id.is_empty()), }, - persist_error: (!resp.persist_error.is_empty()).then_some(resp.persist_error), - persist_note: (!resp.persist_note.is_empty()).then_some(resp.persist_note), + // Escaped once here for every front end: both can quote a path + // whose name a local user chose. + persist_error: (!resp.persist_error.is_empty()) + .then(|| convert::display_safe(&resp.persist_error)), + persist_note: (!resp.persist_note.is_empty()) + .then(|| convert::display_safe(&resp.persist_note)), }) } From 6c4d6cf7d3fc8e566a994cbe4447f965e8ba8d9b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:35:14 +0200 Subject: [PATCH 060/125] fix(rules): pin bundle rules to the binary that connects The daemon matches /proc//exe, but several bundle entries pinned a launcher: /usr/bin/firefox (a shell script on Arch), npm-cli.js, the rustup proxy behind /usr/bin/cargo, and front ends whose helpers do the fetching (git, apt-get). bundle add reported them added and they never fired. Candidates now list the real binaries first and resolve() skips #! scripts and rustup. Epiphany, npm and pip are dropped: their traffic comes from a shared WebKit helper or an interpreter, and allowing those would allow every program using them. bundle remove still removes rules older versions installed under those names. --- README.md | 8 +- crates/cfc-cli/src/rules.rs | 141 ++++++++++++++++++++++++++++-------- docs/TROUBLESHOOTING.md | 2 +- 3 files changed, 115 insertions(+), 36 deletions(-) diff --git a/README.md b/README.md index 80ff406..343962c 100644 --- a/README.md +++ b/README.md @@ -220,7 +220,7 @@ For everything else, there are bundles: cfc rules bundle list # what there is, and what applies here cfc rules bundle add web --dry-run # preview sudo cfc rules bundle add web # installed browsers -> 443 and 80 -sudo cfc rules bundle add dev # git, cargo, npm, pip, docker +sudo cfc rules bundle add dev # git, ssh, cargo, docker/podman sudo cfc rules bundle add updates # apt, dnf, flatpak, yay sudo cfc rules bundle remove web # exactly the rules that bundle owns ``` @@ -229,8 +229,10 @@ Two properties worth knowing. **Every rule names an executable** - there is no way to write "allow tcp/443" here, because a payload phoning home uses 443 exactly like a browser does and a port-shaped rule cannot tell them apart. And entries whose program is not installed on this machine -are **skipped and reported**, so "4 added, 10 skipped" is the normal -outcome of `bundle add web` on a box with two browsers. +are **skipped and reported**, so "4 added, 8 skipped" is the normal +outcome of `bundle add web` on a box with two browsers. Each rule names the +binary that actually connects, never a launcher script, so tools where only +an interpreter connects (npm, pip) have no bundle entry. **3. Give prompts somewhere to go.** On a desktop, launch the GUI: diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index e586df7..e3a1797 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -1378,6 +1378,16 @@ fn apply_simple(s: &OsnSimple, scope: &mut proto::RuleScope) -> anyhow::Result<( /// The first candidate that **exists on this machine** is used; if none does, /// the entry is skipped and said out loud, so "installed 4 of 7, skipped 3 not /// present" is a normal, legible outcome rather than a silent partial success. +/// +/// # Candidates name the program that connects +/// +/// The daemon matches `/proc//exe`, the image that was mapped. A launcher +/// script (`/usr/bin/firefox` on Arch), a proxy that execs another binary +/// (rustup's `cargo`) or a front end whose helper does the fetching (`git`, +/// `apt-get`) never appears there, so a rule pinned to it never fires. Real +/// binaries come first, and launchers are skipped (see [`is_launcher`]). +/// Tools where only an interpreter connects (npm, pip) are not bundled at all: +/// allowing `/usr/bin/node` to reach 443 would allow every Node program. struct BundleRule { name: &'static str, /// Absolute paths to try, in order. First one that exists wins. @@ -1411,11 +1421,26 @@ impl BundleRule { self.exe_candidates .iter() .copied() - .find(|p| std::path::Path::new(p).is_file()) + .filter(|p| std::path::Path::new(p).is_file()) .map(|p| cfc_core::exe_path::resolve(std::path::Path::new(p)).into_path()) + .find(|p| !is_launcher(p)) } } +/// True for a file that execs another image instead of connecting itself: a +/// `#!` script, or the rustup proxy every toolchain binary links to. +fn is_launcher(path: &std::path::Path) -> bool { + use std::io::Read as _; + if path.file_name() == Some(std::ffi::OsStr::new("rustup")) { + return true; + } + let mut head = [0u8; 2]; + std::fs::File::open(path) + .and_then(|mut file| file.read_exact(&mut head)) + .is_ok() + && head == *b"#!" +} + /// A named, selectable set of rules. struct Bundle { name: &'static str, @@ -1561,9 +1586,10 @@ fn bundles() -> Vec { name: "updates", summary: "package managers beyond pacman/paru, which are in `system`", rules: vec![ + // apt-get hands the fetch to its method helpers. BundleRule { name: "updates-apt-https", - exe_candidates: &["/usr/bin/apt-get", "/usr/lib/apt/methods/https"], + exe_candidates: &["/usr/lib/apt/methods/https"], dst_port: Some(443), protocol: Some(Tcp), direction: None, @@ -1573,7 +1599,7 @@ fn bundles() -> Vec { // are signed, so the transport is not what protects them. BundleRule { name: "updates-apt-http", - exe_candidates: &["/usr/bin/apt-get", "/usr/lib/apt/methods/http"], + exe_candidates: &["/usr/lib/apt/methods/http"], dst_port: Some(80), protocol: Some(Tcp), direction: None, @@ -1614,10 +1640,14 @@ fn bundles() -> Vec { name: "dev", summary: "the tools that fetch code and dependencies", rules: vec![ - // git speaks both: HTTPS remotes and ssh:// remotes. + // git speaks both, but through helpers: git-remote-https for + // HTTPS remotes and ssh for ssh:// remotes. BundleRule { name: "dev-git-https", - exe_candidates: &["/usr/bin/git"], + exe_candidates: &[ + "/usr/lib/git-core/git-remote-https", + "/usr/libexec/git-core/git-remote-https", + ], dst_port: Some(443), protocol: Some(Tcp), direction: None, @@ -1625,7 +1655,7 @@ fn bundles() -> Vec { }, BundleRule { name: "dev-git-ssh", - exe_candidates: &["/usr/bin/git"], + exe_candidates: &["/usr/bin/ssh"], dst_port: Some(22), protocol: Some(Tcp), direction: None, @@ -1639,22 +1669,6 @@ fn bundles() -> Vec { direction: None, src_net: None, }, - BundleRule { - name: "dev-npm-https", - exe_candidates: &["/usr/bin/npm", "/usr/bin/node"], - dst_port: Some(443), - protocol: Some(Tcp), - direction: None, - src_net: None, - }, - BundleRule { - name: "dev-pip-https", - exe_candidates: &["/usr/bin/pip", "/usr/bin/pip3"], - dst_port: Some(443), - protocol: Some(Tcp), - direction: None, - src_net: None, - }, BundleRule { name: "dev-docker-https", exe_candidates: &["/usr/bin/dockerd", "/usr/bin/podman"], @@ -1843,16 +1857,28 @@ fn bundles() -> Vec { fn browser_rules() -> Vec { /// `(rule stem, candidate paths)`. const BROWSERS: &[(&str, &[&str])] = &[ - ("firefox", &["/usr/bin/firefox", "/usr/lib/firefox/firefox"]), - ("librewolf", &["/usr/bin/librewolf"]), + ( + "firefox", + &[ + "/usr/lib/firefox/firefox", + "/usr/lib64/firefox/firefox", + "/usr/bin/firefox", + ], + ), + ( + "librewolf", + &["/usr/lib/librewolf/librewolf", "/usr/bin/librewolf"], + ), ( "chromium", - &["/usr/bin/chromium", "/usr/lib/chromium/chromium"], + &["/usr/lib/chromium/chromium", "/usr/bin/chromium"], ), - ("chrome", &["/usr/bin/google-chrome-stable"]), - ("brave", &["/usr/bin/brave"]), - ("vivaldi", &["/usr/bin/vivaldi-stable"]), - ("epiphany", &["/usr/bin/epiphany"]), + ("chrome", &["/opt/google/chrome/chrome"]), + ( + "brave", + &["/opt/brave.com/brave/brave", "/opt/brave-bin/brave"], + ), + ("vivaldi", &["/opt/vivaldi/vivaldi-bin"]), ]; // `&'static str` names are needed by `BundleRule`, and these are built at @@ -1898,7 +1924,7 @@ struct Planned { /// Entries whose program is installed here, with the path as /proc will /// report it - not necessarily the candidate that matched. present: Vec<(&'static str, PathBuf)>, - /// Entries skipped because no candidate path exists. + /// Entries skipped because no candidate exists, or only a launcher does. absent: Vec<&'static str>, } @@ -2159,7 +2185,7 @@ pub async fn bundle_add( // network. if !planned.absent.is_empty() { println!( - "\nnot installed on this machine, so skipped ({}):", + "\nno program here that a rule can match (not installed, or only a launcher), so skipped ({}):", planned.absent.len() ); for n in &planned.absent { @@ -2175,6 +2201,17 @@ pub async fn bundle_add( Ok(()) } +/// Entries a later version dropped because their rule could never fire: +/// Epiphany fetches through the WebKit network process every WebKitGTK app +/// shares, and npm and pip connect as their interpreter. `bundle remove` +/// still removes what older versions installed under these names. +const RETIRED_BUNDLE_RULES: &[(&str, &str)] = &[ + ("web", "web-epiphany-https"), + ("web", "web-epiphany-http"), + ("dev", "dev-npm-https"), + ("dev", "dev-pip-https"), +]; + /// `cfc rules bundle remove ` /// /// Removes only deterministic IDs created by this bundle. Existing rules @@ -2186,10 +2223,16 @@ pub async fn bundle_remove( format: OutputFormat, ) -> CliResult { let bundle = find_bundle(name)?; + let retired = RETIRED_BUNDLE_RULES + .iter() + .filter(|(b, _)| *b == bundle.name) + .map(|(_, name)| *name); let owned: std::collections::HashSet = bundle .rules .iter() - .map(|r| bundle_rule_id(bundle.name, r.name)) + .map(|r| r.name) + .chain(retired) + .map(|name| bundle_rule_id(bundle.name, name)) .collect(); let existing = client.list_rules().await?; @@ -2608,6 +2651,40 @@ mod json_tests { mod bundle_tests { use super::*; + // `/usr/bin/firefox` on Arch is `exec /usr/lib/firefox/firefox`, and + // `/usr/bin/cargo` under rustup is the rustup proxy. Neither is ever + // /proc//exe, so a rule pinned to one never fires. + #[test] + fn launchers_are_skipped_for_the_binary_that_connects() { + let dir = std::env::temp_dir().join(format!("cfc-bundle-{}", uuid::Uuid::new_v4())); + std::fs::create_dir(&dir).unwrap(); + let script = dir.join("firefox"); + let real = dir.join("firefox-bin"); + let proxy = dir.join("rustup"); + std::fs::write(&script, "#!/bin/sh\nexec firefox-bin \"$@\"\n").unwrap(); + std::fs::write(&real, b"\x7fELF").unwrap(); + std::fs::write(&proxy, b"\x7fELF").unwrap(); + let leak = |p: &std::path::Path| -> &'static str { + Box::leak(p.to_str().unwrap().to_owned().into_boxed_str()) + }; + let entry = |candidates: Vec<&'static str>| BundleRule { + name: "test", + exe_candidates: Box::leak(candidates.into_boxed_slice()), + dst_port: Some(443), + protocol: None, + direction: None, + src_net: None, + }; + let resolved = entry(vec![leak(&script), leak(&real)]).resolve(); + assert_eq!( + resolved, + Some(cfc_core::exe_path::resolve(&real).into_path()) + ); + assert_eq!(entry(vec![leak(&script)]).resolve(), None); + assert_eq!(entry(vec![leak(&proxy)]).resolve(), None); + std::fs::remove_dir_all(dir).unwrap(); + } + /// The invariant the whole feature rests on. /// /// A bundle that installed a bare "allow tcp/443" outbound would re-open diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 0818f21..4c7fd06 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -412,7 +412,7 @@ set. For a bounded unattended window - during a package install, say - **2. Pre-seed rules and accept the fallback.** `cfc rules bundle add system` (also spelled `cfc rules bootstrap-defaults`) covers the usual system services. `cfc rules bundle list` shows the others — -`web` for installed browsers, `dev` for git/cargo/npm, `updates` for +`web` for installed browsers, `dev` for git/cargo/docker, `updates` for apt/dnf/flatpak — each scoped to a specific executable, never to a bare port. Entries whose program is not installed here are skipped and reported. Add your own with From c00667e42d4fd80d0b2a35da0efd676272a250de Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:35:49 +0200 Subject: [PATCH 061/125] fix(rules): accept bundle rules seeded before 0.7.0 0.7.0 gave bundle rules deterministic ids and refused any same-named rule under another id, so bootstrap-defaults and bundle add failed on every host seeded by 0.6.0, calling the bundle's own rules outside it. A same-named rule that grants exactly what the entry would now counts as present; a different one still stops the command, and the error names it. bundle add also says when an installed rule pins another path than the bundle now names. The docs no longer promise skipping by name. --- README.md | 5 +-- crates/cfc-cli/src/rules.rs | 65 ++++++++++++++++++++++++++++++++++--- docs/HARDENING.md | 4 ++- 3 files changed, 66 insertions(+), 8 deletions(-) diff --git a/README.md b/README.md index 343962c..40cd181 100644 --- a/README.md +++ b/README.md @@ -208,8 +208,9 @@ This installs twelve allow rules - systemd-resolved DNS (:53), systemd-timesyncd and chronyd NTP (:123/udp), the DHCP clients (dhcpcd, NetworkManager and systemd-networkd, :67 and :547/udp), pacman and paru HTTPS mirrors (:443/tcp), and the SSH client (:22/tcp) - and is -idempotent (already-present rules are skipped by name; `--dry-run` -previews). **Do not skip this step.** No profile allows unmatched remote flows +idempotent (rules it installed are skipped, as are identical same-named +rules seeded before 0.7.0; a different rule with one of its names stops it +before anything changes; `--dry-run` previews). **Do not skip this step.** No profile allows unmatched remote flows on its own. With no rules and no UI connected, unmatched queued remote connections are denied. Filtering starts before the network is configured (see below), and these rules keep DHCP, DNS and NTP usable. diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index e3a1797..1e8d34a 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -1956,6 +1956,15 @@ fn proto_for(spec: &BundleRule, exe: &str) -> proto::RuleInfo { ) } +/// Whether `rule` grants exactly what `wanted` would: same action, duration and +/// scope. Name, id and enabled state are not policy. +fn same_policy(rule: &proto::RuleInfo, wanted: &proto::RuleInfo) -> bool { + rule.action == wanted.action + && rule.duration == wanted.duration + && rule.duration_seconds == wanted.duration_seconds + && rule.scope == wanted.scope +} + fn bundle_rule_id(bundle: &str, name: &str) -> String { use sha2::{Digest, Sha256}; let digest = Sha256::digest(format!("colony-firewall-bundle\0{bundle}\0{name}")); @@ -2133,19 +2142,47 @@ pub async fn bundle_add( let planned = plan(&bundle); let mut added = Vec::new(); let mut skipped_present = 0u32; - for (rule_name, _) in &planned.present { + // A rule with an entry's name but another id was not installed by this + // bundle. One identical to the entry is that entry as seeded before 0.7.0 + // gave bundle rules their own ids, and counts as present; any other one + // stops the command before it changes anything. + let mut legacy = std::collections::HashSet::new(); + for (rule_name, exe) in &planned.present { let id = bundle_rule_id(bundle.name, rule_name); - if existing + let wanted = proto_for(by_name[*rule_name], &exe.to_string_lossy()); + for rule in existing .iter() - .any(|rule| rule.name == *rule_name && rule.id != id) + .filter(|rule| rule.name == *rule_name && rule.id != id) { - return Err(CliError::runtime(format!("bundle rule `{rule_name}` collides with a rule outside this bundle; nothing was changed"))); + if same_policy(rule, &wanted) { + legacy.insert(*rule_name); + } else { + return Err(CliError::runtime(format!( + "bundle rule `{rule_name}` collides with rule {} of the same name, which this \ + bundle did not install and which differs from it; rename or remove that rule \ + and retry. Nothing was changed", + short_id(&rule.id) + ))); + } } } for (rule_name, exe) in &planned.present { let id = bundle_rule_id(bundle.name, rule_name); - if existing.iter().any(|rule| rule.id == id) { + if let Some(rule) = existing.iter().find(|rule| rule.id == id) { + let stored = rule.scope.as_ref().map_or("", |s| s.exe_path.as_str()); + if !format.is_json() && std::path::Path::new(stored) != exe.as_path() { + println!( + "kept: {rule_name} pins {}, but this bundle now names {}; \ + `bundle remove` and `bundle add` replace it", + output::terminal_safe(stored), + output::terminal_safe(&exe.to_string_lossy()) + ); + } + skipped_present += 1; + continue; + } + if legacy.contains(rule_name) { skipped_present += 1; continue; } @@ -2651,6 +2688,24 @@ mod json_tests { mod bundle_tests { use super::*; + // Hosts seeded before 0.7.0 hold the bundle's rules under random ids. + // An identical copy counts as present; an edited one still blocks. + #[test] + fn a_pre_0_7_copy_of_a_bundle_rule_counts_only_while_unchanged() { + let bundle = find_bundle("inbound").unwrap(); + let wanted = proto_for(&bundle.rules[0], ""); + let mut legacy = wanted.clone(); + legacy.id = "11111111-1111-4111-8111-111111111111".into(); + legacy.enabled = false; + assert!(same_policy(&legacy, &wanted)); + let mut deny = legacy.clone(); + deny.action = proto::Action::Deny as i32; + assert!(!same_policy(&deny, &wanted)); + let mut wider = legacy.clone(); + wider.scope.as_mut().unwrap().src_net.clear(); + assert!(!same_policy(&wider, &wanted)); + } + // `/usr/bin/firefox` on Arch is `exec /usr/lib/firefox/firefox`, and // `/usr/bin/cargo` under rustup is the rustup proxy. Neither is ever // /proc//exe, so a rule pinned to one never fires. diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 7a2a668..0dbc890 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -90,7 +90,9 @@ You can install the system service rules with one command: cfc rules bootstrap-defaults ``` -This is idempotent: it skips rules already present by name. +This is idempotent: it skips the rules it installed earlier and identical +same-named rules seeded before 0.7.0. A different rule with one of its names +stops it before anything changes. ## What to *deny* first From 69736f7188c8d04b5d20cfea6e61db9e11c7dee6 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:36:19 +0200 Subject: [PATCH 062/125] fix(rules): keep a bundle rule edited into a deny on bundle remove The GUI editor keeps a rule's id, so a bundle Allow turned into a Deny still carried the bundle's id and bundle remove deleted it, letting the traffic it stopped fall through to the prompt or the default. Bundles install only Allows, so remove now keeps any owned rule that is not one and says so. --- crates/cfc-cli/src/main.rs | 5 ++-- crates/cfc-cli/src/rules.rs | 16 ++++++++++++ crates/cfc-cli/tests/cli_e2e.rs | 43 +++++++++++++++++++++++++++++++++ 3 files changed, 62 insertions(+), 2 deletions(-) diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index 20af694..ea651ab 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -219,8 +219,9 @@ enum BundleCmd { }, /// Remove the rules a bundle installed. /// - /// Matches the bundle's exact rule names, never a prefix, so a rule you - /// wrote yourself is never caught by it. + /// Matches the ids the bundle gave its rules, never a name or a prefix, + /// so a rule you wrote yourself is never caught by it. A bundle rule you + /// edited into a deny or reject is kept. Remove { /// Bundle name (see `cfc rules bundle list`). name: String, diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 1e8d34a..e11d70f 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -2274,7 +2274,22 @@ pub async fn bundle_remove( let existing = client.list_rules().await?; let mut removed = Vec::new(); + let mut kept = Vec::new(); for r in existing.iter().filter(|r| owned.contains(&r.id)) { + // Bundles install only Allows. An editor keeps the id, so one that is + // now a Deny or Reject is the user's decision, and deleting it would + // let the traffic it stops through to the prompt or the default. + if r.action != proto::Action::Allow as i32 { + if !format.is_json() { + println!( + "kept: {} ({}) was changed from allow; remove it by id if it should go", + short_id(&r.id), + output::terminal_safe(&r.name) + ); + } + kept.push(r.name.clone()); + continue; + } if !dry_run { client.delete_rule(&r.id).await?; } @@ -2295,6 +2310,7 @@ pub async fn bundle_remove( "dry_run": dry_run, "removed": removed.len(), "rules": removed, + "kept": kept, })); } println!( diff --git a/crates/cfc-cli/tests/cli_e2e.rs b/crates/cfc-cli/tests/cli_e2e.rs index a9ca0c4..b633b5e 100644 --- a/crates/cfc-cli/tests/cli_e2e.rs +++ b/crates/cfc-cli/tests/cli_e2e.rs @@ -1075,6 +1075,49 @@ async fn removing_a_bundle_preserves_a_manual_rule_with_the_same_name() { std::fs::remove_dir_all(dir).unwrap(); } +// The GUI editor keeps a rule's id, so a bundle rule turned into a Deny still +// carries the bundle's id. Removing the bundle must not delete that Deny. +#[tokio::test] +async fn removing_a_bundle_keeps_its_rule_edited_into_a_deny() { + use sha2::Digest as _; + let digest = sha2::Sha256::digest("colony-firewall-bundle\0inbound\0inbound-ssh-lan"); + let id = uuid::Uuid::from_bytes(digest[..16].try_into().unwrap()).to_string(); + let dir = std::env::temp_dir().join(format!("cfc-test-{}", uuid::Uuid::new_v4())); + std::fs::create_dir(&dir).unwrap(); + let socket = dir.join("cli.sock"); + let mut edited = stub_rule(&id, "inbound-ssh-lan"); + edited.action = pb::Action::Deny as i32; + let fake = FakeDaemon::default(); + fake.existing.lock().unwrap().push(edited); + let calls = fake.calls.clone(); + let server = serve(socket.clone(), fake).await; + let socket_arg = socket.to_string_lossy().into_owned(); + let out = tokio::task::spawn_blocking(move || { + run_cli( + &[ + "--socket", + &socket_arg, + "rules", + "bundle", + "remove", + "inbound", + ], + Duration::from_secs(5), + ) + }) + .await + .unwrap(); + assert!( + out.status.success(), + "{}", + String::from_utf8_lossy(&out.stderr) + ); + assert!(String::from_utf8_lossy(&out.stdout).contains("kept")); + assert!(calls.lock().unwrap().is_empty()); + server.abort(); + std::fs::remove_dir_all(dir).unwrap(); +} + #[tokio::test(flavor = "multi_thread", worker_threads = 2)] async fn remove_by_full_id_reaches_a_rule_the_daemon_does_not_list() { // A quarantined row is not in ListRules; the journal names its id. From dfca4c77b69aee4ffd46bf2acfbe59c65a83bede Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:36:44 +0200 Subject: [PATCH 063/125] fix(rules): validate OpenSnitch addresses and ports per file dest.ip was concatenated with /32 or /128 (10.0.0.0/8 became 10.0.0.0/8/32), dest.network was passed through and dest.port parsed as u32. Each passed conversion and the daemon then refused the whole batch without naming the file. They are now parsed as an IP address, a CIDR network and a u16, so a bad value skips its own file with a reason. --- crates/cfc-cli/src/rules.rs | 40 ++++++++++++++++++++++++++++++------- 1 file changed, 33 insertions(+), 7 deletions(-) diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index e11d70f..3a8c2d2 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -1313,21 +1313,27 @@ fn apply_simple(s: &OsnSimple, scope: &mut proto::RuleScope) -> anyhow::Result<( "dest.host" | "dest.domain" => anyhow::bail!( "hostname policy is unsupported; use an explicit numeric dest.ip or dest.network" ), + // Parsed here, as the daemon will: one bad value must skip its own + // file with a reason, not fail the whole batch at the daemon. "dest.ip" => { + let ip = s.data.parse::().map_err(|_| { + anyhow::anyhow!("operand `dest.ip`: `{}` is not an IP address", s.data) + })?; // single IP -> /32 or /128 - let net = if s.data.contains(':') { - format!("{}/128", s.data) - } else { - format!("{}/32", s.data) - }; + let net = ipnet::IpNet::from(ip).to_string(); set_once("dest.ip", &mut scope.dst_net, net)?; } - "dest.network" => set_once("dest.network", &mut scope.dst_net, s.data.clone())?, + "dest.network" => { + let net = s.data.parse::().map_err(|_| { + anyhow::anyhow!("operand `dest.network`: `{}` is not a CIDR network", s.data) + })?; + set_once("dest.network", &mut scope.dst_net, net.to_string())?; + } "dest.port" => { if scope.has_dst_port { anyhow::bail!("operand `dest.port` appears more than once"); } - scope.dst_port = s.data.parse::().map_err(|_| { + scope.dst_port = s.data.parse::().map(u32::from).map_err(|_| { anyhow::anyhow!("operand `dest.port`: `{}` is not a port number", s.data) })?; scope.has_dst_port = true; @@ -3230,6 +3236,26 @@ mod opensnitch_tests { assert_eq!(r.scope.unwrap().dst_net, "2001:db8::1/128"); } + // These used to pass conversion and then fail the whole batch at the + // daemon, so one bad file stopped the import without being named. + #[test] + fn malformed_addresses_and_ports_fail_their_own_file() { + for (operand, data) in [ + ("dest.ip", "10.0.0.0/8"), + ("dest.ip", "example.org"), + ("dest.network", "10.0.0.0/33"), + ("dest.network", "10.0.0.1"), + ("dest.port", "70000"), + ] { + let source = format!( + r#"{{"action":"deny","duration":"always","operator":{{"type":"simple","operand":"{operand}","data":"{data}"}}}}"# + ); + assert!(parse(&source).is_err(), "{operand} {data}"); + } + let r = parse(r#"{"action":"deny","operator":{"type":"simple","operand":"dest.network","data":"10.0.0.0/8"}}"#).unwrap(); + assert_eq!(r.scope.unwrap().dst_net, "10.0.0.0/8"); + } + #[test] fn empty_rule_rejected() { // No operator at all -> no convertible predicates -> error. From dce4ec088c419f36d920d49c2e72e56514a07419 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:37:26 +0200 Subject: [PATCH 064/125] fix(rules)!: refuse a partial OpenSnitch import unless asked An additive import-opensnitch skipped every rule it could not convert (hostname and regexp rules among them), applied the rest and exited 0, so dropping a narrow deny beside a broad allow imported a wider policy than the source with only a skip count to show for it. Any unconvertible rule now stops the import before anything changes; --skip-unconvertible imports the rest, and --replace still refuses outright. BREAKING CHANGE: import-opensnitch fails when a source rule cannot be converted; pass --skip-unconvertible for the old additive behaviour. --- README.md | 3 ++- crates/cfc-cli/src/main.rs | 16 +++++++++--- crates/cfc-cli/src/rules.rs | 18 ++++++++++--- crates/cfc-cli/tests/cli_e2e.rs | 46 +++++++++++++++++++++++++++++++++ 4 files changed, 76 insertions(+), 7 deletions(-) diff --git a/README.md b/README.md index 40cd181..1ab8a51 100644 --- a/README.md +++ b/README.md @@ -419,7 +419,8 @@ cfc resume # Back up rules cfc rules export --out rules.json -# Migrate from an existing opensnitch install +# Migrate from an existing opensnitch install. A rule with no equivalent +# here (hostname, regexp) stops it; --skip-unconvertible imports the rest. cfc rules import-opensnitch /etc/opensnitchd/rules ``` diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index ea651ab..6887eb1 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -171,12 +171,20 @@ enum RulesCmd { replace: bool, }, /// Import rules from an opensnitch rules directory or single JSON file. + /// + /// Rules with no faithful equivalent here (hostnames, regexps, unknown + /// operands) cannot be converted. By default one of them stops the import + /// before anything changes, because dropping a narrow deny while importing + /// a broad allow imports a wider policy than the source. ImportOpensnitch { /// Path to opensnitch rules dir (e.g. /etc/opensnitchd/rules) or a single .json. path: PathBuf, /// Replace mode: make the rule set match the source in one atomic batch. Every source rule must validate before anything changes. #[arg(long)] replace: bool, + /// Import the rules that convert and skip, with a reason, those that do not. + #[arg(long, conflicts_with = "replace")] + skip_unconvertible: bool, }, /// Install a small set of sensible starter rules: system DNS, NTP /// (timesyncd/chrony), DHCP clients (dhcpcd/NetworkManager/networkd), @@ -362,9 +370,11 @@ async fn dispatch( RulesCmd::Import { file, replace } => { rules::import(client, file, replace, format).await } - RulesCmd::ImportOpensnitch { path, replace } => { - rules::import_opensnitch(client, path, replace, format).await - } + RulesCmd::ImportOpensnitch { + path, + replace, + skip_unconvertible, + } => rules::import_opensnitch(client, path, replace, skip_unconvertible, format).await, RulesCmd::Bundle { cmd } => match cmd { BundleCmd::List => rules::bundle_list(client, format).await, BundleCmd::Add { name, dry_run } => { diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 3a8c2d2..a936d46 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -1101,6 +1101,7 @@ pub async fn import_opensnitch( client: &mut Client, path: PathBuf, replace: bool, + skip_unconvertible: bool, format: OutputFormat, ) -> CliResult { let mut files: Vec = if path.is_dir() { @@ -1124,7 +1125,9 @@ pub async fn import_opensnitch( ))); } - // Unsupported rules may be skipped only for additive imports. + // Unsupported rules may be skipped only when asked, and never in replace + // mode: a skipped deny next to an imported allow is a wider policy than + // the source, and the skip count alone did not say so. let mut pending = Vec::new(); let mut skipped = 0u32; for file in &files { @@ -1155,6 +1158,14 @@ pub async fn import_opensnitch( if replace && skipped > 0 { return Err(anyhow::anyhow!("refusing --replace: {skipped} source rules could not be converted; nothing was changed").into()); } + if skipped > 0 && !skip_unconvertible { + return Err(anyhow::anyhow!( + "refusing to import: {skipped} source rules could not be converted, and \ + importing the rest would drop their restrictions; nothing was changed. \ + Rewrite them, or pass --skip-unconvertible to import the rest anyway" + ) + .into()); + } if replace && pending.is_empty() { return Err(anyhow::anyhow!( "refusing --replace: none of the {} file(s) converted, so this would \ @@ -1185,8 +1196,9 @@ fn convert_opensnitch(file: &std::path::Path, osn: OsnRule) -> anyhow::Result proto::Action::Allow, "deny" | "drop" => proto::Action::Deny, diff --git a/crates/cfc-cli/tests/cli_e2e.rs b/crates/cfc-cli/tests/cli_e2e.rs index b633b5e..05b1c01 100644 --- a/crates/cfc-cli/tests/cli_e2e.rs +++ b/crates/cfc-cli/tests/cli_e2e.rs @@ -1184,3 +1184,49 @@ async fn a_partial_opensnitch_replace_changes_nothing() { server.abort(); std::fs::remove_dir_all(dir).unwrap(); } + +// An additive import used to drop the unconvertible deny, apply the allow and +// exit 0, so the imported policy was wider than the source. +#[tokio::test] +async fn an_opensnitch_import_with_unconvertible_rules_needs_consent() { + let dir = std::env::temp_dir().join(format!("cfc-test-{}", uuid::Uuid::new_v4())); + std::fs::create_dir(&dir).unwrap(); + let source = dir.join("source"); + std::fs::create_dir(&source).unwrap(); + std::fs::write(source.join("allow.json"), r#"{"name":"scoped","action":"allow","duration":"always","operator":{"type":"simple","operand":"dest.port","data":"443"}}"#).unwrap(); + std::fs::write(source.join("deny.json"), r#"{"name":"tracker","action":"deny","duration":"always","operator":{"type":"simple","operand":"dest.host","data":"tracker.example"}}"#).unwrap(); + let socket = dir.join("cli.sock"); + let fake = FakeDaemon::default(); + let calls = fake.calls.clone(); + let server = serve(socket.clone(), fake).await; + let socket_arg = socket.to_string_lossy().into_owned(); + let source_arg = source.to_string_lossy().into_owned(); + let run = |extra: &'static [&'static str]| { + let socket_arg = socket_arg.clone(); + let source_arg = source_arg.clone(); + tokio::task::spawn_blocking(move || { + let mut args = vec![ + "--socket", + &socket_arg, + "rules", + "import-opensnitch", + &source_arg, + ]; + args.extend_from_slice(extra); + run_cli(&args, Duration::from_secs(5)) + }) + }; + let out = run(&[]).await.unwrap(); + assert!(!out.status.success()); + assert!(String::from_utf8_lossy(&out.stderr).contains("nothing was changed")); + assert!(calls.lock().unwrap().is_empty()); + let out = run(&["--skip-unconvertible"]).await.unwrap(); + assert!( + out.status.success(), + "{}", + String::from_utf8_lossy(&out.stderr) + ); + assert_eq!(calls.lock().unwrap().len(), 1); + server.abort(); + std::fs::remove_dir_all(dir).unwrap(); +} From cce9987885cb5de81efe3e52b16144389e18d84c Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:38:27 +0200 Subject: [PATCH 065/125] fix(client): show protocol, uid and a pinned digest in rule summaries rule_summary, the only rendering in the GUI rules list and in cfc rules list, left these three predicates out, so a deny scoped to uid 1000 and udp/53 read as DNS blocked for everyone and a hash-pinned allow looked path-only. They now follow the port. The GUI's saved-rule line no longer appends the protocol itself. --- crates/cfc-client/src/convert.rs | 39 ++++++++++++++++++++++++++++---- crates/cfc-ui/src/main.rs | 8 +------ 2 files changed, 36 insertions(+), 11 deletions(-) diff --git a/crates/cfc-client/src/convert.rs b/crates/cfc-client/src/convert.rs index 7bbf96d..c6f7ae8 100644 --- a/crates/cfc-client/src/convert.rs +++ b/crates/cfc-client/src/convert.rs @@ -185,10 +185,26 @@ pub fn rule_summary(r: &pb::RuleInfo) -> String { } }) .unwrap_or_else(|| "*".into()); - let port = scope - .and_then(|s| s.has_dst_port.then_some(s.dst_port)) - .map(|p| format!(":{p}")) - .unwrap_or_default(); + // Protocol, uid and a pinned digest narrow the rule too; left out, a + // `deny uid 1000 udp/53` read as DNS blocked for everyone. + let port = scope.map_or_else(String::new, |s| { + let mut port = if s.has_dst_port { + format!(":{}", s.dst_port) + } else { + String::new() + }; + if s.has_protocol { + port.push(' '); + port.push_str(protocol_label(s.protocol)); + } + if s.has_uid { + port.push_str(&format!(" uid={}", s.uid)); + } + if !s.exe_sha256.is_empty() { + port.push_str(" [pinned]"); + } + port + }); let exe = scope .and_then(|s| { if s.exe_path.is_empty() { @@ -373,6 +389,21 @@ mod tests { assert!(s.contains(":22"), "{s}"); } + #[test] + fn protocol_uid_and_pinned_digest_are_not_hidden() { + let s = rule_summary(&rule(pb::RuleScope { + exe_sha256: "ab".repeat(32), + uid: 1000, + has_uid: true, + protocol: pb::Protocol::Udp as i32, + has_protocol: true, + dst_port: 53, + has_dst_port: true, + ..Default::default() + })); + assert_eq!(s, "allow * -> *:53 udp uid=1000 [pinned]"); + } + #[test] fn an_outbound_source_restriction_is_not_hidden() { let s = rule_summary(&rule(pb::RuleScope { diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index a49bcfa..1c8321a 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -1720,15 +1720,9 @@ fn saved_rule_line(rule: &proto::RuleInfo) -> String { .split_whitespace() .collect::>() .join(" "); - let protocol = rule - .scope - .as_ref() - .filter(|s| s.has_protocol) - .map(|s| format!(" {}", convert::protocol_label(s.protocol))) - .unwrap_or_default(); let disabled = if rule.enabled { "" } else { ", disabled" }; format!( - "rule saved: {summary}{protocol}, {}{disabled}", + "rule saved: {summary}, {}{disabled}", convert::rule_duration_label(rule) ) } From bd2a524e2c4e7bd59bae26721d5639359c6bc0b9 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:39:17 +0200 Subject: [PATCH 066/125] fix(client): unsubscribe as soon as a resilient stream is dropped pump_once noticed a dropped consumer only when the next event failed to send. The GUI drops its prompt subscription on every disconnect, so after a status timeout with the daemon still up, the orphaned subscription kept counting as a listener and the next prompt was held for the full timeout and then lost. The pump now also waits on the channel closing. --- crates/cfc-cli/tests/cli_e2e.rs | 39 ++++++++++++++++++++++++++++++++- crates/cfc-client/src/lib.rs | 9 +++++++- 2 files changed, 46 insertions(+), 2 deletions(-) diff --git a/crates/cfc-cli/tests/cli_e2e.rs b/crates/cfc-cli/tests/cli_e2e.rs index 05b1c01..d2dbf90 100644 --- a/crates/cfc-cli/tests/cli_e2e.rs +++ b/crates/cfc-cli/tests/cli_e2e.rs @@ -36,6 +36,8 @@ struct FakeDaemon { /// tested. Without this the fake daemon has no failure mode at all and the /// central atomicity claim goes unexercised. upsert_fails_for: Arc>>, + /// Set when a prompt subscriber goes away. + unsubscribed: Arc, } /// One mutation seen by the fake daemon. @@ -55,6 +57,7 @@ impl Firewall for FakeDaemon { _req: Request, ) -> Result, Status> { let (tx, rx) = tokio::sync::mpsc::channel(4); + let unsubscribed = self.unsubscribed.clone(); tokio::spawn(async move { let ev = pb::PromptEvent { prompt_id: "42".into(), @@ -89,7 +92,10 @@ impl Firewall for FakeDaemon { let _ = tx.send(Ok(ev)).await; // Hold the stream open; the CLI is expected to leave on its own // once --count is satisfied. - tokio::time::sleep(Duration::from_secs(30)).await; + tokio::select! { + _ = tx.closed() => unsubscribed.store(true, std::sync::atomic::Ordering::SeqCst), + _ = tokio::time::sleep(Duration::from_secs(30)) => {} + } }); Ok(Response::new(ReceiverStream::new(rx))) } @@ -1230,3 +1236,34 @@ async fn an_opensnitch_import_with_unconvertible_rules_needs_consent() { server.abort(); std::fs::remove_dir_all(dir).unwrap(); } + +// A front end that drops its subscription (the GUI does on every disconnect) +// must unsubscribe at once. The pump used to notice only when the next event +// failed to send, so the daemon held that prompt for an absent listener. +#[tokio::test(flavor = "multi_thread", worker_threads = 2)] +async fn a_dropped_prompt_stream_unsubscribes_at_once() { + use futures::StreamExt as _; + let socket = socket_path("unsubscribe"); + let fake = FakeDaemon::default(); + let unsubscribed = fake.unsubscribed.clone(); + let server = serve(socket.clone(), fake).await; + let mut stream = Box::pin(cfc_client::stream_prompts_resilient(&socket, "test".into())); + loop { + match tokio::time::timeout(Duration::from_secs(5), stream.next()).await { + Ok(Some(cfc_client::StreamItem::Event(_))) => break, + Ok(Some(_)) => continue, + other => panic!("no prompt arrived: {other:?}"), + } + } + drop(stream); + let deadline = Instant::now() + Duration::from_secs(5); + while !unsubscribed.load(std::sync::atomic::Ordering::SeqCst) { + assert!( + Instant::now() < deadline, + "the subscription outlived its consumer" + ); + tokio::time::sleep(Duration::from_millis(20)).await; + } + server.abort(); + let _ = std::fs::remove_file(socket); +} diff --git a/crates/cfc-client/src/lib.rs b/crates/cfc-client/src/lib.rs index e378d9e..69acba8 100644 --- a/crates/cfc-client/src/lib.rs +++ b/crates/cfc-client/src/lib.rs @@ -415,7 +415,14 @@ async fn pump_once( return PumpOutcome::ConsumerGone; } loop { - match stream.message().await { + // Watch the consumer too: a dropped stream must unsubscribe now, not + // when the next event fails to send. Until then the daemon counts it + // as a listener and holds a prompt for it for the full timeout. + let message = tokio::select! { + _ = tx.closed() => return PumpOutcome::ConsumerGone, + message = stream.message() => message, + }; + match message { Ok(Some(ev)) => { if tx.send(StreamItem::Event(ev)).await.is_err() { return PumpOutcome::ConsumerGone; From 2067bf9b61fc4b034b207795622108adff61468b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:40:18 +0200 Subject: [PATCH 067/125] fix(rules): keep a timed rule's deadline through export and import The export carried a timed rule's duration but not when it started, and import sent no creation time, so the daemon started the full lifetime again: a one-hour Allow in a backup was valid for another hour on every restore, even long after it had expired. Exports now carry expires_at_unix_ms for timed rules, and import sends the matching past creation time, which the daemon already keeps. --- crates/cfc-cli/src/rules.rs | 42 ++++++++++++++++++++++++++++++++++++- 1 file changed, 41 insertions(+), 1 deletion(-) diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index a936d46..8cba078 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -733,6 +733,10 @@ pub struct ExportedRule { pub duration: String, #[serde(default)] pub duration_seconds: u32, + /// When a `seconds` rule runs out. Without it a restored backup started + /// the full lifetime again, so an expired temporary Allow came back. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub expires_at_unix_ms: Option, #[serde(default)] pub scope: ExportedScope, } @@ -853,6 +857,19 @@ impl ExportedRule { "rule `{name}`: duration_seconds requires duration `seconds`" )); } + // The daemon keeps a past creation time, so the rule ends when the + // exported one did; one already past is stored expired. + let created_at_unix_ms = match self.expires_at_unix_ms { + None => 0, + Some(_) if duration != proto::Duration::Seconds => { + return Err(format!( + "rule `{name}`: expires_at_unix_ms requires duration `seconds`" + )) + } + Some(at) => at + .saturating_sub(i64::from(self.duration_seconds) * 1000) + .max(1), + }; let direction_idx = match self.scope.direction.as_deref() { None => None, Some(d) => Some(match d.to_ascii_lowercase().as_str() { @@ -1006,7 +1023,7 @@ impl ExportedRule { duration: duration as i32, duration_seconds: self.duration_seconds, scope: Some(scope), - created_at_unix_ms: 0, + created_at_unix_ms, hit_count: 0, }) } @@ -1028,6 +1045,9 @@ pub fn exported_rule(r: &proto::RuleInfo) -> ExportedRule { action: convert::action_label(r.action).to_string(), duration: convert::duration_label(r.duration).to_string(), duration_seconds: r.duration_seconds, + expires_at_unix_ms: (r.duration == proto::Duration::Seconds as i32 + && r.created_at_unix_ms > 0) + .then(|| r.created_at_unix_ms + i64::from(r.duration_seconds) * 1000), scope: ExportedScope { exe_path: scope.and_then(|s| opt_string(&s.exe_path)), exe_sha256: scope.and_then(|s| opt_string(&s.exe_sha256)), @@ -2493,6 +2513,25 @@ mod json_tests { assert!(rule.try_into_proto().is_err()); } + // A restored backup used to start a timed Allow's lifetime again. + #[test] + fn a_timed_rule_keeps_its_deadline_through_export_and_import() { + let mut rule = exported("allow").try_into_proto().unwrap(); + rule.duration = proto::Duration::Seconds as i32; + rule.duration_seconds = 3600; + rule.created_at_unix_ms = 1_000_000; + let back = exported_rule(&rule); + assert_eq!(back.expires_at_unix_ms, Some(4_600_000)); + assert_eq!(back.try_into_proto().unwrap().created_at_unix_ms, 1_000_000); + let mut always = exported("allow"); + always.expires_at_unix_ms = Some(4_600_000); + assert!(always.try_into_proto().is_err()); + assert_eq!( + exported_rule(&exported("allow").try_into_proto().unwrap()).expires_at_unix_ms, + None + ); + } + fn exported(action: &str) -> ExportedRule { ExportedRule { id: String::new(), @@ -2501,6 +2540,7 @@ mod json_tests { action: action.into(), duration: "always".into(), duration_seconds: 0, + expires_at_unix_ms: None, scope: ExportedScope { exe_path: Some("/usr/bin/curl".into()), exe_sha256: None, From f14a48df996a990b07cdf6366071d7da17448733 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:41:01 +0200 Subject: [PATCH 068/125] fix(client): escape the backslash in display-safe strings display_safe wrote control characters as escapes but left the backslash alone, so an argument containing a literal backslash-n rendered exactly like an escaped newline. The backslash is now escaped as well. The places that escaped an already safe string a second time (rule summaries in cfc rules list and show, the app column of cfc live, systemctl errors in confinement) now escape once. --- crates/cfc-cli/src/confinement/mod.rs | 3 ++- crates/cfc-cli/src/live.rs | 3 ++- crates/cfc-cli/src/output.rs | 11 +++++++---- crates/cfc-cli/src/rules.rs | 8 +++----- crates/cfc-client/src/convert.rs | 6 +++++- 5 files changed, 19 insertions(+), 12 deletions(-) diff --git a/crates/cfc-cli/src/confinement/mod.rs b/crates/cfc-cli/src/confinement/mod.rs index 706010f..30ec639 100644 --- a/crates/cfc-cli/src/confinement/mod.rs +++ b/crates/cfc-cli/src/confinement/mod.rs @@ -217,8 +217,9 @@ fn manager(args: &[&str]) -> Result { .context("calling systemd")?; ensure!( result.status.success(), + // main() escapes the whole error once. "systemd rejected the application operation: {}", - crate::output::terminal_safe(&String::from_utf8_lossy(&result.stderr)) + String::from_utf8_lossy(&result.stderr) ); Ok(String::from_utf8(result.stdout)?.trim().to_owned()) } diff --git a/crates/cfc-cli/src/live.rs b/crates/cfc-cli/src/live.rs index 99b09da..39fa564 100644 --- a/crates/cfc-cli/src/live.rs +++ b/crates/cfc-cli/src/live.rs @@ -185,7 +185,8 @@ fn print_row(ev: &proto::ConnectionEvent, conn: &proto::ConnectionInfo) { format!("{time:<8}").if_supports_color(Stdout, |s| s.dimmed()), convert::protocol_label(conn.protocol), pid, - output::truncate(&app, 18), + // process_display already escaped it. + output::clip(&app, 18), src, dst.if_supports_color(Stdout, |s| s.cyan()), verdict, diff --git a/crates/cfc-cli/src/output.rs b/crates/cfc-cli/src/output.rs index 99cf72a..7ebe60b 100644 --- a/crates/cfc-cli/src/output.rs +++ b/crates/cfc-cli/src/output.rs @@ -69,11 +69,14 @@ pub fn rfc3339(unix_ms: i64) -> Option { /// Render untrusted values as one terminal-safe line. JSON keeps raw values. pub use cfc_client::convert::display_safe as terminal_safe; -/// Clips a cell to `width` characters, marking the cut with `~` so a -/// truncated path is never mistaken for a real one. +/// Escapes an untrusted cell, then clips it like [`clip`]. pub fn truncate(s: &str, width: usize) -> String { - let escaped = terminal_safe(s); - let s = escaped.as_str(); + clip(&terminal_safe(s), width) +} + +/// Clips an already display-safe cell to `width` characters, marking the cut +/// with `~` so a truncated path is never mistaken for a real one. +pub fn clip(s: &str, width: usize) -> String { if s.chars().count() <= width || width == 0 { return s.to_string(); } diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 8cba078..6be33b4 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -139,7 +139,8 @@ pub async fn list(client: &mut Client, format: OutputFormat) -> CliResult { convert::rule_duration_label(r), r.hit_count, output::truncate(&r.name, name_w), - output::terminal_safe(&convert::rule_summary(r)) + // Already display-safe. + convert::rule_summary(r) ); } Ok(()) @@ -172,10 +173,7 @@ pub async fn show(client: &mut Client, needle: &str, format: OutputFormat) -> Cl } ); println!("hits {}", rule.hit_count); - println!( - "summary {}", - output::terminal_safe(&convert::rule_summary(&rule)) - ); + println!("summary {}", convert::rule_summary(&rule)); println!("scope:"); // Not dashed when unset: an absent direction is not "unconstrained", it // means outbound (the matcher's contract - unset kept the meaning every diff --git a/crates/cfc-client/src/convert.rs b/crates/cfc-client/src/convert.rs index c6f7ae8..1460e5b 100644 --- a/crates/cfc-client/src/convert.rs +++ b/crates/cfc-client/src/convert.rs @@ -10,12 +10,15 @@ use cfc_proto::v1 as pb; /// chosen by the program being judged, or by whoever named its file, and DNS /// names by whoever answers the query: a U+202E in a directory name reverses /// the rest of a Path row, and an embedded newline adds a fake line to a -/// prompt. Only for display; rules and copies keep the raw value. +/// prompt. The backslash is escaped too, so a literal `\n` in a name cannot +/// pass for an escaped newline. Escape once: a second pass doubles every +/// backslash. Only for display; rules and copies keep the raw value. pub fn display_safe(value: &str) -> String { value .chars() .flat_map(|character| { if character.is_control() + || character == '\\' || matches!(character, '\u{202a}'..='\u{202e}' | '\u{2066}'..='\u{2069}') { character.escape_default().collect::>() @@ -271,6 +274,7 @@ mod tests { #[test] fn display_safe_escapes_controls_and_bidi_only() { assert_eq!(display_safe("line\nnext\tcell"), "line\\nnext\\tcell"); + assert_eq!(display_safe("x\\n"), "x\\\\n"); assert!(display_safe("\u{202e}\u{2066}").is_ascii()); assert_eq!(display_safe("/usr/bin/caf\u{e9}"), "/usr/bin/caf\u{e9}"); let p = pb::ProcessInfo { From 4fe5333995ea6d8dc2e0cdf132f14df8eb8ee5e4 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:41:28 +0200 Subject: [PATCH 069/125] docs(changelog): list the CLI and confinement fixes in this bundle --- CHANGELOG.md | 53 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 53 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index ec9467d..f55fa35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -28,9 +28,43 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). a rule the editor refuses is also reported in the footer. - GUI: a prompt arriving while others are pending no longer switches to the Prompts tab; only the first one does, and raises the window. +- `cfc rules import-opensnitch` stops before changing anything when a source + rule cannot be converted (hostname and regexp rules among them), because + dropping a narrow deny next to a broad allow imported a wider policy than + the source with only a skip count to show for it. `--skip-unconvertible` + imports the rest, naming each skipped file. **Breaking** for scripts that + relied on the old additive behaviour. +- Bundles name the binary that connects. The rules for Firefox on Arch, git, + cargo under rustup and apt pinned a launcher or front end that never shows + up as the connecting executable, so they never fired. `web` and `dev` drop + Epiphany, npm and pip, whose traffic comes from a shared WebKit helper or + an interpreter. Hosts that installed `web`, `dev` or `updates` before keep + the old rules: `cfc rules bundle remove NAME` then `bundle add NAME` + replaces them, and `bundle add` names each one that pins an old path. +- Rule summaries in the GUI and `cfc rules list` show the protocol, a uid and + `[pinned]` for a hash-pinned rule, so a scoped rule no longer reads as + global. +- `cfc rules export` writes `expires_at_unix_ms` for timed rules, and import + keeps that deadline instead of starting the full lifetime again. Older + versions refuse an export that contains the field. +- Confinement: a refused launch exits 125 and its reason is in + `journalctl -u cfc-app-ID.service`; it used to be discarded and reported + as the application's status 1. ### Security +- `cfc prompts`: keys typed while no prompt was shown, such as an answer + typed just as a prompt expired, answered the next prompt as soon as it was + printed, and an arrow key skipped one prompt and left `A` or `D` to answer + the next. On a terminal, pending input is now discarded before each prompt. +- The CLI printed the daemon's reason for not saving a rule raw, and that + reason can quote an executable path a local user named, escape sequences + included. The client now escapes it for every front end, and escapes the + backslash in every escaped string so a literal `\n` cannot pass for an + escaped newline. +- Confinement: on kernels 6.17 to 7.1 the root gate's `BPF_PROG_QUERY` + attribute was 32 bytes and the kernel wrote 8 bytes past it on the stack. + The attribute now has its full size. - GUI: `A`, `D`, `Shift+A` and `Shift+D` answered the newest prompt, the bottom card and often off-screen, from any tab and with Ctrl, Alt or Super held, so `Shift+A` on the card being read could write an always-allow rule @@ -156,6 +190,25 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). so later prompts only reached the overflow bubble, which cannot answer them. Slots are freed once their prompt's deadline has passed, and the stale bubbles are closed. +- Confinement refused every launch while the daemon's DNS observer, on by + default, was attached at the cgroup root, because the gate required the + unit's effective filters to be exactly its own pair. Programs inherited + from ancestors are now accepted; the unit's own pair must still be exact. +- Confinement: Ctrl-C while systemctl ran, a closed terminal or a dropped + SSH session killed the launcher and left the tree running with its + grants. SIGINT, SIGHUP, SIGQUIT and SIGTERM now stop the tree at any + point, and its identity is printed before it starts. +- `cfc rules bootstrap-defaults` and `bundle add` failed on hosts seeded + before 0.7.0, calling the bundle's own rules outside it. An identical + same-named rule now counts as present; a different one still stops the + command. +- `cfc rules bundle remove` deleted a bundle rule the user had edited into a + deny. It now keeps any of its rules that is no longer an allow. +- OpenSnitch import passed `dest.ip` networks (`10.0.0.0/8/32`), bad CIDRs + and ports above 65535 to the daemon, which refused the whole import + without naming the file. They now fail their own file. +- The GUI's prompt subscription stayed open after the GUI dropped it, so the + daemon held the next prompt for an absent listener until it timed out. ## [0.7.0] - 2026-09-30 From e87a8c0b6e18a674e50afa62a78413de0f36ff5b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:57:02 +0200 Subject: [PATCH 070/125] fix(packaging): keep nft unit changes away from the network managers The documented way to lift filtering was a plain stop, which propagates through the managers' Requires= and stops NetworkManager or systemd-networkd; RPM erase did the same through %systemd_preun's --no-reload stop. Docs, unit comments and the pacman message now say `systemctl disable --now`, and %preun disables with a reload before the macro runs. Restart is documented as unsupported for the nft units. Upgrades reenabled colony-firewalld, whose Also= re-enabled an nft unit the admin had disabled; the loops now reenable only enabled nft units. TROUBLESHOOTING no longer recommends a same-named table in /etc/nftables.conf, which the unit replaces at boot, daemon start and upgrade; local changes go in a copy loaded through a drop-in. The pacman message seeds the starter rules with sudo right after enabling. --- README.md | 5 +- docs/TROUBLESHOOTING.md | 101 +++++++++++++++----- packaging/rpm/colony-firewall-control.spec | 11 ++- pkg/colony-firewall-control.install | 20 ++-- pkg/colony.json | 2 +- scripts/check-startup-protection.py | 37 +++++++ systemd/colony-firewall-nft-inbound.service | 8 +- systemd/colony-firewall-nft.service | 11 ++- systemd/nftables-snippet.conf | 3 +- 9 files changed, 155 insertions(+), 43 deletions(-) diff --git a/README.md b/README.md index 1ab8a51..d61d020 100644 --- a/README.md +++ b/README.md @@ -264,8 +264,9 @@ distribution's `nftables.service` when it is in the same boot transaction. Enabling enforcement creates native requirements from those two network managers: a failed nft load blocks their startup. A failed daemon start leaves the loaded tables dropping new flows. The daemon also requires the outbound -table before initialization. Tables survive daemon stops and restarts; stop -the nft unit explicitly to remove its table. Inbound stays opt-in. Its lockout +table before initialization. Tables survive daemon stops and restarts; to +remove one, `sudo systemctl disable --now` its nft unit (a plain stop also +stops the network managers that require it). Inbound stays opt-in. Its lockout guard reads saved SQLite rules without a running daemon. With inbound enabled, ping and other ICMP requests need a rule like any other inbound flow, so allow your monitoring hosts: diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 4c7fd06..56a5140 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -9,14 +9,30 @@ before enabling enforcement on any machine you reach over SSH. Once loaded, both nft tables survive daemon stops and restarts. With no queue listener, new tracked flows drop; established and related traffic retains its -authorization. To intentionally remove filtering, stop the corresponding nft -unit. An active network manager that requires this unit also stops; disabling -enforcement removes its requirement for subsequent starts. Uninstall removes -both tables and Colony's pinned BPF directory. +authorization. To intentionally remove filtering, disable the corresponding +nft unit with `--now`: + +```sh +sudo systemctl disable --now colony-firewall-nft colony-firewalld +sudo systemctl disable --now colony-firewall-nft-inbound +``` + +`disable` removes the network managers' requirement on the unit and reloads +systemd before stopping it. A plain `systemctl stop` keeps that requirement +loaded, so an active NetworkManager or systemd-networkd stops with the unit, +and starting the manager again loads the table again. Disable the daemon too +if filtering should stay off: starting it loads the outbound table. Uninstall +removes both tables and Colony's pinned BPF directory. + +Never `restart` an nft unit, including from configuration management: it +deletes the table before loading it again, which leaves new flows unfiltered +for a moment, and it restarts the daemon and the network managers that +require the unit. `reload` replaces the table in one transaction. Package upgrades reload active nft units atomically. After a manual upgrade, -run `systemctl daemon-reload`, then `systemctl reenable colony-firewalld -colony-firewall-nft` (and the inbound unit only if already enabled), and +run `systemctl daemon-reload`, then `systemctl reenable colony-firewall-nft` +(and the inbound unit only if already enabled; never the daemon, whose +`Also=` would enable a disabled nft unit), and `systemctl reload colony-firewall-nft` (and the inbound unit if active) before relying on the new rules. Reenable installs the native network-manager requirements on existing deployments. A startup error saying the @@ -49,17 +65,15 @@ you cannot open a new one. Three layers of protection, use all of them the first time: -**1. Allow SSH above the queue rule.** Edit your copy of the snippet so -port 22 never reaches NFQUEUE at all: +**1. Allow SSH above the queue rule.** In a local copy of the snippet +(see [Changing the shipped ruleset](#changing-the-shipped-ruleset)), add one +line above the two queue rules of `chain output` so port 22 never reaches +NFQUEUE at all: ``` -table inet colony_firewall { - chain output { - type filter hook output priority 0; policy accept; tcp dport 22 accept + oifname "lo" ct state new queue num 0 bypass ct state new queue num 0 - } -} ``` (This exempts *outbound* SSH from filtering - for a remote machine you @@ -73,9 +87,9 @@ shell that survives your SSH session: sudo setsid sh -c 'sleep 300 && nft delete table inet colony_firewall' & ``` -Then apply the snippet. If you still have connectivity after testing, -cancel the timer (`sudo pkill -f 'nft delete table'`, or just re-apply the -snippet after the timer fires). If you locked yourself out, wait out the +Then enable enforcement. If you still have connectivity after testing, +cancel the timer (`sudo pkill -f 'nft delete table'`, or just +`sudo systemctl reload colony-firewall-nft` after the timer fires). If you locked yourself out, wait out the five minutes and the table deletes itself. **3. Know the console recovery.** From a local console, serial console, or @@ -115,8 +129,8 @@ If this errors with "No such file or directory", nothing is being enqueued - the daemon runs but enforces *nothing*, silently. This is the usual state after a reboot if you only ever applied the snippet manually with `nft -f`: nftables rules do not persist across reboots on their own. -Enable the companion unit (`colony-firewall-nft.service`) or merge the -snippet into `/etc/nftables.conf` so the rule comes back at boot. +Enable the companion unit (`colony-firewall-nft.service`) so the rule comes +back at boot. **Do the queue numbers match?** The snippet says `queue num 0`; the daemon binds the queue from `[nfqueue] queue_num` in `daemon.toml` (default 0). @@ -125,7 +139,8 @@ as a dead daemon. **The fail-open alternative.** If you would rather lose filtering than lose the network when the daemon is down, add the `bypass` keyword to the -final queue rule too: +final queue rule of a local copy of the snippet (see +[Changing the shipped ruleset](#changing-the-shipped-ruleset)): ``` ct state new queue num 0 bypass @@ -283,11 +298,11 @@ hashed, for instance) leaves new loopback flows, local DNS included, waiting in the same queue; once it fills they drop until the watchdog restarts the daemon, which takes up to about 90 seconds. -If you carry an older copy of the snippet in your own `/etc/nftables.conf`, -compare it with the shipped one: a copy without the loopback rule drops -every new loopback flow whenever the daemon is down, and one with an -explicit `oifname lo accept` skips the daemon for loopback entirely, so -loopback rules never apply. +If you load a local copy of the snippet, compare it with the shipped one +after every upgrade: a copy without the loopback rule drops every new +loopback flow whenever the daemon is down, and one with an explicit +`oifname lo accept` skips the daemon for loopback entirely, so loopback +rules never apply. Note the daemon already exempts its *own* reverse-DNS lookups internally (they would otherwise deadlock the queue); the loopback rule is about @@ -308,8 +323,42 @@ Two kinds of packet are settled in the kernel instead: explicit `notrack` rule touched (a busy DNS or NTP server's tuning, for instance). An `accept` in another table does not override this chain's `policy drop`. To keep such flows, load a local copy of the snippet with - an accept for them above the queue rules, and point the unit at it with - a drop-in. + an accept for them above the queue rules (see + [Changing the shipped ruleset](#changing-the-shipped-ruleset)). + +## Changing the shipped ruleset + +`colony-firewall-nft.service` loads +`/usr/share/colony-firewall/nftables-snippet.conf`, and that file starts by +deleting any `table inet colony_firewall` already loaded. A table of that name +from `/etc/nftables.conf` or a manual `nft -f` is therefore replaced at boot, +whenever the daemon starts (it requires the unit) and on every package +upgrade (which reloads the unit). Carry changes as a local copy that the unit +loads instead: + +```sh +sudo install -Dm644 /usr/share/colony-firewall/nftables-snippet.conf \ + /etc/colony-firewall/nftables-snippet.conf +sudoedit /etc/colony-firewall/nftables-snippet.conf +sudo systemctl edit colony-firewall-nft +``` + +and in the drop-in, override both commands (upgrades reload, so +`ExecReload=` matters as much as `ExecStart=`): + +``` +[Service] +ExecStart= +ExecStart=/usr/bin/nft -f /etc/colony-firewall/nftables-snippet.conf +ExecReload= +ExecReload=/usr/bin/nft -f /etc/colony-firewall/nftables-snippet.conf +``` + +Then `sudo systemctl reload colony-firewall-nft`. Keep the `add table` and +`delete table` lines at the top of the copy: they are what lets a reload +replace the table in one transaction. Upgrades do not touch the copy, so +compare it with the shipped file after each one. The inbound unit takes the +same drop-in with `nftables-inbound.conf`. ## Fail-open vs fail-closed matrix diff --git a/packaging/rpm/colony-firewall-control.spec b/packaging/rpm/colony-firewall-control.spec index 1adcfd8..6ddcd5c 100644 --- a/packaging/rpm/colony-firewall-control.spec +++ b/packaging/rpm/colony-firewall-control.spec @@ -172,8 +172,10 @@ cargo test --workspace --locked --no-fail-fast %sysusers_create_compat %{_sysusersdir}/colony-firewall.conf if [ $1 -gt 1 ]; then # Reload active nft units atomically before the daemon restart in postun. + # Reenable only enabled nft units: the daemon's Also= would enable an nft + # unit the admin disabled. systemctl daemon-reload - for unit in colony-firewalld.service colony-firewall-nft.service colony-firewall-nft-inbound.service; do + for unit in colony-firewall-nft.service colony-firewall-nft-inbound.service; do if systemctl is-enabled --quiet "$unit"; then systemctl reenable "$unit" || exit 1 fi @@ -185,6 +187,13 @@ if [ $1 -gt 1 ]; then fi %preun +if [ $1 -eq 0 ]; then + # Disable with a daemon reload before the stop. %systemd_preun stops with + # --no-reload, so the network managers' Requires= on the nft units is + # still loaded and the stop would take NetworkManager or systemd-networkd + # down with them, with nothing to start them again. + systemctl disable --now colony-firewall-nft-inbound.service colony-firewall-nft.service colony-firewalld.service >/dev/null 2>&1 || : +fi %systemd_preun colony-firewalld.service colony-firewall-nft.service colony-firewall-nft-inbound.service %postun diff --git a/pkg/colony-firewall-control.install b/pkg/colony-firewall-control.install index ffc1231..ea0fde7 100644 --- a/pkg/colony-firewall-control.install +++ b/pkg/colony-firewall-control.install @@ -8,18 +8,20 @@ post_install() { systemctl enable --now colony-firewalld colony-firewall-nft (colony-firewall-nft loads 'table inet colony_firewall'; it is fail-closed while loaded, except new loopback flows, and survives - daemon restarts/stops. - Stop colony-firewall-nft explicitly to remove filtering.) + daemon restarts/stops. To remove filtering, run + 'systemctl disable --now colony-firewalld colony-firewall-nft'; + a plain stop also stops NetworkManager/systemd-networkd.) - 2. Let your desktop user talk to the daemon socket: + 2. Seed sensible default rules right away (until they exist, unmatched + DHCP, DNS and NTP flows are denied): + sudo cfc rules bootstrap-defaults + + 3. Let your desktop user talk to the daemon socket: usermod -aG colony-firewall then log out/in. The 'colony-firewall' group is created by systemd-sysusers from /usr/lib/sysusers.d/colony-firewall.conf (applied automatically by the pacman sysusers hook). - 3. Seed sensible default rules: - cfc rules bootstrap-defaults - ==> The GUI autostarts for all desktop users via /etc/xdg/autostart/colony-firewall.desktop; per-user opt-out: copy it to ~/.config/autostart/ and set Hidden=true. @@ -31,8 +33,10 @@ post_upgrade() { # Refresh active units only; inbound remains opt-in. ExecReload applies # the new snippet atomically and removes old Fast Allow acceptance. systemctl daemon-reload - # Recreate native network-manager dependencies for enabled deployments. - for unit in colony-firewalld.service colony-firewall-nft.service colony-firewall-nft-inbound.service; do + # Recreate native network-manager dependencies for enabled nft units. Never + # reenable the daemon: its Also= would enable an nft unit the admin + # disabled, together with the network managers' Requires= on it. + for unit in colony-firewall-nft.service colony-firewall-nft-inbound.service; do if systemctl is-enabled --quiet "$unit"; then systemctl reenable "$unit" || return 1 fi diff --git a/pkg/colony.json b/pkg/colony.json index 455a26a..b89f219 100644 --- a/pkg/colony.json +++ b/pkg/colony.json @@ -31,7 +31,7 @@ "install -D -m 0644 cfc-ebpf.o /usr/lib/colony-firewall/cfc-ebpf.o", "systemd-sysusers", "systemctl daemon-reload", - "for unit in colony-firewalld.service colony-firewall-nft.service colony-firewall-nft-inbound.service; do if systemctl is-enabled --quiet \"$unit\"; then systemctl reenable \"$unit\" || exit 1; fi; done", + "for unit in colony-firewall-nft.service colony-firewall-nft-inbound.service; do if systemctl is-enabled --quiet \"$unit\"; then systemctl reenable \"$unit\" || exit 1; fi; done", "systemctl try-reload-or-restart colony-firewall-nft.service colony-firewall-nft-inbound.service || { echo \"Firewall rules could not be refreshed; reload colony-firewall-nft and inspect the journal before relying on filtering.\" >&2; exit 1; }" ], "preRemove": [ diff --git a/scripts/check-startup-protection.py b/scripts/check-startup-protection.py index 83a1c38..efc1b80 100644 --- a/scripts/check-startup-protection.py +++ b/scripts/check-startup-protection.py @@ -5,6 +5,7 @@ import json import os from pathlib import Path +import shlex import shutil import sqlite3 import subprocess @@ -28,6 +29,17 @@ def unit(name): assert "colony-firewalld.service" not in outbound["Unit"].get("After", "").split() assert "colony-firewalld.service" not in outbound["Unit"].get("Requires", "").split() +# %systemd_preun stops with --no-reload, under the managers' loaded Requires=, +# which would stop NetworkManager on erase. A reload-first disable must run +# before it. +spec = (ROOT / "packaging/rpm/colony-firewall-control.spec").read_text() +preun = spec.split("\n%preun\n", 1)[1].split("\n%postun", 1)[0] +before_macro = preun.split("\n%systemd_preun", 1)[0] +disable = [line for line in before_macro.splitlines() if "systemctl disable --now" in line] +assert disable and "--no-reload" not in disable[0] and all( + name in disable[0] for name in ("colony-firewall-nft.service", "colony-firewall-nft-inbound.service")), \ + "RPM erase must disable the nft units with a reload before %systemd_preun stops them" + with tempfile.TemporaryDirectory(prefix="cfc-startup-check-") as directory: stage = Path(directory) units = stage / "usr/lib/systemd/system" @@ -61,6 +73,31 @@ def unit(name): for manager in MANAGERS: assert not (stage / "etc/systemd/system" / (manager + ".requires") / name).is_symlink() + # An upgrade must not re-enable an nft unit the admin disabled: reenabling + # the daemon would bring it back through Also=. + shim = stage / "shim" + shim.mkdir() + (shim / "systemctl").write_text( + "#!/bin/sh\ncase \"$1\" in daemon-reload|try-reload-or-restart) exit 0 ;; esac\n" + f"exec {shlex.quote(shutil.which('systemctl'))} --root {shlex.quote(str(stage))} \"$@\"\n") + (shim / "systemctl").chmod(0o755) + subprocess.run(["systemctl", "--root", str(stage), "enable", "colony-firewalld.service"], + check=True, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) + subprocess.run(["systemctl", "--root", str(stage), "disable", "colony-firewall-nft.service"], + check=True, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) + subprocess.run(["bash", "-c", '. "$1" && post_upgrade >/dev/null', "post_upgrade", + str(ROOT / "pkg/colony-firewall-control.install")], + env=dict(os.environ, PATH=f"{shim}:{os.environ['PATH']}"), check=True) + for manager in MANAGERS: + assert not (stage / "etc/systemd/system" / (manager + ".requires") / + "colony-firewall-nft.service").is_symlink(), "upgrade re-enabled a disabled nft unit" + loop = "for unit in colony-firewall-nft.service colony-firewall-nft-inbound.service; do" + for recipe in ("pkg/colony-firewall-control.install", "packaging/rpm/colony-firewall-control.spec", + "pkg/colony.json"): + assert loop in (ROOT / recipe).read_text(), f"{recipe} must reenable only enabled nft units" + subprocess.run(["systemctl", "--root", str(stage), "disable", "colony-firewalld.service"], + check=True, stdout=subprocess.DEVNULL, stderr=subprocess.PIPE) + binaries = stage / "bin" binaries.mkdir() ss = binaries / "ss" diff --git a/systemd/colony-firewall-nft-inbound.service b/systemd/colony-firewall-nft-inbound.service index abe8d4c..b262123 100644 --- a/systemd/colony-firewall-nft-inbound.service +++ b/systemd/colony-firewall-nft-inbound.service @@ -7,9 +7,11 @@ # # systemctl enable --now colony-firewall-nft-inbound # -# Stopping this unit explicitly removes inbound filtering. Its table remains -# loaded across daemon restarts and stops; a missing queue listener drops new -# inbound flows until the daemon returns. +# `systemctl disable --now colony-firewall-nft-inbound` removes inbound +# filtering. A plain stop or restart also stops or restarts the network +# managers that require this unit; refresh the rules with reload. Its table +# remains loaded across daemon restarts and stops; a missing queue listener +# drops new inbound flows until the daemon returns. [Unit] Description=Colony Firewall inbound nftables rules diff --git a/systemd/colony-firewall-nft.service b/systemd/colony-firewall-nft.service index e24206e..c9f53eb 100644 --- a/systemd/colony-firewall-nft.service +++ b/systemd/colony-firewall-nft.service @@ -1,6 +1,13 @@ # Keep the fail-closed NFQUEUE table installed across daemon restarts and stops. -# Stop this unit explicitly to remove filtering, or uninstall the package. # Load before the daemon, including when its initialization fails. +# +# To remove filtering, `systemctl disable --now colony-firewall-nft` (add +# colony-firewalld to keep it off: starting the daemon loads the table again). +# disable drops the network managers' Requires= before the stop. A plain stop +# or restart propagates through that Requires= to NetworkManager and +# systemd-networkd, and restart also leaves a moment without the table: +# refresh the rules with reload. To load a local copy of the snippet, override +# both ExecStart= and ExecReload= in a drop-in (docs/TROUBLESHOOTING.md). [Unit] Description=Colony Firewall nftables NFQUEUE rules @@ -26,4 +33,6 @@ ExecStop=/usr/bin/nft delete table inet colony_firewall WantedBy=multi-user.target # Native .requires links make nft load failure prevent manager startup. # They exist only after enforcement is enabled and are removed on disable. +# Requires= also carries an explicit stop or restart of this unit to the +# managers; see the header for the supported way to lift filtering. RequiredBy=NetworkManager.service systemd-networkd.service diff --git a/systemd/nftables-snippet.conf b/systemd/nftables-snippet.conf index b55208d..b306a88 100644 --- a/systemd/nftables-snippet.conf +++ b/systemd/nftables-snippet.conf @@ -5,7 +5,8 @@ # flows follow explicit application policy; unmatched local IPC is allowed # without prompting. See docs/HARDENING.md. # Enable colony-firewall-nft.service for persistence. Its rules remain loaded -# across daemon restarts and stops. Stop that unit explicitly to lift filtering. +# across daemon restarts and stops. Lift filtering with +# `systemctl disable --now colony-firewall-nft`, not a plain stop. # One nft -f transaction replaces an existing table without a filtering gap. # add is idempotent and makes the same batch work on the first installation. From eb2db811199d3e609744783c3d9f430bff9da711 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:57:02 +0200 Subject: [PATCH 071/125] fix(systemd): give the daemon a private /dev and drop AF_PACKET uid 0 owns the block devices through plain DAC, so the daemon could write the disks underneath ProtectSystem=strict. It opens no device node, so PrivateDevices= costs nothing and also drops CAP_MKNOD, CAP_SYS_RAWIO and @raw-io. Nothing opens a packet socket either. The hand-written ReadOnlyPaths now cover more of what ProtectKernelTunables= does, and HARDENING.md no longer lists that directive as set. --- docs/HARDENING.md | 6 ++++-- systemd/colony-firewalld.service | 20 +++++++++++++++----- 2 files changed, 19 insertions(+), 7 deletions(-) diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 0dbc890..30959d6 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -401,9 +401,11 @@ to shrink what a code-execution bug could reach: | `SystemCallArchitectures=native` | Closes the 32-bit-syscall bypass of that filter | | `MemoryDenyWriteExecute` | Nothing here JITs; no W+X memory | | `ProtectSystem=strict`, `ProtectHome`, `ReadWritePaths` | Read-only filesystem apart from the state, runtime and log directories | -| `RestrictAddressFamilies` | AF_UNIX, AF_INET, AF_INET6, AF_NETLINK, AF_PACKET only | +| `PrivateDevices` | Private `/dev` with only pseudo devices: uid 0 cannot open the block devices and write underneath `ProtectSystem` | +| `RestrictAddressFamilies` | AF_UNIX, AF_INET, AF_INET6, AF_NETLINK only; no packet sockets | | `RestrictNamespaces`, `LockPersonality`, `RestrictRealtime`, `RestrictSUIDSGID` | Namespace and personality lockdown | -| `ProtectKernelTunables`, `ProtectKernelLogs`, `ProtectControlGroups`, `ProtectClock`, `ProtectHostname` | No writing kernel state | +| `ProtectKernelLogs`, `ProtectControlGroups`, `ProtectClock`, `ProtectHostname` | No writing kernel state | +| `ReadOnlyPaths` (in place of `ProtectKernelTunables`) | The `/proc` and `/sys` entries `ProtectKernelTunables` covers, listed by hand so `/sys/fs/bpf` stays writable for the pinned links. An entry missing from the list stays writable | | `UMask=0077` | Closes the window between `bind` and the explicit chmod of the control socket | | `PrivateTmp` | No shared `/tmp` | diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index 12378d1..b7102b7 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -109,6 +109,11 @@ LogsDirectory=colony-firewall # Sandboxing PrivateTmp=true +# The daemon opens no device node. Without this, uid 0 owns the block devices +# through plain DAC (root:disk 0660) and could write the disks underneath +# ProtectSystem. It also drops CAP_MKNOD, CAP_SYS_RAWIO and @raw-io. The nft +# and rpm children only need /dev/null, which the private /dev keeps. +PrivateDevices=true # ProtectKernelTunables is deliberately OFF, and replaced by hand below. # # It does the right thing and one wrong thing: it remounts /sys read-only @@ -130,9 +135,9 @@ PrivateTmp=true # read-only, /sys/kernel is read-only, and /sys/kernel/btf and # /sys/kernel/tracing remain readable - which the eBPF loader needs. # -# A new top-level directory in a future kernel's sysfs would be writable to -# this unit until it is added here. That is the cost of the trade, and it is -# smaller than losing the pin. +# A new top-level directory in a future kernel's sysfs, or a /sys/fs or /proc +# entry not listed here, would be writable to this unit until it is added +# here. That is the cost of the trade, and it is smaller than losing the pin. # # EVERY path is prefixed with "-", and that is not defensive tidiness. Without # it, systemd treats a missing path as a fatal error and the unit dies at @@ -152,10 +157,13 @@ PrivateTmp=true # is the half that was missed. ProtectKernelTunables=false ReadOnlyPaths=-/proc/sys -/proc/sysrq-trigger -/proc/irq -/proc/acpi -/proc/fs +ReadOnlyPaths=-/proc/apm -/proc/asound -/proc/bus -/proc/latency_stats -/proc/mtrr +ReadOnlyPaths=-/proc/scsi -/proc/timer_list -/proc/timer_stats ReadOnlyPaths=-/sys/block -/sys/bus -/sys/class -/sys/dev -/sys/devices -/sys/firmware ReadOnlyPaths=-/sys/hypervisor -/sys/kernel -/sys/module -/sys/power ReadOnlyPaths=-/sys/fs/btrfs -/sys/fs/cgroup -/sys/fs/ext4 -/sys/fs/fuse ReadOnlyPaths=-/sys/fs/pstore -/sys/fs/resctrl -/sys/fs/tmpfs -/sys/fs/virtiofs +ReadOnlyPaths=-/sys/fs/erofs -/sys/fs/f2fs -/sys/fs/selinux -/sys/fs/xfs ProtectKernelLogs=true ProtectControlGroups=true RestrictNamespaces=true @@ -224,8 +232,10 @@ SystemCallArchitectures=native # (root:colony-firewall 0660); a tight umask closes the pre-chmod window. UMask=0077 -# Required for NFQUEUE -RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 AF_NETLINK AF_PACKET +# AF_NETLINK carries NFQUEUE, sock_diag and nftables; AF_INET/AF_INET6 the +# Reject raw sockets; AF_UNIX the control socket. Nothing opens a packet +# socket, and AF_PACKET with CAP_NET_RAW would sniff or inject below netfilter. +RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 AF_NETLINK [Install] WantedBy=multi-user.target From ebb35febd124f25a01a6fd97a749eea4194e5840 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:57:02 +0200 Subject: [PATCH 072/125] fix(release): ship a tarball installer and attest its provenance No Colony app store client reads pkg/colony.json or runs its postInstall/preRemove, so the release tarball had no installer and the documented store channel did not exist. scripts/tarball-installers.sh now generates install.sh and uninstall.sh from the manifest into the tarball, and the docs describe it as a manual channel. The publish job attests the tarball with Sigstore build provenance, and a dispatched draft tags the commit it was built from. SECURITY.md and pkg/README say what SHA256SUMS and the PKGBUILD checksum do not prove. The release LLVM pairing check read a version bpf-linker 0.11 does not print, so it never fired; it now runs ebpf.yml's two checks. --- .github/workflows/check.yml | 2 +- .github/workflows/ebpf.yml | 4 +- .github/workflows/release.yml | 55 ++++++++++++----- SECURITY.md | 17 ++++++ TODO.md | 2 +- crates/cfc-daemon/src/ebpf.rs | 6 +- docs/ROADMAP.md | 7 ++- pkg/README.md | 102 ++++++++++++++++---------------- scripts/check-release-assets.sh | 12 ++-- scripts/tarball-installers.sh | 34 +++++++++++ 10 files changed, 162 insertions(+), 79 deletions(-) create mode 100755 scripts/tarball-installers.sh diff --git a/.github/workflows/check.yml b/.github/workflows/check.yml index b594194..8147f15 100644 --- a/.github/workflows/check.yml +++ b/.github/workflows/check.yml @@ -33,7 +33,7 @@ jobs: # Guards the release tarball against pkg/colony.json: every file the # manifest's postInstall installs has to be staged by release.yml, or the - # Colony channel silently ships without it (this is how the sysusers + # tarball's install.sh silently ships without it (this is how the sysusers # fragment went missing and left the control socket root-only). release-assets: runs-on: ubuntu-latest diff --git a/.github/workflows/ebpf.yml b/.github/workflows/ebpf.yml index 8e76eb4..472bb33 100644 --- a/.github/workflows/ebpf.yml +++ b/.github/workflows/ebpf.yml @@ -140,8 +140,8 @@ jobs: # - and its rustc side read stable (see the install step). # # (1) The system LLVM the install step provided must actually be the - # major it asked llvm.sh for - the script can fall back when a - # major is not packaged for the runner's distro yet. Comparing + # major it requested from apt.llvm.org, which may not carry that + # major for the runner's release yet. Comparing # the nightly's LLVM against itself would assert nothing; this # compares it against what landed on disk. WANT="${{ steps.linker.outputs.llvm_major }}" diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 69cd189..4b61c1d 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -93,11 +93,13 @@ jobs: # down, because the assert below requires the two to be equal and a # second hardcoded number could only drift away from the first. - name: Install bpf-linker + id: linker run: | LLVM_MAJOR="$(rustc +${{ steps.bpfpins.outputs.nightly }} -vV | sed -n 's/^LLVM version: \([0-9]*\).*/\1/p')" if [ -z "${LLVM_MAJOR}" ]; then echo "::error::could not read rustc's LLVM version"; exit 1 fi + echo "llvm_major=${LLVM_MAJOR}" >> "${GITHUB_OUTPUT}" echo "pinned nightly uses LLVM ${LLVM_MAJOR}; building bpf-linker against it" sudo apt-get update @@ -125,15 +127,22 @@ jobs: # whatever libLLVM the system provides -- so assert it rather than assume. - name: Assert the LLVM versions pair run: | - RUSTC_LLVM="$(rustc +${{ steps.bpfpins.outputs.nightly }} -vV | sed -n 's/^LLVM version: \([0-9]*\).*/\1/p')" - LINKER_LLVM="$(bpf-linker --version 2>/dev/null | sed -n 's/.*LLVM \([0-9]*\).*/\1/p')" - echo "rustc LLVM=${RUSTC_LLVM} bpf-linker LLVM=${LINKER_LLVM}" - if [ -z "${RUSTC_LLVM}" ]; then - echo "::error::could not determine rustc's LLVM version" + # The same two checks as ebpf.yml. `bpf-linker --version` prints no + # LLVM version in 0.11, so comparing against it never fired. + # (1) The system LLVM must be the major requested from apt.llvm.org. + WANT="${{ steps.linker.outputs.llvm_major }}" + GOT="$("/usr/lib/llvm-${WANT}/bin/llvm-config" --version 2>/dev/null | cut -d. -f1 || true)" + echo "requested LLVM=${WANT} system llvm-config says=${GOT}" + if [ "${GOT}" != "${WANT}" ]; then + echo "::error::asked apt.llvm.org for LLVM ${WANT} but /usr/lib/llvm-${WANT} answers '${GOT}' - the pinned nightly's LLVM is not installable here, so the bitcode handoff to bpf-linker will fail with 'Invalid record'. Bump crates/cfc-ebpf/bpf-linker-version or move the nightly pin back." exit 1 fi - if [ -n "${LINKER_LLVM}" ] && [ "${RUSTC_LLVM}" -gt "${LINKER_LLVM}" ]; then - echo "::error::rustc LLVM ${RUSTC_LLVM} is newer than bpf-linker's ${LINKER_LLVM}; the link will fail with 'Invalid record'. Bump crates/cfc-ebpf/bpf-linker-version or move the nightly pin back." + # (2) What the installed binary actually links, read from the binary. + LINKED="$(ldd "$(command -v bpf-linker)" 2>/dev/null | grep -oE 'libLLVM[-.so]*[0-9]+' | grep -oE '[0-9]+$' | head -1 || true)" + if [ -z "${LINKED}" ]; then + echo "::warning::could not read a libLLVM major from the bpf-linker binary (static link or layout change); relying on check (1) only" + elif [ "${LINKED}" != "${WANT}" ]; then + echo "::error::bpf-linker links libLLVM ${LINKED} but was requested against LLVM ${WANT}" exit 1 fi @@ -158,19 +167,19 @@ jobs: "${STAGE}/" # Executable, and not a binary: systemd execs it as the inbound - # unit's ExecStartPre. colony.json re-applies 0755 on install, but + # unit's ExecStartPre. install.sh re-applies 0755 on install, but # staging it 644 here would be a lie about what it is. install -m755 \ scripts/inbound-lockout-guard.sh \ "${STAGE}/" # Everything pkg/colony.json's postInstall installs. The tarball - # layout is FLAT: postInstall runs with the extracted directory - # as its working directory, so each file must be present under + # layout is FLAT: install.sh, generated from postInstall, runs + # from the extracted directory, so each file must be present under # its own basename. scripts/check-release-assets.sh gates this # list against the manifest on every push, so a file added to # postInstall but not staged here fails CI instead of silently - # vanishing on the Colony channel. + # vanishing from tarball installs. install -m644 \ systemd/colony-firewalld.service \ systemd/colony-firewall-nft.service \ @@ -185,9 +194,8 @@ jobs: pkg/colony-firewall.svg \ "${STAGE}/" - # The kernel-side object. Fatal if absent: colony.json's postInstall - # cannot express an optional file, so a tarball without it produces a - # package whose install step fails outright. + # The kernel-side object. Fatal if absent: install.sh cannot express + # an optional file, so a tarball without it would fail to install. install -m644 \ crates/cfc-ebpf/target/bpfel-unknown-none/release/cfc-ebpf.o \ "${STAGE}/" @@ -204,6 +212,9 @@ jobs: docs/TROUBLESHOOTING.md \ "${STAGE}/" + # The tarball's installer and uninstaller, from the same manifest. + ./scripts/tarball-installers.sh "${STAGE}" + tar --zstd -C "$(dirname "${STAGE}")" -cf "${NAME}.tar.zst" "${NAME}" sha256sum "${NAME}.tar.zst" > SHA256SUMS @@ -380,11 +391,23 @@ jobs: runs-on: ubuntu-latest permissions: contents: write + # Signed build provenance for the tarball (actions/attest below). + id-token: write + attestations: write steps: - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: name: release-assets path: release-assets + # SHA256SUMS is uploaded beside the tarball, so whoever can edit the + # release can replace both. This Sigstore-signed SLSA provenance binds + # the tarball's digest to this repository's release workflow and commit + # instead: `gh attestation verify --repo + # Project-Colony/Colony-Firewall-Control`. It runs here rather than in + # build so no build script can mint the job's OIDC token. + - uses: actions/attest@1e69f48acb82d1966a394da916b4c1698aa569d6 # v4.2.2 + with: + subject-path: release-assets/colony-firewall-control-*.tar.zst - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 if: github.ref_type == 'tag' with: @@ -395,6 +418,10 @@ jobs: with: draft: true tag_name: ${{ github.ref_type == 'tag' && github.ref_name || format('v{0}', needs.build.outputs.version) }} + # A dispatch has no tag yet: publishing the draft creates it, and + # without this it would point at the default branch rather than + # the commit the assets were built from. Ignored when the tag exists. + target_commitish: ${{ github.sha }} name: ${{ github.ref_type == 'tag' && github.ref_name || format('v{0}', needs.build.outputs.version) }} body_path: release-assets/RELEASE_BODY.md files: | diff --git a/SECURITY.md b/SECURITY.md index 0c9dc9a..715bf94 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -39,3 +39,20 @@ What to expect: Dependency vulnerabilities are scanned continuously in CI with `cargo deny` (RustSec advisory database); a report is still welcome if you spot an exploitable path through a dependency before CI does. + +## Verifying Releases + +Release assets are not signed with a project key. From 0.8.0 the release +tarball carries a Sigstore-signed build provenance attestation, which ties its +digest to this repository's release workflow and the commit it built: + +```sh +gh attestation verify colony-firewall-control--linux-x86_64.tar.zst \ + --repo Project-Colony/Colony-Firewall-Control +``` + +`SHA256SUMS` comes from the same job and is uploaded beside the tarball, so it +detects a damaged download, not a replaced one. The attached `PKGBUILD` pins +the hash of the source archive GitHub served when the tag was built; nothing +compares that archive with the tagged tree, so it binds the recipe to those +bytes and no further. diff --git a/TODO.md b/TODO.md index 0d869f2..073adc8 100644 --- a/TODO.md +++ b/TODO.md @@ -221,7 +221,7 @@ stable. An AUR install therefore gets `Degrade::ObjectMissing` and runs on Three ways out, none free: -1. leave it (what happens today - the Colony tarball has the object, AUR does not); +1. leave it (what happens today - the release tarball has the object, AUR does not); 2. ship the object as a second `source=()` from the release assets - but that deadlocks against draft releases, and it would be the one shipped component no AUR user builds from source, which for kernel code deserves a hard think; diff --git a/crates/cfc-daemon/src/ebpf.rs b/crates/cfc-daemon/src/ebpf.rs index a0041fc..934f487 100644 --- a/crates/cfc-daemon/src/ebpf.rs +++ b/crates/cfc-daemon/src/ebpf.rs @@ -137,9 +137,9 @@ pub enum NoteLevel { /// Where the BPF object is expected to live. /// -/// The Colony package installs it here (`pkg/colony.json` postInstall, 0644 -/// root:root, which is also what `loader::vet_object` requires before loading -/// it unasked). `.github/workflows/release.yml` builds it with +/// The release tarball's `install.sh` installs it here (generated from +/// `pkg/colony.json` postInstall, 0644 root:root, which is also what +/// `loader::vet_object` requires before loading it unasked). `.github/workflows/release.yml` builds it with /// `cargo xtask build-ebpf` and stages it into the tarball; /// `scripts/check-release-assets.sh` fails the build if the manifest and the /// tarball ever disagree about it. diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 6471335..5ddd398 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -21,7 +21,9 @@ is still manual. - [x] systemd unit + nft snippet - [x] CI: cargo fmt + clippy + test + build - [x] AUR PKGBUILD draft -- [x] Colony app store manifest (`pkg/colony.json`) +- [x] Release tarball recipe (`pkg/colony.json`; no Colony store client reads + it, `scripts/tarball-installers.sh` turns it into the tarball's + `install.sh`) ## Phase 1 - Daemon MVP [done] @@ -175,7 +177,8 @@ kernel 7.1.8. - [ ] VirusTotal lookup integration (optional, opt-in) - [x] Profile presets: relaxed / balanced / strict - [x] Import rules from opensnitch JSON -- [x] Colony app store manifest (`colony.json`) +- [ ] Colony app store listing (needs a root `colony.json` in Colony's real + schema; the store installs one per-user binary, never units or tables) - [x] AUR PKGBUILD draft (now AUR-ready in `pkg/`; not yet published, signed release pending) - [x] Shell completions + man pages diff --git a/pkg/README.md b/pkg/README.md index cbc8b65..75dbabc 100644 --- a/pkg/README.md +++ b/pkg/README.md @@ -20,7 +20,7 @@ Everything a package ships, at a glance: | `README.md`, `docs/{ARCHITECTURE,HARDENING,TROUBLESHOOTING}.md` | `/usr/share/doc/colony-firewall-control/` | | `LICENSE` | `/usr/share/licenses/colony-firewall-control/` | | generated by `cfc` | completions in `/usr/share/{bash-completion/completions,zsh/site-functions,fish/vendor_completions.d}/`, man pages in `/usr/share/man/man1/` | -| `cfc-ebpf.o` (built by `cargo xtask build-ebpf`) | `/usr/lib/colony-firewall/` — Colony tarball only; the AUR and RPM packages deliberately do not ship it (see `DEFAULT_OBJECT_PATH` in `crates/cfc-daemon/src/ebpf.rs`), and the daemon degrades to `sock_diag` + `/proc` attribution without it | +| `cfc-ebpf.o` (built by `cargo xtask build-ebpf`) | `/usr/lib/colony-firewall/`, release tarball only; the AUR and RPM packages deliberately do not ship it (see `DEFAULT_OBJECT_PATH` in `crates/cfc-daemon/src/ebpf.rs`), and the daemon degrades to `sock_diag` + `/proc` attribution without it | Key design points: @@ -33,7 +33,7 @@ Key design points: unit creates native `Requires` links from NetworkManager and systemd-networkd, ordered after nft loading; load failure blocks those managers' startup. Enabling the daemon also enables outbound enforcement. - Upgrade scripts reenable existing deployments to create those links. + Upgrade scripts reenable enabled nft units to create those links. Explicit disable and uninstall remove them. The units load after a distribution `nftables.service` in the same boot transaction. This contract excludes initramfs networking, already configured interfaces, other network @@ -65,6 +65,12 @@ pattern anyone would guess could download). Rename `SRCINFO` to push without a file of exactly that name. Releases up to v0.2.3 carry the old `default.SRCINFO` spelling. +The checksum `updpkgsums` writes is trust on first use: it is the hash of +whatever GitHub served for `archive/v$pkgver.tar.gz` when the tag was built, +and nothing compares that archive with the tagged tree. Release assets are +not signed with a project key; the tarball's provenance attestation is +described in `SECURITY.md`. + Per release, by hand: ```sh @@ -92,8 +98,8 @@ belong in the AUR repository — `.gitignore` everything else `makepkg` leaves behind. The install scriptlet (`colony-firewall-control.install`) prints first-run -steps on install (enable units, join the `colony-firewall` group, -`cfc rules bootstrap-defaults`) and — critically — its `pre_remove` stops +steps on install (enable units, `sudo cfc rules bootstrap-defaults`, join +the `colony-firewall` group) and, critically, its `pre_remove` stops `colony-firewall-nft` + `colony-firewalld` and deletes `table inet colony_firewall`, so removing the package can never leave the fail-closed queue rule behind and blackhole outbound traffic. @@ -120,33 +126,49 @@ and the install scriptlet are exercised without needing a published tag. sourceability, `makepkg --printsrcinfo`, `namcap`, `shellcheck` on the scriptlet) for both recipes. -## Colony app store +## Release tarball + +The release attaches `colony-firewall-control--linux-x86_64.tar.zst`, the +only channel that ships `cfc-ebpf.o`. Install and remove it by hand, as root: + +```sh +tar --zstd -xf colony-firewall-control--linux-x86_64.tar.zst +cd colony-firewall-control--linux-x86_64 +sudo ./install.sh # then First run in README.md +sudo ./uninstall.sh +``` + +Verify the download first with `gh attestation verify` (see `SECURITY.md`). -`colony.json` follows the Colony manifest format (camelCase, single asset -per platform). Notes on this manifest: +Both scripts are generated by `scripts/tarball-installers.sh` from +`colony.json`. Despite the name, no Colony app store client reads that +file: the store only lists repositories with a root `colony.json` in its own +schema, and it installs a single per-user binary and runs no scripts, so it +could not install units, nft tables or the sysusers group anyway. The +manifest is the tarball's install recipe: - `postInstall` is **idempotent**: `daemon.toml` is only installed if - absent, so upgrades never clobber user config. -- `preRemove` mirrors the pacman `pre_remove`: stop the units, delete - both Colony nft tables and `/sys/fs/bpf/colony-firewall`, then remove the non-binary files - `postInstall` placed (user config in `/etc/colony-firewall/` is kept). - Stores too old to know the `preRemove` key ignore it — on those, run the - `preRemove` commands manually before uninstalling, or outbound traffic - stays blackholed by the orphaned fail-closed table. + absent, so upgrades never clobber user config. `install.sh` runs it + after copying `binaries` to `installPath`. +- `preRemove` mirrors the pacman `pre_remove`: disable and stop the units, + delete both Colony nft tables and `/sys/fs/bpf/colony-firewall`, then + remove the files `postInstall` placed (user config in + `/etc/colony-firewall/` is kept). `uninstall.sh` runs it and then removes + the binaries. The tarball is built by the `build` job in `.github/workflows/release.yml`, which is the source of truth for its contents. It unpacks to a single `colony-firewall-control--linux-x86_64/` directory, and inside that -the layout is **flat**: `postInstall` runs with the extracted directory -as its working directory, so every file it names must sit directly in -it under exactly that basename. +the layout is **flat**: `install.sh` runs from the extracted directory, so +every file `postInstall` names must sit directly in it under exactly that +basename. `scripts/check-release-assets.sh` (a hard gate in `check.yml`, re-run in `release.yml`) parses this manifest's `postInstall` commands and fails if any source they install is not staged by the workflow. Adding a line to `postInstall` without adding the file to the workflow is therefore a CI -failure rather than a silent no-op on the store channel — which is how +failure rather than a silent gap in tarball installs, which is how the `colony-firewall.sysusers` fragment previously went missing, leaving no `colony-firewall` group and a root-only control socket. @@ -154,7 +176,7 @@ To reproduce the tarball locally, from the repo root: ```sh cargo build --workspace --release --locked -cargo xtask build-ebpf # cfc-ebpf.o; postInstall fails outright without it +cargo xtask build-ebpf # cfc-ebpf.o; install.sh fails outright without it V=0.7.0 NAME="colony-firewall-control-${V}-linux-x86_64" @@ -185,6 +207,7 @@ install -m644 \ README.md CHANGELOG.md LICENSE \ docs/ARCHITECTURE.md docs/HARDENING.md docs/TROUBLESHOOTING.md \ "${STAGE}/" +./scripts/tarball-installers.sh "${STAGE}" tar --zstd -C "$(dirname "${STAGE}")" -cf "${NAME}.tar.zst" "${NAME}" ``` @@ -194,38 +217,16 @@ the two ever disagree, the workflow wins — it is the copy `scripts/check-release-assets.sh` gates against the manifest, this one is not — so fix this list to match it. -Then upload the tarball as a GitHub Release asset and link it from the -`asset` field of `colony.json` (the version in that filename is checked -by `scripts/check-versions.sh`). +The version in the manifest's `asset` filename is checked by +`scripts/check-versions.sh`. ## Manual install -After `cargo build --release`: - -```sh -sudo install -Dm755 target/release/colony-firewalld /usr/bin/colony-firewalld -sudo install -Dm755 target/release/colony-firewall /usr/bin/colony-firewall -sudo install -Dm755 target/release/cfc /usr/bin/cfc -sudo install -Dm644 systemd/colony-firewalld.service /usr/lib/systemd/system/colony-firewalld.service -sudo install -Dm644 systemd/colony-firewall-nft.service /usr/lib/systemd/system/colony-firewall-nft.service -sudo install -Dm644 systemd/colony-firewall-nft-inbound.service /usr/lib/systemd/system/colony-firewall-nft-inbound.service -sudo install -Dm644 systemd/colony-firewall.sysusers /usr/lib/sysusers.d/colony-firewall.conf -sudo install -Dm644 systemd/nftables-snippet.conf /usr/share/colony-firewall/nftables-snippet.conf -sudo install -Dm644 systemd/nftables-inbound.conf /usr/share/colony-firewall/nftables-inbound.conf -sudo install -Dm755 scripts/inbound-lockout-guard.sh /usr/lib/colony-firewall/inbound-lockout-guard.sh -sudo install -Dm644 systemd/daemon.toml.sample /etc/colony-firewall/daemon.toml -sudo install -Dm644 pkg/colony-firewall.desktop /usr/share/applications/colony-firewall.desktop -sudo install -Dm644 pkg/colony-firewall-autostart.desktop /etc/xdg/autostart/colony-firewall.desktop -sudo install -Dm644 pkg/colony-firewall.svg /usr/share/icons/hicolor/scalable/apps/colony-firewall.svg -sudo systemd-sysusers -sudo systemctl daemon-reload -sudo systemctl enable --now colony-firewalld colony-firewall-nft -sudo usermod -aG colony-firewall "$USER" # then log out/in -cfc rules bootstrap-defaults -``` - -`colony-firewall-nft` replaces the old manual `nft -f -systemd/nftables-snippet.conf` step and survives reboots. +Follow the Manual section of the top-level `README.md`, then its First +run section: enable enforcement, then seed the starter rules with +`sudo cfc rules bootstrap-defaults` right away. `sudo` matters there, because +group membership from `usermod -aG colony-firewall` only applies after a new +login, and until the rules exist, unmatched DHCP, DNS and NTP flows are denied. ## Uninstall behavior (all channels) @@ -240,5 +241,6 @@ Order matters because the nftables snippet is fail-closed: 4. Remove files. `/etc/colony-firewall/daemon.toml` is user config and is left behind (pacman saves it as `.pacsave`). -The AUR package does this automatically via `pre_remove`; the Colony -manifest via `preRemove`; manual installs should follow the steps above. +The AUR package does this automatically via `pre_remove`, the RPM via +`%preun`/`%postun` and the release tarball via `uninstall.sh`; manual +installs should follow the steps above. diff --git a/scripts/check-release-assets.sh b/scripts/check-release-assets.sh index 5e850a6..099542a 100755 --- a/scripts/check-release-assets.sh +++ b/scripts/check-release-assets.sh @@ -2,12 +2,12 @@ # Asserts that the release tarball actually contains every file the # packaging recipes expect to find in it. # -# The Colony app-store channel installs from a flat tarball: the commands -# in pkg/colony.json's "postInstall" run with the extracted directory as -# their working directory, so every relative source they name must have -# been staged by the "Assemble tarball" step in -# .github/workflows/release.yml. When it is not, the install silently -# skips that file -- which is how the sysusers fragment went missing and +# The release tarball installs from a flat directory: its install.sh, +# generated from pkg/colony.json's "postInstall" by +# scripts/tarball-installers.sh, runs from the extracted directory, so every +# relative source it names must have been staged by the "Assemble tarball" +# step in .github/workflows/release.yml. When it is not, the install fails +# or skips that file -- which is how the sysusers fragment went missing and # left the control socket root-only. # # Checks, all fatal: diff --git a/scripts/tarball-installers.sh b/scripts/tarball-installers.sh new file mode 100755 index 0000000..bc679a1 --- /dev/null +++ b/scripts/tarball-installers.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +# Writes install.sh and uninstall.sh into a staged release tarball directory, +# generated from pkg/colony.json. +# +# No Colony app store client reads that manifest: Colony installs a single +# per-user binary and runs no scripts, so it could not install units, nft +# tables or sysusers anyway. These two scripts are what run its postInstall +# and preRemove lists, so the tarball is a manual channel with a real +# installer. Called by the "Assemble tarball" step of release.yml. +# +# Usage: scripts/tarball-installers.sh + +set -euo pipefail + +STAGE="${1:?usage: $0 }" +MANIFEST="$(cd "$(dirname "$0")/.." && pwd)/pkg/colony.json" +PLATFORM=linux-x86_64 + +{ + printf '#!/bin/sh\n# Generated from pkg/colony.json. Run as root: sudo ./install.sh\nset -e\ncd "$(dirname "$0")"\n' + jq -er --arg p "${PLATFORM}" '.platforms[$p] + | "install -m 0755 \(.binaries | join(" ")) \(.installPath)/", .postInstall[]' "${MANIFEST}" + printf 'echo "Installed. Enable enforcement as described under First run in README.md."\n' +} >"${STAGE}/install.sh" + +{ + printf '#!/bin/sh\n# Generated from pkg/colony.json. Run as root: sudo ./uninstall.sh\nset -e\n' + jq -er --arg p "${PLATFORM}" '.platforms[$p] + | .preRemove[], "rm -f \(.installPath as $dir | .binaries | map("\($dir)/\(.)") | join(" "))"' "${MANIFEST}" +} >"${STAGE}/uninstall.sh" + +chmod 0755 "${STAGE}/install.sh" "${STAGE}/uninstall.sh" +sh -n "${STAGE}/install.sh" +sh -n "${STAGE}/uninstall.sh" From 165492069a67a79aa528377c757ee22886475dc6 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:57:02 +0200 Subject: [PATCH 073/125] fix(pkg): refuse the SKIP checksum in build() too makepkg --noprepare skips prepare(), where the guard lived, and built the unverified archive. Both functions now call the same check. --- pkg/PKGBUILD | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/pkg/PKGBUILD b/pkg/PKGBUILD index fb16b85..1f2449e 100644 --- a/pkg/PKGBUILD +++ b/pkg/PKGBUILD @@ -46,17 +46,23 @@ sha256sums=('SKIP') # GitHub release tarballs extract to - (no leading 'v'). _srcname=Colony-Firewall-Control-$pkgver -prepare() { +# Called from build() too: `makepkg --noprepare` skips prepare(). +_require_checksum() { if [[ "${sha256sums[0]}" == 'SKIP' ]]; then printf '%s\n' 'Release template has no source checksum. Run updpkgsums before building.' >&2 return 1 fi +} + +prepare() { + _require_checksum cd "$srcdir/$_srcname" export RUSTUP_TOOLCHAIN=stable cargo fetch --locked --target "$(rustc -vV | sed -n 's/host: //p')" } build() { + _require_checksum cd "$srcdir/$_srcname" export RUSTUP_TOOLCHAIN=stable export CARGO_TARGET_DIR=target From 4cd077ab40cf672ae663e17190ab949179088ea4 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 02:57:02 +0200 Subject: [PATCH 074/125] docs(changelog): list the packaging and systemd fixes in this bundle --- CHANGELOG.md | 33 +++++++++++++++++++++++++++++++++ 1 file changed, 33 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index f55fa35..6b73ce5 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -51,6 +51,17 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). `journalctl -u cfc-app-ID.service`; it used to be discarded and reported as the application's status 1. +- The release tarball ships `install.sh` and `uninstall.sh`, generated from + `pkg/colony.json`. No Colony app store client ever read that manifest or + ran its `postInstall`/`preRemove`, so the tarball had no installer and the + documented store channel did not exist; the docs now say so. +- Lifting filtering is `systemctl disable --now colony-firewall-nft` (plus + `colony-firewalld` to keep it off), not a plain stop. A stop or restart + propagates through the network managers' `Requires=` to NetworkManager and + systemd-networkd. The docs now carry ruleset changes as a local copy loaded + through a unit drop-in: a same-named table from `/etc/nftables.conf` is + replaced at boot, on daemon start and on every upgrade. + ### Security - `cfc prompts`: keys typed while no prompt was shown, such as an answer @@ -84,6 +95,15 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). working when the daemon is down. Loopback Deny rules are not enforced then. Every other new flow stays fail-closed. +- The daemon unit sets `PrivateDevices=` (uid 0 could otherwise open block + devices and write underneath `ProtectSystem=`), drops `AF_PACKET`, which + nothing used, and lists more of the `/proc` and `/sys/fs` entries + `ProtectKernelTunables=` covers. `docs/HARDENING.md` no longer claims + `ProtectKernelTunables=` is set. +- The release tarball carries a Sigstore-signed build provenance + attestation; `SECURITY.md` explains how to verify it and what + `SHA256SUMS` and the attached `PKGBUILD` checksum do not prove. + ### Removed - The Fast Allow userspace path, disabled since 0.7.0 because a socket mark @@ -210,6 +230,19 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - The GUI's prompt subscription stayed open after the GUI dropped it, so the daemon held the next prompt for an absent listener until it timed out. +- RPM erase stopped NetworkManager: `%systemd_preun` stops the nft units + with `--no-reload`, under the managers' loaded `Requires=`. `%preun` now + disables them with a reload first. +- Package upgrades re-enabled an nft unit the admin had disabled, because + reenabling the daemon follows its `Also=`. Only enabled nft units are + reenabled now (pacman, RPM and the tarball). +- `pkg/PKGBUILD` refuses the `SKIP` checksum in `build()` too, so + `makepkg --noprepare` cannot build an unverified archive. +- The release's LLVM pairing check compared against a version + `bpf-linker --version` does not print, so it never fired; it now runs the + same checks as `ebpf.yml`. A dispatched release's draft now tags the commit + it was built from. + ## [0.7.0] - 2026-09-30 ### Added From 0ce644476f240dd579ccdd6696a976fa2cfa58cc Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:28 +0200 Subject: [PATCH 075/125] fix(daemon): keep at most 4 KiB of a process's arguments The resolver read the whole of /proc//cmdline, which the kernel lets reach several MiB, and a copy rode every parked prompt, observation and client message, so a program with a huge argument list multiplied the daemon's and every client's memory. Arguments are only displayed, never matched, so the read now stops at 4 KiB and marks a cut argument with "...". Arguments that are not UTF-8 are decoded lossily instead of being dropped. --- crates/cfc-daemon/src/process_resolve.rs | 58 ++++++++++++++++++++++-- 1 file changed, 54 insertions(+), 4 deletions(-) diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 29345b4..052a2fd 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -96,6 +96,11 @@ const SHA256_MAX_LEN: u64 = cfc_core::rule::SHA256_MAX_LEN; const CACHE_CAP: usize = 1024; +/// How much of `/proc//cmdline` is kept. Arguments are only shown, never +/// matched by a rule, and the kernel allows several MiB of them, a copy of +/// which rides every parked prompt, observation and client message. +const CMDLINE_MAX_BYTES: usize = 4096; + static INODE_PID_CACHE: LazyLock>> = LazyLock::new(|| Mutex::new(TtlCache::new(INODE_CACHE_TTL, CACHE_CAP))); @@ -192,10 +197,7 @@ fn resolve_inner( // cannot pair an old path with a new digest. let proc_exe_path = PathBuf::from(format!("/proc/{pid}/exe")); let image = MappedImage::open(&proc_exe_path); - let mut cmdline = p - .as_ref() - .and_then(|p| p.cmdline().ok()) - .unwrap_or_default(); + let mut cmdline = p.as_ref().map(read_cmdline).unwrap_or_default(); let mut cwd = p.as_ref().and_then(|p| p.cwd().ok()); let (ppid, uid, gid) = match &kern { @@ -322,6 +324,37 @@ fn resolve_inner( /// flow to the process *listening* on the port is a real and separate thing, /// and it would be one netlink round trip rather than two /proc scans; it is /// not done here because it would change which rules match, not just how fast. +/// The process's arguments, at most [`CMDLINE_MAX_BYTES`] of them. A cut +/// argument ends in `...`. +fn read_cmdline(p: &ProcFsProcess) -> Vec { + use std::io::Read as _; + let mut buf = Vec::new(); + let read = p.open_relative("cmdline").and_then(|f| { + f.take(CMDLINE_MAX_BYTES as u64 + 1) + .read_to_end(&mut buf) + .map_err(Into::into) + }); + if read.is_err() { + return Vec::new(); + } + split_cmdline(&buf) +} + +fn split_cmdline(buf: &[u8]) -> Vec { + let cut = buf.len() > CMDLINE_MAX_BYTES; + let mut args: Vec = buf[..buf.len().min(CMDLINE_MAX_BYTES)] + .split(|b| *b == 0) + .filter(|a| !a.is_empty()) + .map(|a| String::from_utf8_lossy(a).into_owned()) + .collect(); + if cut { + if let Some(last) = args.last_mut() { + last.push_str("..."); + } + } + args +} + pub fn socket_owner( protocol: Protocol, direction: Direction, @@ -1006,6 +1039,23 @@ mod tests { // -- hex address formatting / parsing --------------------------------- + #[test] + fn cmdline_is_bounded_and_marks_a_cut() { + assert_eq!(split_cmdline(b"curl\0-s\0\0x\0"), ["curl", "-s", "x"]); + let mut long = b"prog\0".to_vec(); + long.resize(CMDLINE_MAX_BYTES * 4, b'a'); + let args = split_cmdline(&long); + assert_eq!(args[0], "prog"); + assert!(args[1].ends_with("...")); + assert_eq!( + args.iter().map(String::len).sum::(), + CMDLINE_MAX_BYTES - 1 + 3 + ); + // Our own process reads through the same path. + let me = ProcFsProcess::myself().unwrap(); + assert!(!read_cmdline(&me).is_empty()); + } + #[test] fn formats_ipv4() { let s = format_addr_port(IpAddr::V4(Ipv4Addr::new(127, 0, 0, 1)), 80); From a91e41d933b940c526f6bae761af5cb1c734f2ca Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:35 +0200 Subject: [PATCH 076/125] fix(daemon): strip " (deleted)" only from an image with no link left The kernel appends " (deleted)" to /proc//exe once the image is unlinked, and the resolver and the eBPF resync dropped that suffix by text alone so rules keep matching across an upgrade. A file literally named "curl (deleted)", for instance one mounted at /usr/bin/curl (deleted) in a user's own mount namespace, therefore passed for /usr/bin/curl and matched its path rules. The suffix is now dropped only when the mapped image's link count is zero; a linked file keeps its whole name. --- crates/cfc-daemon/src/ebpf/enforce.rs | 26 ++++------- crates/cfc-daemon/src/process_resolve.rs | 58 +++++++++++++++++------- 2 files changed, 49 insertions(+), 35 deletions(-) diff --git a/crates/cfc-daemon/src/ebpf/enforce.rs b/crates/cfc-daemon/src/ebpf/enforce.rs index 7e499c3..2341db6 100644 --- a/crates/cfc-daemon/src/ebpf/enforce.rs +++ b/crates/cfc-daemon/src/ebpf/enforce.rs @@ -490,15 +490,7 @@ impl VerdictSink { // the time this runs, whatever the exec program wrote for a // mismatched spelling is overwritten or cleared. The residual window // is the exec-to-consumer latency, and the packet path covers it. - let resolved = std::fs::read_link(format!("/proc/{pid}/exe")) - .ok() - .map(|exe| { - let s = exe.to_string_lossy(); - match s.strip_suffix(crate::process_resolve::DELETED_SUFFIX) { - Some(stripped) => std::path::PathBuf::from(stripped), - None => exe, - } - }); + let resolved = proc_exe(pid); // The uid here stays the event's, not a fresh read: at the moment of an // execve that *is* the process's uid, and a drop of privileges between // the kernel's tracepoint and this consumer is both vanishingly narrow @@ -656,16 +648,14 @@ impl VerdictSink { } } -/// `/proc//exe`, with the kernel's `" (deleted)"` suffix stripped. +/// `/proc//exe` as rules match it (`policy_exe_path`). `None` when the +/// process is gone or its image cannot be read. fn proc_exe(pid: u32) -> Option { - let exe = std::fs::read_link(format!("/proc/{pid}/exe")).ok()?; - let s = exe.to_string_lossy(); - Some( - match s.strip_suffix(crate::process_resolve::DELETED_SUFFIX) { - Some(stripped) => std::path::PathBuf::from(stripped), - None => exe, - }, - ) + use std::os::unix::fs::MetadataExt as _; + let link = format!("/proc/{pid}/exe"); + let path = std::fs::read_link(&link).ok()?; + let unlinked = std::fs::metadata(&link).ok()?.nlink() == 0; + Some(crate::process_resolve::policy_exe_path(path, unlinked)) } /// Everything a resync decision needs about one pid, read from /proc, plus the diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 052a2fd..78f87f0 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -217,7 +217,7 @@ fn resolve_inner( let image = image.and_then(|image| image.finish(&proc_exe_path)); let image = image.filter(|_| read_starttime(pid) == starttime); - let (exe, sha256) = match image { + let (exe, sha256, unlinked) = match image { Some(identity) => identity, None => { // The event filename is an exec argument, including aliases. @@ -225,7 +225,7 @@ fn resolve_inner( // or changed during resolution. Kernel credentials remain useful. cmdline.clear(); cwd = None; - (PathBuf::from(cfc_core::UNKNOWN_EXE), None) + (PathBuf::from(cfc_core::UNKNOWN_EXE), None, false) } }; @@ -251,14 +251,11 @@ fn resolve_inner( // digest below is of the *old* bytes, and comparing those against the new // file on disk would report `Modified` for every process running across an // upgrade - so it is passed the original. - let replaced_on_disk = exe.to_string_lossy().ends_with(DELETED_SUFFIX); + // + // Only an image with no link left is stripped (`policy_exe_path`): a file + // literally named "curl (deleted)" must not pass for "curl". let exe_for_provenance = exe.clone(); - let exe = if replaced_on_disk { - let s = exe.to_string_lossy(); - PathBuf::from(&s[..s.len() - DELETED_SUFFIX.len()]) - } else { - exe - }; + let exe = policy_exe_path(exe, unlinked); // Package provenance reuses the digest computed just above rather than // re-hashing. That digest comes from /proc/{pid}/exe -- the binary the @@ -819,6 +816,23 @@ fn parse_starttime(stat: &str) -> Option { /// gone, provenance wants to know it was there. pub(crate) const DELETED_SUFFIX: &str = " (deleted)"; +/// The path rules match for an image `/proc//exe` names `path`. +/// +/// The kernel appends [`DELETED_SUFFIX`] once the image has no link left, as +/// after a package upgrade under a running program; it is dropped so a rule +/// for the path keeps matching. A file that is merely named "x (deleted)" +/// still has a link and keeps its whole name, so it cannot pass for "x". +pub(crate) fn policy_exe_path(path: PathBuf, unlinked: bool) -> PathBuf { + let stripped = unlinked + .then(|| { + path.to_str()? + .strip_suffix(DELETED_SUFFIX) + .map(PathBuf::from) + }) + .flatten(); + stripped.unwrap_or(path) +} + /// One opened mapped image. Its link and metadata must still describe this /// file after hashing; otherwise no executable identity is published. struct MappedImage { @@ -842,7 +856,9 @@ impl MappedImage { Some(Self { file, path, key }) } - fn finish(self, link: &Path) -> Option<(PathBuf, Option)> { + /// The image's path as `/proc` renders it, its digest, and whether it has + /// no link left. + fn finish(self, link: &Path) -> Option<(PathBuf, Option, bool)> { let meta = self.file.metadata().ok()?; if image_key(&meta) != self.key { return None; @@ -860,7 +876,7 @@ impl MappedImage { { return None; } - Some((self.path, sha256)) + Some((self.path, sha256, meta.nlink() == 0)) } } @@ -1078,15 +1094,22 @@ mod tests { // path string, so leaving it there means the rule the user wrote - or // the one a prompt created - matches nothing. let raw = PathBuf::from("/usr/lib/firefox/firefox (deleted)"); - let s = raw.to_string_lossy(); - assert!(s.ends_with(DELETED_SUFFIX)); - let cleaned = PathBuf::from(&s[..s.len() - DELETED_SUFFIX.len()]); - assert_eq!(cleaned, PathBuf::from("/usr/lib/firefox/firefox")); + assert_eq!( + policy_exe_path(raw.clone(), true), + PathBuf::from("/usr/lib/firefox/firefox") + ); + + // A linked file literally named that keeps its name: it is not the + // firefox an upgrade replaced. + assert_eq!( + policy_exe_path(raw, false).to_str(), + Some("/usr/lib/firefox/firefox (deleted)") + ); // And a path that merely *contains* the words is left alone: the // suffix is a suffix, not a substring. let odd = PathBuf::from("/opt/my (deleted) app/bin"); - assert!(!odd.to_string_lossy().ends_with(DELETED_SUFFIX)); + assert_eq!(policy_exe_path(odd.clone(), true), odd); } #[test] @@ -1558,8 +1581,9 @@ mod tests { fs::write(&second, b"another image").unwrap(); symlink(&first, &link).unwrap(); - let (path, digest) = MappedImage::open(&link).unwrap().finish(&link).unwrap(); + let (path, digest, unlinked) = MappedImage::open(&link).unwrap().finish(&link).unwrap(); assert_eq!(path, first); + assert!(!unlinked); assert_eq!( digest.as_deref(), Some("b94d27b9934d3e08a52e52d7da7dabfac484efe37a5380ee9088f7ace2efcde9") From 837997f794686f12c92a5416119083669f1594a8 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:43 +0200 Subject: [PATCH 077/125] test(daemon): wait for the child's exec before resolving it spawn() can return before the child has exec'd sh, so under load the first resolve saw the test binary or an image mid-exec and the identity test failed. The test now waits until the child's image changes. --- crates/cfc-daemon/src/process_resolve.rs | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 78f87f0..58831e5 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -1607,6 +1607,13 @@ mod tests { .unwrap(); let pid = child.id(); let link = format!("/proc/{pid}/exe"); + // spawn() can return before the child has exec'd sh; until then its + // image is this test binary. + let me = std::env::current_exe().unwrap(); + let deadline = Instant::now() + Duration::from_secs(5); + while fs::read_link(&link).unwrap() == me && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(10)); + } let before = resolve(pid); assert_eq!(before.exe, fs::read_link(&link).unwrap()); From a65a4ad9cd338c3feaced7595e3406d30285e10b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:43 +0200 Subject: [PATCH 078/125] fix(storage): create the rule store private whatever the umask The store's directory and database were created with the inherited umask, so a daemon started by hand under the usual 022 left rules and other users' command lines world-readable, and under 000 writable. The packaged unit's UMask=0077 hid this. The directory is now created 0700 and the database 0600 (tightened if it already exists) before SQLite opens it; its -wal and -shm files copy that mode. --- crates/cfc-daemon/src/storage.rs | 40 +++++++++++++++++++++++++++++++- 1 file changed, 39 insertions(+), 1 deletion(-) diff --git a/crates/cfc-daemon/src/storage.rs b/crates/cfc-daemon/src/storage.rs index 8a9637d..b1deb09 100644 --- a/crates/cfc-daemon/src/storage.rs +++ b/crates/cfc-daemon/src/storage.rs @@ -10,6 +10,7 @@ use anyhow::Context; use cfc_core::{Duration as RuleDuration, Rule, RuleSet}; use parking_lot::Mutex; use rusqlite::Connection; +use std::os::unix::fs::{DirBuilderExt, OpenOptionsExt, PermissionsExt}; use std::path::Path; use std::sync::atomic::{AtomicUsize, Ordering}; use std::sync::Arc; @@ -176,9 +177,30 @@ fn tune(conn: &Connection, durable: bool) -> anyhow::Result<()> { impl RuleStore { pub fn open(path: &Path) -> anyhow::Result { + anyhow::ensure!( + path != Path::new(":memory:"), + "durable storage requires a database file" + ); + // Explicit modes: a daemon started by hand inherits the shell's umask, + // and SQLite creates the database as 0666 minus that umask, with the + // -wal and -shm files copying the database's mode. The rules and other + // users' command lines in it are root's alone. The packaged unit's + // UMask=0077 gives the same result. if let Some(parent) = path.parent() { - std::fs::create_dir_all(parent).ok(); + std::fs::DirBuilder::new() + .recursive(true) + .mode(0o700) + .create(parent) + .ok(); } + std::fs::OpenOptions::new() + .create(true) + .append(true) + .mode(0o600) + .open(path) + .with_context(|| format!("creating {}", path.display()))?; + std::fs::set_permissions(path, std::fs::Permissions::from_mode(0o600)) + .with_context(|| format!("restricting {}", path.display()))?; let conn = Connection::open(path).context("opening sqlite")?; let store = Self::from_conn(conn, true)?; @@ -652,6 +674,22 @@ mod tests { assert!(timeout > 0 && timeout <= 500, "busy timeout = {timeout}"); } + #[test] + fn store_files_are_private_whatever_the_umask() { + let dir = tempfile::tempdir().unwrap(); + let state = dir.path().join("state"); + let path = state.join("rules.db"); + drop(RuleStore::open(&path).unwrap()); + let mode = |p: &Path| std::fs::metadata(p).unwrap().permissions().mode() & 0o777; + assert_eq!(mode(&state), 0o700); + assert_eq!(mode(&path), 0o600); + + // A database left world-writable by an earlier run is tightened too. + std::fs::set_permissions(&path, std::fs::Permissions::from_mode(0o666)).unwrap(); + drop(RuleStore::open(&path).unwrap()); + assert_eq!(mode(&path), 0o600); + } + #[test] fn event_commit_bounds_contention_from_another_sqlite_writer() { let dir = tempfile::tempdir().unwrap(); From 73c0ee33ee373d03db01daa3847a6747d73a1e6f Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:43 +0200 Subject: [PATCH 079/125] fix(ebpf): read the BPF object that was vetted vet_object resolved symlinks and checked the target's ownership, then the loader read the configured path again, following the links anew. Whoever controls a link along the way could swap the file between the check and the read. The loader now reads the canonical path it vetted. --- crates/cfc-daemon/src/ebpf/loader.rs | 83 +++++++++++++++++----------- 1 file changed, 52 insertions(+), 31 deletions(-) diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index 5b952cb..1be3c69 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -72,9 +72,11 @@ use cfc_core::exe_path::{dir_is_sealed as dir_is_safe, file_is_sealed as file_is /// directory some ordinary user can rename, is a short path from "unprivileged /// local account" to "decides what the firewall believes". /// -/// Returns the offending path so the note can name it. Symlinks are resolved -/// first: vetting the link and loading the target would check the wrong file. -fn vet_object(path: &Path) -> anyhow::Result<()> { +/// The error names the offending path so the note can name it. Symlinks are +/// resolved first and the vetted target is returned for the caller to read: +/// vetting the target and reading through the link again would let whoever +/// controls a link along the way swap the file in between. +fn vet_object(path: &Path) -> anyhow::Result { use std::os::unix::fs::MetadataExt as _; let real = @@ -107,7 +109,7 @@ fn vet_object(path: &Path) -> anyhow::Result<()> { )); } } - Ok(()) + Ok(real) } /// Classifies a failure from `EbpfLoader::load` - parsing the ELF, creating @@ -221,38 +223,42 @@ pub(super) fn load_and_attach( // Vet before read, so a file we would refuse is never even pulled into // memory, and so the "not there at all" case is distinguishable from the // "there but not ours" one. - if let Err(e) = vet_object(object_path) { - // `NotFound` from canonicalize is the ordinary "no object installed" - // case, not a trust failure, and it must stay that way: under an - // automatic default it is the single most common outcome on earth and - // logging it as a security event would be noise. - let missing = errno_of(&e) == Some(libc::ENOENT); - if missing { - return Err(LoadError::new( - Degrade::ObjectMissing, - e.context(format!( - "no BPF object at {} (build it with `cargo xtask build-ebpf` \ + let read_path = match vet_object(object_path) { + Ok(real) => real, + Err(e) => { + // `NotFound` from canonicalize is the ordinary "no object installed" + // case, not a trust failure, and it must stay that way: under an + // automatic default it is the single most common outcome on earth and + // logging it as a security event would be noise. + let missing = errno_of(&e) == Some(libc::ENOENT); + if missing { + return Err(LoadError::new( + Degrade::ObjectMissing, + e.context(format!( + "no BPF object at {} (build it with `cargo xtask build-ebpf` \ and install it there, or set [ebpf] object_path)", - object_path.display() - )), - )); - } - match trust { - Trust::Refuse => { - return Err(LoadError::new(Degrade::ObjectUntrusted, e)); + object_path.display() + )), + )); } - // Somebody pointed the daemon at this file on purpose. Say what is - // wrong with it and do as asked. - Trust::Warn => { - warn!("loading an unvetted BPF object because it was configured explicitly: {e:#}"); - report - .notes - .push(format!("BPF object failed its ownership check: {e:#}")); + match trust { + Trust::Refuse => { + return Err(LoadError::new(Degrade::ObjectUntrusted, e)); + } + // Somebody pointed the daemon at this file on purpose. Say what is + // wrong with it and do as asked. + Trust::Warn => { + warn!("loading an unvetted BPF object because it was configured explicitly: {e:#}"); + report + .notes + .push(format!("BPF object failed its ownership check: {e:#}")); + } } + object_path.to_path_buf() } - } + }; - let object = std::fs::read(object_path).map_err(|e| { + let object = std::fs::read(&read_path).map_err(|e| { let degrade = if e.kind() == std::io::ErrorKind::NotFound { Degrade::ObjectMissing } else { @@ -1499,6 +1505,21 @@ mod tests { ); } + #[test] + fn the_vetted_target_is_what_gets_read() { + // Any root-sealed file stands in for an installed object. + let Ok(real) = vet_object(Path::new("/usr/bin/env")) else { + return; + }; + let dir = tempfile::tempdir().expect("tempdir"); + let link = dir.path().join("cfc-ebpf.o"); + std::os::unix::fs::symlink(&real, &link).expect("symlink"); + // The link sits in a directory its owner can rewrite; the path handed + // back is the target, so swapping the link after vetting changes + // nothing that is read. + assert_eq!(vet_object(&link).expect("vetted"), real); + } + #[test] fn an_absent_object_is_missing_not_untrusted() { let dir = tempfile::tempdir().expect("tempdir"); From a4c44e13f4567abaeef268b16b9009ec8b4433ae Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:43 +0200 Subject: [PATCH 080/125] fix(ui): keep an edited rule's executable path verbatim Saving a rule trimmed the executable field, so a rule for a file whose name ends in a space was retargeted to another path, or became unsaveable, on any edit. The path is now kept as typed; only an all-blank field still counts as no executable. --- crates/cfc-ui/src/main.rs | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 1c8321a..a8afddf 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -1801,8 +1801,10 @@ fn build_rule_from_editor(ed: &RuleEditor) -> Result { // Validate in the user's namespace as well as the daemon's: ProtectHome // and PrivateTmp hide exactly the paths a person commonly enters. - let typed = ed.exe.trim(); - let exe = if typed.is_empty() { + // Not trimmed: a file name may end in a space, and saving a rule for + // "/opt/app " must not quietly retarget it to "/opt/app". + let typed = ed.exe.as_str(); + let exe = if typed.trim().is_empty() { String::new() } else { cfc_core::exe_path::resolve_policy(std::path::Path::new(typed))? @@ -1962,6 +1964,14 @@ mod tests { assert!(build_rule_from_editor(&ed).is_ok()); } + #[test] + fn editor_keeps_an_executable_path_verbatim() { + let mut ed = editor_with_scope(); + ed.exe = "/usr/bin/cfc-test-app ".into(); + let rule = build_rule_from_editor(&ed).unwrap(); + assert_eq!(rule.scope.unwrap().exe_path, "/usr/bin/cfc-test-app "); + } + #[test] fn editor_builds_a_persistable_rule() { let rule = build_rule_from_editor(&editor_with_scope()).unwrap(); From c326d61bb9fa7195a2c827fc08fc7988a5250335 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:51 +0200 Subject: [PATCH 081/125] fix: build and run the demo in private temporary directories The Arch build recipes copied the PKGBUILD into a fixed /tmp/cfc-build (one with mkdir -p, which accepts a directory another user created first) and the prompt demo kept its store and socket in /tmp/cfc-demo, which another local user could create or point elsewhere with a symlink. Both now use a fresh mktemp directory; the demo logs its socket path. --- README.md | 8 ++++---- crates/cfc-daemon/examples/prompt_demo.rs | 22 +++++++++++++--------- pkg/PKGBUILD-git | 6 +++--- pkg/README.md | 4 ++-- 4 files changed, 22 insertions(+), 18 deletions(-) diff --git a/README.md b/README.md index d61d020..9f7d66d 100644 --- a/README.md +++ b/README.md @@ -103,10 +103,10 @@ Not on the AUR yet. Two recipes ship in `pkg/`; the `-git` one works today, without a published release: ```sh -mkdir -p /tmp/cfc-build -cp pkg/PKGBUILD-git /tmp/cfc-build/PKGBUILD -cp pkg/colony-firewall-control.install /tmp/cfc-build/ -cd /tmp/cfc-build && makepkg -si +build=$(mktemp -d) +cp pkg/PKGBUILD-git "$build"/PKGBUILD +cp pkg/colony-firewall-control.install "$build"/ +cd "$build" && makepkg -si ``` `pkg/PKGBUILD` is the AUR release recipe instead: it builds from the diff --git a/crates/cfc-daemon/examples/prompt_demo.rs b/crates/cfc-daemon/examples/prompt_demo.rs index 55b566e..fd6f033 100644 --- a/crates/cfc-daemon/examples/prompt_demo.rs +++ b/crates/cfc-daemon/examples/prompt_demo.rs @@ -7,9 +7,12 @@ //! //! ```sh //! cargo run -p cfc-daemon --example prompt_demo -//! CFC_SOCKET=/tmp/cfc-demo/cfc.sock colony-firewall-tray # or the GUI/cfc +//! CFC_SOCKET= colony-firewall-tray # or the GUI/cfc //! ``` //! +//! The store and socket live in a fresh private temporary directory; the +//! socket path is logged at startup. +//! //! Verdicts coming back over the worker channel are printed, so you can see //! a notification button click arrive where the NFQUEUE worker would //! normally apply it. Packets are imaginary; nothing is filtered. @@ -28,8 +31,6 @@ use std::path::PathBuf; use std::sync::Arc; use tokio::sync::broadcast; -const SOCKET: &str = "/tmp/cfc-demo/cfc.sock"; - /// A rotating cast of pretend applications, so successive prompts look /// different in the notification. const CAST: &[(&str, &str, u16)] = &[ @@ -46,10 +47,12 @@ async fn main() -> anyhow::Result<()> { .with_target(false) .init(); - let dir = PathBuf::from("/tmp/cfc-demo"); - std::fs::create_dir_all(&dir)?; + // mkdtemp, not a fixed /tmp name another user could create first or + // point somewhere else with a symlink. + let dir = tempfile::Builder::new().prefix("cfc-demo-").tempdir()?; + let socket = dir.path().join("cfc.sock"); - let store = RuleStore::open(&dir.join("rules.db"))?; + let store = RuleStore::open(&dir.path().join("rules.db"))?; // 45s per prompt: enough time to read a notification and pick a // button. Timeout denies, like every shipped profile — an unanswered // question is not consent. @@ -68,10 +71,11 @@ async fn main() -> anyhow::Result<()> { let (_ipc, prompt_tx) = ipc::spawn( IpcOptions { - socket_path: PathBuf::from(SOCKET), + socket_path: socket.clone(), ipc: IpcConfig { group: "colony-firewall".into(), - // Demo socket in /tmp: let the invoking user talk to it. + // Demo socket in the user's own temporary directory: let the + // invoking user talk to it. require_group: false, }, pause_default_secs: 120, @@ -101,7 +105,7 @@ async fn main() -> anyhow::Result<()> { let uid = nix::unistd::Uid::current().as_raw(); tracing::info!( - socket = SOCKET, + socket = %socket.display(), uid, "demo daemon up - a prompt fires every 25s" ); diff --git a/pkg/PKGBUILD-git b/pkg/PKGBUILD-git index 1f06fe8..b6e4add 100644 --- a/pkg/PKGBUILD-git +++ b/pkg/PKGBUILD-git @@ -3,9 +3,9 @@ # Development / VCS variant of colony-firewall-control. # # Usage (from anywhere): -# mkdir /tmp/cfc-git && cp pkg/PKGBUILD-git /tmp/cfc-git/PKGBUILD -# cp pkg/colony-firewall-control.install /tmp/cfc-git/ -# cd /tmp/cfc-git && makepkg -si +# build=$(mktemp -d) && cp pkg/PKGBUILD-git "$build"/PKGBUILD +# cp pkg/colony-firewall-control.install "$build"/ +# cd "$build" && makepkg -si # # To build from your local checkout instead of GitHub (fast iteration, # uses your working tree's HEAD), point _giturl at the repo root, e.g.: diff --git a/pkg/README.md b/pkg/README.md index 75dbabc..19cf85f 100644 --- a/pkg/README.md +++ b/pkg/README.md @@ -83,8 +83,8 @@ Per release, by hand: cd pkg && updpkgsums PKGBUILD # from pacman-contrib grep '^sha256sums=' PKGBUILD # must no longer say SKIP # 4. Test build (needs colony-firewall-control.install next to PKGBUILD): -mkdir /tmp/cfc-build && cp PKGBUILD colony-firewall-control.install /tmp/cfc-build/ -cd /tmp/cfc-build && makepkg -si +build=$(mktemp -d) && cp PKGBUILD colony-firewall-control.install "$build"/ +cd "$build" && makepkg -si namcap PKGBUILD ./*.pkg.tar.zst # no E: lines # 5. Regenerate .SRCINFO and push to the AUR: makepkg --printsrcinfo > .SRCINFO From b01a8bf2e0c57ded24d8ebf2e0f7013cc22a1976 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:10:51 +0200 Subject: [PATCH 082/125] docs(hardening): inbound replies, deleted images and FUSE images Three limits were not written down: with inbound filtering off, a program outbound rules deny still answers connections a remote peer opens; an image deleted while it runs is matched by its former path, which a mount namespace can also present; and digests and the root-sealed test trust what a FUSE filesystem the daemon can read reports. --- README.md | 3 +++ docs/HARDENING.md | 10 ++++++++++ 2 files changed, 13 insertions(+) diff --git a/README.md b/README.md index 9f7d66d..e85339b 100644 --- a/README.md +++ b/README.md @@ -280,6 +280,9 @@ managers, or a later external ruleset flush. Early unmatched flows use **Scope.** Normal mode decides new tracked IP flows from socket attribution; established and related traffic retains its connection-wide authorization. +With inbound filtering off (the default) a connection a remote peer opens is +never judged at all, so a program that outbound rules deny still answers on +any port it listens on. Passed or inherited sockets are not reauthorized for each sending executable. A current descriptor holder does not prove which process sent a packet. While the daemon runs, new direct loopback flows follow explicit rules; diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 30959d6..1e67088 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -202,6 +202,10 @@ is a separate launch mode. remote flows delegated through AF_UNIX or D-Bus brokers. Existing local connections retain their authorization. Hostname rules and observed answers do not isolate DNS queries. +- **Inbound-initiated connections**: with inbound filtering off (the + default) a connection a remote peer opens is never queued: its replies are + established traffic. A program that outbound rules deny still answers on + any port it listens on. Enable inbound filtering to decide those flows. - **Inherited or passed sockets**: established/related traffic keeps its connection-wide authorization. An inherited or passed descriptor is not reauthorized for each sending executable. Current descriptor ownership @@ -217,6 +221,12 @@ is a separate launch mode. matched from a mount namespace, so pin its hash. Code already running as a user can also borrow an allowed program's identity by running it with chosen arguments or with `LD_PRELOAD`, which a hash pin does not prevent. + An image deleted while it runs is matched by the path it had (an upgraded + program keeps its rules), which the daemon cannot check against anything, + so a mount namespace can present a path that way too. Digests and the + root-sealed test trust what the filesystem reports: on a FUSE filesystem + the daemon can read (`user_allow_other` in `/etc/fuse.conf`), the user who + mounted it controls both. - **Raw and packet sockets**: applications with `CAP_NET_RAW` can use AF_PACKET outside the shipped `inet OUTPUT` hook. Raw IP packets can coincide with another socket's tuple even when TCP matching is strict. Tuple and inode From a702b643b0319925cfa697eb39d75ce17be19846 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:11:05 +0200 Subject: [PATCH 083/125] docs(changelog): list the remaining audit fixes in this bundle --- CHANGELOG.md | 23 +++++++++++++++++++++++ 1 file changed, 23 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 6b73ce5..acddccf 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -104,6 +104,20 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). attestation; `SECURITY.md` explains how to verify it and what `SHA256SUMS` and the attached `PKGBUILD` checksum do not prove. +- A running program whose file was literally named `curl (deleted)`, for + instance in a user's own mount namespace, matched the rules for `curl`: + the kernel's `" (deleted)"` suffix was dropped by text alone. It is now + dropped only from an image with no link left. +- A daemon started by hand created its rule store with the shell's umask, so + rules and other users' command lines were world-readable (or writable + under umask 000). The store directory is now created 0700 and the database + 0600. +- The BPF object was vetted through its symlinks and then read through them + again, so whoever controlled a link could swap it in between. The vetted + target is what gets read now. +- The Arch build recipes and the prompt demo used fixed `/tmp` directories + another local user could create first; they use `mktemp -d` now. + ### Removed - The Fast Allow userspace path, disabled since 0.7.0 because a socket mark @@ -243,6 +257,15 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). same checks as `ebpf.yml`. A dispatched release's draft now tags the commit it was built from. +- A process's arguments were read whole, up to several MiB, and copied into + every prompt, observation and client message. At most 4 KiB is kept now; + a cut argument ends in `...`. +- GUI: saving a rule trimmed its executable path, retargeting a rule for a + file whose name ends in a space. +- HARDENING.md says that with inbound filtering off a program outbound rules + deny still answers inbound connections, that a deleted image is matched by + its former path, and what a readable FUSE filesystem controls. + ## [0.7.0] - 2026-09-30 ### Added From 79c023a4da554e3ea26a7b56c75c38686b1133f4 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:17:33 +0200 Subject: [PATCH 084/125] docs(troubleshooting): drop the port 22 exemption from the SSH advice The guide said a dropped SSH session could not be reopened and told remote administrators to load a `policy accept` output chain with `tcp dport 22 accept`. Inbound SSH replies are established traffic and never queued, so the exemption did nothing for reaching the box, while it let every process reach any host on port 22 unjudged, accepted INVALID and UNTRACKED packets and dropped the loopback rule. The section now names the real lockout risks: the opt-in inbound table, which drops new sessions while the daemon is down, and outbound lookups the login makes. The dead-man's switch and console recovery remove both tables. README and HARDENING agree with it. --- README.md | 21 +++++---- docs/HARDENING.md | 4 +- docs/TROUBLESHOOTING.md | 100 ++++++++++++++++++++++------------------ 3 files changed, 72 insertions(+), 53 deletions(-) diff --git a/README.md b/README.md index e85339b..d20ec5e 100644 --- a/README.md +++ b/README.md @@ -254,9 +254,12 @@ box, not a passing one, and answering it with an allow would mean those hosts had no outbound firewall whatsoever. Stored rules are what such a machine runs on; `cfc prompts` is how you add more without a GUI. -This cannot lock you out of a remote machine: the ruleset hooks `output` -on `ct state new` only, so an inbound SSH session's replies are -`ct state established` and are never queued. +This does not refuse inbound SSH: the ruleset hooks `output` on +`ct state new` only, so an inbound SSH session's replies are +`ct state established` and are never queued. Outbound lookups the login +itself makes (reverse DNS, LDAP or Kerberos from `sshd`'s PAM stack) are +new flows and are judged like any other; see +[docs/TROUBLESHOOTING.md](docs/TROUBLESHOOTING.md#testing-over-ssh-without-locking-yourself-out). **Boot behaviour.** The nft units load independently before the daemon, `network-pre.target`, NetworkManager and systemd-networkd, after the @@ -310,11 +313,13 @@ cfc status # "enforcing yes", and it warns on stderr when it is not > **WARNING - remote / SSH machines:** the shipped nftables snippet is > fail-closed for everything except new loopback flows, which are allowed > while no daemon listens. If the daemon is down while the rule is loaded, -> **all new non-loopback outbound connections drop**, and a mistake can lock you out of a box you -> only reach over SSH. Read -> [docs/TROUBLESHOOTING.md](docs/TROUBLESHOOTING.md) - specifically the -> SSH exemption and dead-man's-switch patterns - *before* enabling -> enforcement remotely. +> **all new non-loopback outbound connections drop**. Inbound SSH still +> connects, but a login that needs the network (LDAP, Kerberos, reverse DNS) +> can fail, and the opt-in inbound table drops new SSH sessions outright +> while the daemon is down. Read +> [docs/TROUBLESHOOTING.md](docs/TROUBLESHOOTING.md#testing-over-ssh-without-locking-yourself-out) - +> specifically the dead-man's switch - *before* enabling enforcement +> remotely. ### Explicit application confinement diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 1e67088..34c7ed9 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -471,7 +471,9 @@ The other half of the security posture is the nftables side, not the daemon: whether the kernel drops or accepts new connections when nobody is answering the queue. The shipped snippet is fail-closed for everything except new loopback flows, which are allowed while no daemon listens. That -is the safer default and also the one that can lock you out of a remote box. +is the safer default, and also the one that cuts a box off from everything +it reaches out to, including any network lookup an SSH login needs (the +opt-in inbound table drops new SSH sessions outright while the daemon is down). The full matrix - daemon up or down, table loaded or not, with and without `bypass` - is in [TROUBLESHOOTING.md](TROUBLESHOOTING.md#fail-open-vs-fail-closed-matrix). diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 56a5140..9019bcd 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -2,8 +2,9 @@ The failure modes of an outbound firewall are unusually punishing: when it breaks, *the network* breaks, and the tool you'd use to debug it may be on -the other side of the connection it just dropped. Read the first section -before enabling enforcement on any machine you reach over SSH. +the other side of the connection it just dropped. Read +[Testing over SSH](#testing-over-ssh-without-locking-yourself-out) before +enabling enforcement on any machine you reach over SSH. ## Daemon restarts and rule upgrades @@ -53,56 +54,64 @@ this check. `CFC_INBOUND_FORCE=1` remains the explicit console override. ## Testing over SSH without locking yourself out -The shipped nftables snippet is **fail-closed for everything except new -loopback flows, which are allowed while no daemon listens**: the final -`queue num 0` without the `bypass` keyword means that if nothing is -listening on NFQUEUE 0 (daemon stopped, crashed, or not yet started), the -kernel drops every *new* non-loopback outbound connection. Only the rule -just above it, `oifname "lo" ct state new queue num 0 bypass`, lets new -loopback flows through in that state. Your established SSH session -survives (`ct state new` only matches new flows), but the moment it drops -you cannot open a new one. - -Three layers of protection, use all of them the first time: - -**1. Allow SSH above the queue rule.** In a local copy of the snippet -(see [Changing the shipped ruleset](#changing-the-shipped-ruleset)), add one -line above the two queue rules of `chain output` so port 22 never reaches -NFQUEUE at all: - -``` - tcp dport 22 accept - oifname "lo" ct state new queue num 0 bypass - ct state new queue num 0 -``` - -(This exempts *outbound* SSH from filtering - for a remote machine you -manage, also make sure your *inbound* SSH path doesn't depend on any -process this firewall could deny, e.g. a DNS lookup in `sshd`'s PAM stack.) - -**2. Arm a dead-man's switch BEFORE applying the rules.** In a detached +The outbound table cannot refuse a new inbound SSH session. It hooks +`output` and queues only `ct state new`, and everything `sshd` sends your +client is a reply on a connection the client opened, so it is +`ct state established` and never queued. That holds while the daemon is +down too. Two things can still cut you off: + +- **The inbound table** (`colony-firewall-nft-inbound`, opt-in). It queues + every new inbound connection, SSH included, and only an inbound Allow rule + admits one. Its final `queue num 0` has no `bypass`, so while no daemon + listens it drops every new inbound connection whatever the rules say. The + session you enabled it from survives; the next one does not. Its lockout + guard (see above) refuses to load the table when no inbound Allow rule + could admit an established session, but it does not check the rule's + source network, and no rule admits anything while the daemon is down. +- **Outbound lookups your login makes.** `sshd` and its PAM and NSS stack + can open new outbound flows while you log in: reverse DNS with + `UseDNS yes`, an LDAP, Kerberos, SSSD or RADIUS server. They are root + processes, so with no root prompt subscriber they get `no_ui_action` at + once (a denial under every profile), and while the daemon is down they + drop. Accounts that resolve locally are unaffected. Find these flows with + `sudo cfc prompts` or `cfc log --action deny` and allow each one scoped to + its program and server, for example + `cfc rules add --exe --dst-net --dst-port --protocol tcp --name login-ldap`. + +Do not exempt port 22 in the outbound chain. It does nothing for reaching +the box, and `tcp dport 22 accept` lets every process on the host, attributed +or not, reach any address on that port without a verdict or a log line. + +Use both of these the first time: + +**1. Arm a dead-man's switch BEFORE applying the rules.** In a detached shell that survives your SSH session: ```sh -sudo setsid sh -c 'sleep 300 && nft delete table inet colony_firewall' & +sudo setsid sh -c 'sleep 300; nft delete table inet colony_firewall_inbound; nft delete table inet colony_firewall' & ``` Then enable enforcement. If you still have connectivity after testing, -cancel the timer (`sudo pkill -f 'nft delete table'`, or just -`sudo systemctl reload colony-firewall-nft` after the timer fires). If you locked yourself out, wait out the -five minutes and the table deletes itself. - -**3. Know the console recovery.** From a local console, serial console, or +cancel the timer (`sudo pkill -f '[n]ft delete table'`; the brackets keep +the pattern from matching the `sudo` running it), or, after it fired, +`sudo systemctl reload colony-firewall-nft` (and +`colony-firewall-nft-inbound` if it is enabled) to load the tables again. +If you locked yourself out, wait out the five minutes and both tables +delete themselves. Deleting the inbound table fails harmlessly when it was +never loaded. + +**2. Know the console recovery.** From a local console, serial console, or your VPS provider's emergency shell: ```sh -nft delete table inet colony_firewall # stop enqueueing entirely +nft delete table inet colony_firewall_inbound # admit inbound again +nft delete table inet colony_firewall # stop enqueueing outbound # or -systemctl start colony-firewalld # give the queue a consumer again +systemctl start colony-firewalld # give the queue a consumer again ``` -Either one restores traffic; the first disables enforcement, the second -resumes it. +Deleting the tables disables enforcement; starting the daemon resumes it, +and with it any inbound Allow rule you wrote. ## No network after enabling @@ -116,7 +125,8 @@ cfc status ``` If `systemctl` shows the unit dead while the nftables rule is loaded, you -are in the fail-closed state described above: non-loopback packets are +are in the fail-closed state (see the +[matrix](#fail-open-vs-fail-closed-matrix)): new non-loopback flows are queued to NFQUEUE 0 and nobody answers. Start the daemon or delete the table. **Is the nftables table actually loaded?** @@ -436,9 +446,11 @@ cfc status # prompt policy 30s timeout -> Deny, no UI -> Deny ``` -Inbound SSH is unaffected — the ruleset hooks `output` on `ct state new`, -and an established session's replies are never queued — so you always -have a way back in to fix it. +Inbound SSH is unaffected: the outbound ruleset never queues a session's +replies, and the opt-in inbound table judges by your inbound rules and +`inbound_action`, not `no_ui_action`. A login that needs the network (LDAP, +Kerberos, reverse DNS) is the exception; see +[Testing over SSH](#testing-over-ssh-without-locking-yourself-out). Then pick one of three fixes: From 953035d0678a846623be8156ae7b32c428ba7e21 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:17:37 +0200 Subject: [PATCH 085/125] docs: state the hashing limit and other residual limits Executables over 64 MiB are never hashed: a hash-pinned rule naming one refuses its flows, and outside a root-sealed path "Allow always" cannot be saved, so such a program prompts for every new flow. README and HARDENING now say so and give the workarounds. HARDENING also notes that Docker grants CAP_NET_RAW by default, and its recovery steps use `cfc pause` instead of a profile switch, which changes nothing since every profile denies. SECURITY.md links the documented non-goals. The unit comment no longer says ProtectKernelTunables is on. --- README.md | 12 +++++++++++- SECURITY.md | 8 ++++++++ docs/HARDENING.md | 26 ++++++++++++++++++++++++-- systemd/colony-firewalld.service | 7 ++++--- 4 files changed, 47 insertions(+), 6 deletions(-) diff --git a/README.md b/README.md index d20ec5e..e78d493 100644 --- a/README.md +++ b/README.md @@ -299,7 +299,17 @@ remote flows delegated through local brokers, including AF_UNIX and D-Bus. Applications with `CAP_NET_RAW` can use AF_PACKET outside the `inet OUTPUT` hook. Raw IP packets can also coincide with another socket's tuple; socket attribution does not prove their origin. Use explicit application confinement -or OS containment for those cases. +or OS containment for those cases. Docker grants `CAP_NET_RAW` to containers +by default, so a `--network=host` container has it; drop it with +`--cap-drop NET_RAW` for workloads CFC should govern. + +Executables over 64 MiB (Chromium, Electron apps, VS Code) are never hashed. +A hash-pinned rule naming one refuses its flows, and outside a root-owned +path an "Allow always" for one cannot be saved, so it prompts for every new +flow. See [docs/HARDENING.md](docs/HARDENING.md#rule-design-principles). +The complete list of non-goals is in +[docs/HARDENING.md](docs/HARDENING.md#what-this-firewall-does-not-protect-against). + Fast Allow was removed: a socket mark cannot prove which process sends, so it opened bypasses. The old `[ebpf] fast_allow` and `fast_allow_mark` keys are ignored with a warning, and allowed flows use the normal NFQUEUE path. diff --git a/SECURITY.md b/SECURITY.md index 715bf94..560efde 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -28,6 +28,14 @@ Please do **not** open a public issue for anything you believe is exploitable (privilege escalation via the daemon, rule-bypass of the NFQUEUE filter, crafted-packet parsing crashes, socket permission problems, etc.). +Out of scope: the limits documented in +[What this firewall does not protect against](docs/HARDENING.md#what-this-firewall-does-not-protect-against), +such as root processes, established or inherited flows, replies on +inbound-initiated connections while inbound filtering is off, local relays, +AF_PACKET and raw sockets, and code running as an allowed program's user. +A way around the firewall that this list does not describe, or a +description that turns out to be wrong, is in scope. + What to expect: - This is a single-maintainer hobby project; acknowledgement is best-effort, diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 34c7ed9..4096242 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -178,6 +178,23 @@ with zero hits after weeks of use is probably obsolete or wrong. different binary after an interpreter upgrade. When in doubt, target the real path under `/usr/lib/...` or pin by SHA-256 (`scope.exe_sha256`). +**Executables over 64 MiB have no digest.** The daemon does not hash an +image larger than 64 MiB, and Chromium, Electron apps and VS Code are +usually past it. For such a program: + +- `cfc rules add --pin-hash` refuses the file. A digest supplied another way + (`--sha256`, an import) is stored but can never be compared, so a rule + carrying one, Allow or Deny, refuses the program's flows wherever its other + fields match, and no rule below it can allow them. +- On a root-sealed path (root-owned, with root-owned ancestors, as a package + installs it) nothing else changes: "Allow always" saves a path-only rule. +- On any other path (under a home directory, a user-writable `/opt` + tree), an Allow cannot be bound to the image, so "Allow always" applies + once and saves no rule (the UI says why), and every new flow, each + retransmit included, opens its own prompt. A hand-written path-only rule + (`cfc rules add --exe `) works, but whoever can write that file + inherits it. Installing the program root-owned is the better fix. + ## What this firewall does *not* protect against Normal mode follows the desktop application firewall model of OpenSnitch and @@ -231,6 +248,9 @@ is a separate launch mode. outside the shipped `inet OUTPUT` hook. Raw IP packets can coincide with another socket's tuple even when TCP matching is strict. Tuple and inode checks do not prove raw packet provenance or provide layer-2 containment. + Docker grants `CAP_NET_RAW` to containers by default, so a + `--network=host` container can send frames on the host's interfaces this + way; run workloads CFC should govern with `--cap-drop NET_RAW`. - **DNS-over-HTTPS embedded in browsers**: the firewall sees the outer HTTPS flow. Domain isolation requires an application-aware proxy or separate containment. - **Container traffic**: Docker / Podman / LXC route through their own @@ -488,8 +508,10 @@ uses it only on the loopback rule (`oifname "lo"`). Order of operations: -1. Switch profile back to `balanced` so the daemon stops actively denying - things while you debug. +1. `cfc pause --for 15m`: unmatched outbound flows pass instead of being + denied while you debug, explicit rules still apply, and it resumes on its + own. Every profile denies unmatched flows, so switching profile changes + nothing. 2. `cfc live` and reproduce the failure - the deny verdict will show in real time. 3. `cfc rules list | grep ` - is the rule too narrow? diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index b7102b7..459e339 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -94,9 +94,10 @@ NoNewPrivileges=true # Filesystem ProtectSystem=strict ProtectHome=true -# /sys/fs/bpf is listed because ProtectKernelTunables below remounts /sys -# read-only, and the enforcement links are *pinned* into the bpffs mounted -# there. Without this line the pin fails with EROFS, the daemon falls back to +# /sys/fs/bpf must stay writable: the enforcement links are *pinned* into the +# bpffs mounted there. This line alone does not achieve that against +# ProtectKernelTunables, which is why that setting is replaced by hand below. +# A read-only bpffs fails the pin with EROFS, the daemon falls back to # process-lifetime links, and `kill -9` on this unit lifts every in-kernel deny # it holds - which is the one thing that layer exists to prevent. The daemon # only ever creates /sys/fs/bpf/colony-firewall/; it writes nothing else there. From e67e3df1750406a0e4feecdf8130e28a155c21e1 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 03:17:37 +0200 Subject: [PATCH 086/125] docs(changelog): list the documentation corrections Records the SSH guidance rewrite and the newly documented limits. --- CHANGELOG.md | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index acddccf..5ba39c7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -265,6 +265,18 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - HARDENING.md says that with inbound filtering off a program outbound rules deny still answers inbound connections, that a deleted image is matched by its former path, and what a readable FUSE filesystem controls. +- TROUBLESHOOTING.md told remote administrators to add `tcp dport 22 accept` + to a `policy accept` copy of the outbound chain. That let every process + reach any host on port 22 unjudged, accepted INVALID and UNTRACKED traffic, + dropped the loopback rule, and did nothing for reaching the box, since + inbound SSH replies are never queued. The guide now names the real lockout + risks (the inbound table, network lookups the login makes) and its + dead-man's switch removes both tables. +- The docs now say what the 64 MiB hashing limit costs (hash-pinned rules + refuse such a program; outside a root-owned path "Allow always" is not + saved), that Docker grants `CAP_NET_RAW` by default, and that `cfc pause`, + not a profile switch, lets unmatched flows through while debugging. + SECURITY.md links the documented non-goals. ## [0.7.0] - 2026-09-30 From b9b982abbe0ebe969476a5d4377ccde949e4c18a Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:27:43 +0200 Subject: [PATCH 087/125] fix(daemon): keep " (deleted)" for a process in another user namespace A process in its own user and mount namespace could mount its bytes at /usr/bin/curl, run them and delete them. The kernel then reports "/usr/bin/curl (deleted)", a path that names no file to check the image against, and stripping the suffix handed it every path-only rule for the host's curl. The suffix is now dropped only for a process in the daemon's own user namespace, where only root sets up mounts. A program in another one keeps the suffix after its own upgrade until it restarts. --- CHANGELOG.md | 6 ++ crates/cfc-daemon/src/process_resolve.rs | 95 +++++++++++++++++++++++- docs/ARCHITECTURE.md | 4 +- docs/HARDENING.md | 8 +- 4 files changed, 109 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5ba39c7..72b3193 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -177,6 +177,12 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). different file in the daemon's view is now reported as unknown. Container and Flatpak runtime binaries at such paths therefore lose their executable identity instead of borrowing the host's. +- The same namespace could still borrow the host's path by deleting its + bytes once running: the kernel's `" (deleted)"` suffix was dropped, and a + deleted image names no file to compare with. The suffix is now dropped only + for a process in the daemon's user namespace, so a program in another one + (`unshare -U`, a rootless container) that runs across its own upgrade + matches its rules again only after a restart. - Executables over 64 MiB, such as Chromium, Electron apps and VS Code, have no digest, so every queued packet from them, each retransmit and parallel connection, opened its own prompt. On a root-sealed path they now share one diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 58831e5..36f25f8 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -215,7 +215,10 @@ fn resolve_inner( } }; - let image = image.and_then(|image| image.finish(&proc_exe_path)); + // Read before the start time check, so a recycled pid cannot answer. + let image = image + .and_then(|image| image.finish(&proc_exe_path)) + .map(|(exe, sha256, unlinked)| (exe, sha256, unlinked && in_our_user_namespace(pid))); let image = image.filter(|_| read_starttime(pid) == starttime); let (exe, sha256, unlinked) = match image { Some(identity) => identity, @@ -254,6 +257,13 @@ fn resolve_inner( // // Only an image with no link left is stripped (`policy_exe_path`): a file // literally named "curl (deleted)" must not pass for "curl". + // + // Nor is it stripped for a process in another user namespace. Such a + // process can mount its own bytes at /usr/bin/curl, run them and remove + // them; the former path then names no file to compare the image with + // (`path_names_image`), and stripping would hand it the host curl's + // rules. In our user namespace only root, or a setuid helper root + // installed, sets up mounts. let exe_for_provenance = exe.clone(); let exe = policy_exe_path(exe, unlinked); @@ -895,6 +905,18 @@ fn path_names_image(path: &Path, key: &ImageKey) -> bool { fs::metadata(path).map_or(true, |m| (m.dev(), m.ino()) == (key.0, key.1)) } +/// Whether `pid` runs in the daemon's user namespace. False when either link +/// cannot be read. +fn in_our_user_namespace(pid: u32) -> bool { + match ( + fs::read_link(format!("/proc/{pid}/ns/user")), + fs::read_link("/proc/self/ns/user"), + ) { + (Ok(theirs), Ok(ours)) => theirs == ours, + _ => false, + } +} + fn image_key(meta: &fs::Metadata) -> ImageKey { ( meta.dev(), @@ -1633,6 +1655,77 @@ mod tests { assert_ne!(after.sha256, before.sha256); } + /// Runs a fresh copy of `sleep` under `wrapper`, deletes the copy once it + /// is mapped and resolves the process. `None` when it never got mapped. + fn resolve_deleted_image(sleep: &Path, image: &Path, wrapper: &[&str]) -> Option { + fs::copy(sleep, image).unwrap(); + let mut command = match wrapper.split_first() { + Some((program, args)) => { + let mut c = std::process::Command::new(program); + c.args(args).arg(image); + c + } + None => std::process::Command::new(image), + }; + command.arg("30"); + // Another test thread's fork() can briefly hold the new copy open for + // writing. + let mut child = (0..20) + .find_map(|_| match command.spawn() { + Err(e) if e.raw_os_error() == Some(libc::ETXTBSY) => { + std::thread::sleep(Duration::from_millis(50)); + None + } + other => Some(other), + })? + .ok()?; + let link = format!("/proc/{}/exe", child.id()); + let deadline = Instant::now() + Duration::from_secs(5); + let mapped = loop { + if fs::read_link(&link).ok().as_deref() == Some(image) { + break true; + } + if Instant::now() > deadline || child.try_wait().ok().flatten().is_some() { + break false; + } + std::thread::sleep(Duration::from_millis(10)); + }; + fs::remove_file(image).unwrap(); + let exe = mapped.then(|| resolve(child.id()).exe); + let _ = child.kill(); + let _ = child.wait(); + exe + } + + #[test] + fn a_deleted_image_keeps_its_suffix_in_another_user_namespace() { + // In its own user and mount namespace a process can mount its bytes + // at /usr/bin/curl, run them and delete them. Without the suffix the + // path would be the host curl's; a temporary copy stands in for it. + let Some(sleep) = ["/usr/bin/sleep", "/bin/sleep"] + .into_iter() + .map(Path::new) + .find(|p| p.exists()) + else { + return; + }; + let dir = tempfile::tempdir().unwrap(); + let image = dir.path().join("sleep"); + + // Here, an upgraded program keeps the path its rules name. + assert_eq!( + resolve_deleted_image(sleep, &image, &[]).as_deref(), + Some(image.as_path()) + ); + + let Some(exe) = resolve_deleted_image(sleep, &image, &["unshare", "--user"]) else { + return; // no unprivileged user namespaces here + }; + let mut deleted = image.into_os_string(); + deleted.push(DELETED_SUFFIX); + assert_eq!(exe.as_os_str(), deleted); + } + #[test] fn a_path_naming_another_file_here_is_not_the_image() { let dir = tempfile::tempdir().unwrap(); diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 4ad8f5b..fa333fa 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -190,7 +190,9 @@ The same opened file supplies metadata and bytes. Content changes during hashing are rejected; the mapped link, metadata and process start time must still agree before publishing executable identity. The link is rendered in the process's own mount namespace, so a path that names a different file in -the daemon's view leaves the executable unknown. A digest is cached only +the daemon's view leaves the executable unknown. A deleted image's +`" (deleted)"` suffix is dropped only for a process in the daemon's user +namespace; elsewhere the former path cannot be checked. A digest is cached only when the image's ctime was at least 2 seconds old as hashing began. Userspace cannot set ctime and any write moves it, so a changed image misses the cache; an unchanged one is never rehashed per packet on the single worker. The diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 4096242..10c78b8 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -239,8 +239,12 @@ is a separate launch mode. user can also borrow an allowed program's identity by running it with chosen arguments or with `LD_PRELOAD`, which a hash pin does not prevent. An image deleted while it runs is matched by the path it had (an upgraded - program keeps its rules), which the daemon cannot check against anything, - so a mount namespace can present a path that way too. Digests and the + program keeps its rules), which the daemon cannot check against anything. + Only a process in the daemon's own user namespace gets that path; one in + another user namespace (`unshare -U`, a rootless container) keeps the + `" (deleted)"` suffix, so its rules stop matching until it restarts. A + setuid mount helper such as setuid `bwrap` lets a user present a path + that way too. Digests and the root-sealed test trust what the filesystem reports: on a FUSE filesystem the daemon can read (`user_allow_other` in `/etc/fuse.conf`), the user who mounted it controls both. From 80ef7ff344db1ac92853ceac73e3da0af8383545 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:27:43 +0200 Subject: [PATCH 088/125] fix(provenance): retry a failed rpm query instead of keeping its index A failed rpm -qa, such as a query timing out under boot I/O, produced an empty index stamped with the database's current mtime. It counted as current, so every binary read as not from a package until the next package transaction changed the stamp. The index from a failed query now carries no stamp: the packet thread answers NotReady, shown as unknown, and the warmer retries it at the next two-minute refresh. --- CHANGELOG.md | 5 +++++ crates/cfc-daemon/src/provenance.rs | 33 +++++++++++++++++++++-------- 2 files changed, 29 insertions(+), 9 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 72b3193..310c5cd 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -193,6 +193,11 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). transaction. The packet thread now skips a busy or stale index, and that "not ready" answer is no longer cached for an hour as "not from a package"; it shows as unknown until the index is ready. +- A failed `rpm -qa` (a query timing out at boot, for instance) left an + empty package index that counted as current, so every binary showed as + "not from a package" until the next package transaction. The index from a + failed query is now retried at the next refresh, every two minutes, and + provenance shows as unknown meanwhile. - A refused `UpsertRule` or `ApplyRules` left nothing in the journal unless authorization refused it, so "never sent" and "sent and refused" looked the same (#46). Every refusal now logs the RPC, the caller's uid and pid, the diff --git a/crates/cfc-daemon/src/provenance.rs b/crates/cfc-daemon/src/provenance.rs index 1cc416a..7203778 100644 --- a/crates/cfc-daemon/src/provenance.rs +++ b/crates/cfc-daemon/src/provenance.rs @@ -877,20 +877,18 @@ impl Rpm { /// One `rpm -qa` pass, streamed. /// /// A failure of any kind - rpm missing, the database locked, the query - /// timing out - yields an *empty* index rather than propagating. That is - /// the same answer this host gave before this backend existed - /// (`Unpackaged` everywhere), and it is the only answer that keeps a - /// package transaction from being able to stall the firewall. + /// timing out - yields an empty index rather than propagating, which keeps + /// a package transaction from being able to stall the firewall. That index + /// carries no stamp, so it is never current: the packet thread answers + /// `NotReady` (shown as unknown) and the next [`warm`] retries. Stamped, + /// it reported every binary as unpackaged until the next transaction. fn build_index(&self, stamp: Option) -> Index { let mut idx = Index::empty(stamp); let out = match self.run_query() { Ok(out) => out, Err(e) => { - warn!( - "rpm provenance query failed: {e}; \ - binaries on this host will report as unpackaged" - ); - return idx; + warn!("rpm provenance query failed: {e}; retrying on the next index refresh"); + return Index::empty(None); } }; let mut current: Option<(String, u32)> = None; @@ -1747,6 +1745,23 @@ mod tests { assert!(rpm.lookup(Path::new("/usr/bin/curl")).unwrap().is_none()); } + #[test] + fn a_failed_rpm_query_is_not_kept_as_the_current_index() { + // A query that timed out at boot was stamped like a good one, so every + // binary read as unpackaged until the next package transaction. + let tmp = tempfile::tempdir().unwrap(); + let db = tmp.path().join("rpmdb"); + std::fs::create_dir_all(&db).unwrap(); + let rpm = Rpm::with_program(db, tmp.path().join("no-such-rpm")); + assert!(rpm.lookup(Path::new("/usr/bin/curl")).unwrap().is_none()); + std::thread::scope(|s| { + s.spawn(|| { + mark_datapath_thread(); + assert_eq!(rpm.lookup(Path::new("/usr/bin/curl")), Err(NotReady)); + }); + }); + } + #[test] fn rpm_line_parsing_rejects_what_is_not_a_file_record() { // Warnings and placeholders share the stream with real records. From a8eb41a3bf698b269b2b29e62dde30a52747eb75 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:27:43 +0200 Subject: [PATCH 089/125] docs(hardening): an image the daemon cannot open has no identity An AppImage, or any image on a FUSE mount without allow_other, cannot be opened even by root, so the resolver publishes no executable identity for it and path-scoped Allows never apply. Publishing the path unchecked would let a mount namespace present any path that way. --- docs/HARDENING.md | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 10c78b8..723018f 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -247,7 +247,9 @@ is a separate launch mode. that way too. Digests and the root-sealed test trust what the filesystem reports: on a FUSE filesystem the daemon can read (`user_allow_other` in `/etc/fuse.conf`), the user who - mounted it controls both. + mounted it controls both. An image the daemon cannot open at all, such as + an AppImage or anything on a FUSE mount without `allow_other`, has no + executable identity, so an Allow scoped to its path never applies to it. - **Raw and packet sockets**: applications with `CAP_NET_RAW` can use AF_PACKET outside the shipped `inet OUTPUT` hook. Raw IP packets can coincide with another socket's tuple even when TCP matching is strict. Tuple and inode From fb794565e62cc94cb832d6a74530bea6e1103bac Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:35:12 +0200 Subject: [PATCH 090/125] fix(rules): remove pre-0.7.0 copies of a bundle's rules on bundle remove bundle add already counted a same-named rule identical to the entry as the bundle's own copy seeded before 0.7.0, but bundle remove matched only the deterministic ids, so on an upgraded host `bundle remove system` removed nothing and left those rules behind. One helper now finds the identical copies for both commands; an edited copy is still neither skipped nor removed. --- CHANGELOG.md | 4 +- README.md | 2 +- crates/cfc-cli/src/main.rs | 5 ++- crates/cfc-cli/src/rules.rs | 89 ++++++++++++++++++++++++++----------- 4 files changed, 69 insertions(+), 31 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 310c5cd..be7519a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -245,8 +245,8 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). point, and its identity is printed before it starts. - `cfc rules bootstrap-defaults` and `bundle add` failed on hosts seeded before 0.7.0, calling the bundle's own rules outside it. An identical - same-named rule now counts as present; a different one still stops the - command. + same-named rule now counts as present, and `bundle remove` removes it; a + different one still stops the command. - `cfc rules bundle remove` deleted a bundle rule the user had edited into a deny. It now keeps any of its rules that is no longer an allow. - OpenSnitch import passed `dest.ip` networks (`10.0.0.0/8/32`), bad CIDRs diff --git a/README.md b/README.md index e78d493..616d32c 100644 --- a/README.md +++ b/README.md @@ -209,7 +209,7 @@ systemd-timesyncd and chronyd NTP (:123/udp), the DHCP clients (dhcpcd, NetworkManager and systemd-networkd, :67 and :547/udp), pacman and paru HTTPS mirrors (:443/tcp), and the SSH client (:22/tcp) - and is idempotent (rules it installed are skipped, as are identical same-named -rules seeded before 0.7.0; a different rule with one of its names stops it +rules seeded before 0.7.0, which `bundle remove` also removes; a different rule with one of its names stops it before anything changes; `--dry-run` previews). **Do not skip this step.** No profile allows unmatched remote flows on its own. With no rules and no UI connected, unmatched queued remote connections are denied. Filtering starts before the network is configured diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index 6887eb1..dc528ac 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -228,8 +228,9 @@ enum BundleCmd { /// Remove the rules a bundle installed. /// /// Matches the ids the bundle gave its rules, never a name or a prefix, - /// so a rule you wrote yourself is never caught by it. A bundle rule you - /// edited into a deny or reject is kept. + /// so a rule you wrote yourself is never caught by it. Rules seeded + /// before 0.7.0 are removed only while identical to the bundle's entry. + /// A bundle rule you edited into a deny or reject is kept. Remove { /// Bundle name (see `cfc rules bundle list`). name: String, diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 6be33b4..89d5d3b 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -2001,6 +2001,31 @@ fn same_policy(rule: &proto::RuleInfo, wanted: &proto::RuleInfo) -> bool { && rule.scope == wanted.scope } +/// Ids of the rules that are this bundle's own entries as seeded before 0.7.0 +/// gave bundle rules deterministic ids: same name, and granting exactly what +/// the entry would install here. `bundle add` counts them as present and +/// `bundle remove` removes them; an edited copy is neither. +fn legacy_copies( + bundle: &Bundle, + present: &[(&'static str, PathBuf)], + existing: &[proto::RuleInfo], +) -> std::collections::HashSet { + let mut ids = std::collections::HashSet::new(); + for (rule_name, exe) in present { + let Some(spec) = bundle.rules.iter().find(|r| r.name == *rule_name) else { + continue; + }; + let wanted = proto_for(spec, &exe.to_string_lossy()); + ids.extend( + existing + .iter() + .filter(|rule| rule.name == *rule_name && same_policy(rule, &wanted)) + .map(|rule| rule.id.clone()), + ); + } + ids +} + fn bundle_rule_id(bundle: &str, name: &str) -> String { use sha2::{Digest, Sha256}; let digest = Sha256::digest(format!("colony-firewall-bundle\0{bundle}\0{name}")); @@ -2179,27 +2204,21 @@ pub async fn bundle_add( let mut added = Vec::new(); let mut skipped_present = 0u32; // A rule with an entry's name but another id was not installed by this - // bundle. One identical to the entry is that entry as seeded before 0.7.0 - // gave bundle rules their own ids, and counts as present; any other one + // bundle. A copy seeded before 0.7.0 counts as present; any other one // stops the command before it changes anything. - let mut legacy = std::collections::HashSet::new(); - for (rule_name, exe) in &planned.present { + let legacy = legacy_copies(&bundle, &planned.present, &existing); + for (rule_name, _) in &planned.present { let id = bundle_rule_id(bundle.name, rule_name); - let wanted = proto_for(by_name[*rule_name], &exe.to_string_lossy()); - for rule in existing + if let Some(rule) = existing .iter() - .filter(|rule| rule.name == *rule_name && rule.id != id) + .find(|rule| rule.name == *rule_name && rule.id != id && !legacy.contains(&rule.id)) { - if same_policy(rule, &wanted) { - legacy.insert(*rule_name); - } else { - return Err(CliError::runtime(format!( - "bundle rule `{rule_name}` collides with rule {} of the same name, which this \ - bundle did not install and which differs from it; rename or remove that rule \ - and retry. Nothing was changed", - short_id(&rule.id) - ))); - } + return Err(CliError::runtime(format!( + "bundle rule `{rule_name}` collides with rule {} of the same name, which this \ + bundle did not install and which differs from it; rename or remove that rule \ + and retry. Nothing was changed", + short_id(&rule.id) + ))); } } @@ -2218,7 +2237,10 @@ pub async fn bundle_add( skipped_present += 1; continue; } - if legacy.contains(rule_name) { + if existing + .iter() + .any(|rule| rule.name == *rule_name && legacy.contains(&rule.id)) + { skipped_present += 1; continue; } @@ -2287,8 +2309,9 @@ const RETIRED_BUNDLE_RULES: &[(&str, &str)] = &[ /// `cfc rules bundle remove ` /// -/// Removes only deterministic IDs created by this bundle. Existing rules -/// imported by older versions lack this ownership evidence and are preserved. +/// Removes the deterministic IDs this bundle gives its rules, and the copies +/// of its entries seeded before 0.7.0 that still grant exactly what the entry +/// would (see [`legacy_copies`]). Any other rule, same name or not, is kept. pub async fn bundle_remove( client: &mut Client, name: &str, @@ -2309,9 +2332,13 @@ pub async fn bundle_remove( .collect(); let existing = client.list_rules().await?; + let legacy = legacy_copies(&bundle, &plan(&bundle).present, &existing); let mut removed = Vec::new(); let mut kept = Vec::new(); - for r in existing.iter().filter(|r| owned.contains(&r.id)) { + for r in existing + .iter() + .filter(|r| owned.contains(&r.id) || legacy.contains(&r.id)) + { // Bundles install only Allows. An editor keeps the id, so one that is // now a Deny or Reject is the user's decision, and deleting it would // let the traffic it stops through to the prompt or the default. @@ -2761,21 +2788,31 @@ mod bundle_tests { use super::*; // Hosts seeded before 0.7.0 hold the bundle's rules under random ids. - // An identical copy counts as present; an edited one still blocks. + // An identical copy is the bundle's own, for add and for remove; an + // edited one is not. #[test] fn a_pre_0_7_copy_of_a_bundle_rule_counts_only_while_unchanged() { let bundle = find_bundle("inbound").unwrap(); - let wanted = proto_for(&bundle.rules[0], ""); + let entry = &bundle.rules[0]; + let present = [(entry.name, PathBuf::from("/usr/bin/sshd"))]; + let wanted = proto_for(entry, "/usr/bin/sshd"); let mut legacy = wanted.clone(); legacy.id = "11111111-1111-4111-8111-111111111111".into(); legacy.enabled = false; - assert!(same_policy(&legacy, &wanted)); let mut deny = legacy.clone(); + deny.id = "22222222-2222-4222-8222-222222222222".into(); deny.action = proto::Action::Deny as i32; - assert!(!same_policy(&deny, &wanted)); let mut wider = legacy.clone(); + wider.id = "33333333-3333-4333-8333-333333333333".into(); wider.scope.as_mut().unwrap().src_net.clear(); - assert!(!same_policy(&wider, &wanted)); + let mut elsewhere = legacy.clone(); + elsewhere.id = "44444444-4444-4444-8444-444444444444".into(); + elsewhere.scope.as_mut().unwrap().exe_path = "/usr/local/bin/sshd".into(); + let found = legacy_copies(&bundle, &present, &[legacy, deny, wider, elsewhere]); + assert_eq!( + found, + std::collections::HashSet::from(["11111111-1111-4111-8111-111111111111".to_owned()]) + ); } // `/usr/bin/firefox` on Arch is `exec /usr/lib/firefox/firefox`, and From dd690163fd254fa872ec994f995523a6abc44059 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:39:33 +0200 Subject: [PATCH 091/125] fix(dns): drop observed names with non host-name characters Observed DNS answers are copied byte for byte from unauthenticated packets and every client printed the name before the real address and its trust label, so a name such as "google.com (1.2.3.4; verified hostname)" could pose as that label. The daemon now keeps an observed name only when it is ASCII letters, digits, '-', '_' and '.'. --- CHANGELOG.md | 4 ++++ crates/cfc-daemon/src/dns.rs | 27 ++++++++++++++++++++++++++- crates/cfc-ebpf-common/src/dns.rs | 5 ++++- docs/HARDENING.md | 3 +++ 4 files changed, 37 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index be7519a..d912e6d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -90,6 +90,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). by dunst and mako, so a path could hide part of itself. They are now escaped as the CLI already did. The GUI's Remote row says whether the name is verified. +- An observed DNS answer could name an address with spaces and brackets, + such as `google.com (1.2.3.4; verified hostname)`, and every client printed + it before the real address and trust label. Observed names with anything + but letters, digits, `-`, `_` and `.` are now dropped. - While no daemon listens on the queue, new loopback flows are allowed (`oifname "lo" ct state new queue num 0 bypass`), so local services keep working when the daemon is down. Loopback Deny rules are not enforced then. diff --git a/crates/cfc-daemon/src/dns.rs b/crates/cfc-daemon/src/dns.rs index 063e316..96b51c7 100644 --- a/crates/cfc-daemon/src/dns.rs +++ b/crates/cfc-daemon/src/dns.rs @@ -162,7 +162,15 @@ impl DnsCache { // Trailing dots and case are presentation details of the wire format; // rules and the UI compare bare lowercase names. let name = name.trim_end_matches('.').to_ascii_lowercase(); - if name.is_empty() { + // The record is copied byte for byte from an unauthenticated packet, + // and the tray, GUI and CLI show it next to the real address. Keep + // only host-name characters, so a name such as + // "google.com (1.2.3.4; verified hostname)" cannot pose as that label. + if name.is_empty() + || !name + .bytes() + .all(|b| b.is_ascii_alphanumeric() || matches!(b, b'-' | b'_' | b'.')) + { return; } let ttl = @@ -524,6 +532,23 @@ mod tests { assert!(cache.lookup_at(ip("5.6.7.8"), now).is_none()); } + #[test] + fn observed_names_outside_host_name_characters_are_dropped() { + let cache = DnsCache::new(); + let now = Instant::now(); + for (addr, name) in [ + ("5.6.7.8", "google.com (1.2.3.4; verified hostname)"), + ("5.6.7.9", "a\nb.example"), + ("5.6.7.10", "x.example"), + ("5.6.7.11", "caf\u{e9}.example"), + ] { + cache.observe_answer_at(ip(addr), name, 300, now); + assert!(cache.lookup_at(ip(addr), now).is_none(), "{name}"); + } + cache.observe_answer_at(ip("5.6.7.12"), "_dmarc.x-1.example", 300, now); + assert!(cache.lookup_at(ip("5.6.7.12"), now).is_some()); + } + #[test] fn an_observation_beats_an_existing_ptr_name() { // Diagnostic naming prefers a fresh observation; policy identity does diff --git a/crates/cfc-ebpf-common/src/dns.rs b/crates/cfc-ebpf-common/src/dns.rs index 0bee27f..729cc31 100644 --- a/crates/cfc-ebpf-common/src/dns.rs +++ b/crates/cfc-ebpf-common/src/dns.rs @@ -250,7 +250,10 @@ pub fn skip_questions(c: &DnsCursor<'_>, qdcount: u16) -> Option { Some(off) } -/// Materialises the name at `start` into `out` as dot-separated ASCII. +/// Materialises the name at `start` into `out` as dot-separated labels. +/// +/// Label bytes are copied as received, unvalidated: the daemon keeps only +/// host-name characters before showing a name anywhere. /// /// Returns `(bytes_written, offset_just_past_the_name_in_the_stream)`. A single /// NUL is written at `out[bytes_written]` (when it fits); bytes beyond that are diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 723018f..3df65e1 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -129,6 +129,9 @@ It does not validate a resolver transaction, sender or question. These records remain untrusted diagnostics in a separate cache. They cannot satisfy a policy rule. Observations and forward-confirmed PTR diagnostics use separate caches. Diagnostic entries retain the record TTL, clamped to 60s..1h; the policy cache remains separate. +A name with anything but ASCII letters, digits, `-`, `_` and `.` is dropped, +so a crafted answer cannot print text that looks like the address or trust +label shown beside it. ## Deny or Reject? From 7639b98c646995c25057c31a2c76a2f8aa254c04 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:42:06 +0200 Subject: [PATCH 092/125] docs(hardening): say what the hand-written ReadOnlyPaths leave out The sandbox table said the unit's ReadOnlyPaths cover every entry ProtectKernelTunables does. They leave /proc/kallsyms and /proc/kcore visible, which ProtectKernelTunables hides, and any /sys/fs filesystem or new top-level /sys directory they do not name stays writable. The table and the unit comment now say so, and why the two proc files expose nothing extra: kcore needs CAP_SYS_RAWIO and kallsyms shows addresses without CAP_SYSLOG only when it shows them to every process. --- CHANGELOG.md | 4 ++++ docs/HARDENING.md | 2 +- systemd/colony-firewalld.service | 9 +++++++-- 3 files changed, 12 insertions(+), 3 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d912e6d..84e5c49 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -292,6 +292,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). saved), that Docker grants `CAP_NET_RAW` by default, and that `cfc pause`, not a profile switch, lets unmatched flows through while debugging. SECURITY.md links the documented non-goals. +- HARDENING.md no longer says the unit's hand-written `ReadOnlyPaths` + cover everything `ProtectKernelTunables` does: they leave `/proc/kallsyms` + and `/proc/kcore` visible and any `/sys/fs` filesystem they do not name + writable. ## [0.7.0] - 2026-09-30 diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 3df65e1..655aa64 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -444,7 +444,7 @@ to shrink what a code-execution bug could reach: | `RestrictAddressFamilies` | AF_UNIX, AF_INET, AF_INET6, AF_NETLINK only; no packet sockets | | `RestrictNamespaces`, `LockPersonality`, `RestrictRealtime`, `RestrictSUIDSGID` | Namespace and personality lockdown | | `ProtectKernelLogs`, `ProtectControlGroups`, `ProtectClock`, `ProtectHostname` | No writing kernel state | -| `ReadOnlyPaths` (in place of `ProtectKernelTunables`) | The `/proc` and `/sys` entries `ProtectKernelTunables` covers, listed by hand so `/sys/fs/bpf` stays writable for the pinned links. An entry missing from the list stays writable | +| `ReadOnlyPaths` (in place of `ProtectKernelTunables`) | The `/proc` and `/sys` entries `ProtectKernelTunables` makes read-only, listed by hand so `/sys/fs/bpf` stays writable for the pinned links. An entry missing from the list stays writable: a `/sys/fs` filesystem or `/proc` entry it does not name, or a new top-level `/sys` directory. `/proc/kallsyms` and `/proc/kcore` stay visible, which `ProtectKernelTunables` would hide; without `CAP_SYS_RAWIO` the daemon cannot open `/proc/kcore`, and without `CAP_SYSLOG` it sees kernel addresses in `/proc/kallsyms` only when every process does | | `UMask=0077` | Closes the window between `bind` and the explicit chmod of the control socket | | `PrivateTmp` | No shared `/tmp` | diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index 459e339..5954670 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -126,8 +126,8 @@ PrivateDevices=true # logged at info, on a machine that was perfectly capable of pinning - which is # to say the "denials survive kill -9" property was silently absent. # -# The ReadOnlyPaths below reproduce what ProtectKernelTunables covers, minus -# the one directory that has to be writable. Every entry of /sys is listed +# The ReadOnlyPaths below reproduce what ProtectKernelTunables makes +# read-only, minus the one directory that has to be writable. Every entry of /sys is listed # individually rather than /sys as a whole, because listing the parent and then # excepting the child is exactly what does not work: systemd makes the parent # writable to reach it. @@ -140,6 +140,11 @@ PrivateDevices=true # entry not listed here, would be writable to this unit until it is added # here. That is the cost of the trade, and it is smaller than losing the pin. # +# ProtectKernelTunables also hides /proc/kallsyms and /proc/kcore; this list +# does not. Opening /proc/kcore needs CAP_SYS_RAWIO, which the bounding set +# drops, and without CAP_SYSLOG /proc/kallsyms shows this unit addresses only +# when it shows them to every process. +# # EVERY path is prefixed with "-", and that is not defensive tidiness. Without # it, systemd treats a missing path as a fatal error and the unit dies at # 226/NAMESPACE before the daemon is ever executed: From fd746b624f2a69f6d67a29ce0d18a33a16812d12 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 093/125] refactor(core,daemon): delete helpers nothing calls Engine::enabled_exe_paths lost its last caller when compile_rules moved to compilable_exe_paths, and calling it again would bring back the uid-scope defect that change fixed; its lock warning moves to compilable_exe_paths. Verdict::allow_from_rule, Process::display_name, CoreError, cfc_core::Result, the ResolvedExe re-export, exe_path::resolve_scope and Resolved::path had no caller. Resolved::note only ever sees resolve_policy's Ok values, so its text for the refused variants is gone. The provenance test hashes curl with the shared sha256_file fixture instead of its own copy. --- crates/cfc-core/src/exe_path.rs | 81 +++++------------------------ crates/cfc-core/src/lib.rs | 17 ------ crates/cfc-core/src/process.rs | 7 --- crates/cfc-core/src/verdict.rs | 4 -- crates/cfc-daemon/src/decision.rs | 28 ++-------- crates/cfc-daemon/src/provenance.rs | 28 +--------- 6 files changed, 20 insertions(+), 145 deletions(-) diff --git a/crates/cfc-core/src/exe_path.rs b/crates/cfc-core/src/exe_path.rs index 9a56e5c..ad4c155 100644 --- a/crates/cfc-core/src/exe_path.rs +++ b/crates/cfc-core/src/exe_path.rs @@ -50,14 +50,6 @@ pub enum Resolved { } impl Resolved { - /// The path to store, whatever happened. - pub fn path(&self) -> &Path { - match self { - Self::Unchanged(p) | Self::Missing(p) | Self::Relative(p) => p, - Self::Rewritten { to, .. } | Self::RewrittenButMissing { to, .. } => to, - } - } - /// Consumes into the path to store. pub fn into_path(self) -> PathBuf { match self { @@ -75,34 +67,19 @@ impl Resolved { } /// One line for a human, or `None` when there is nothing worth saying. + /// + /// Only an `Ok` from [`resolve_policy`] reaches a human, and that is + /// always `Unchanged` or `Missing`: every other outcome is refused there + /// with its own message. pub fn note(&self) -> Option { match self { - Self::Unchanged(_) => None, - Self::Rewritten { from, to } => Some(format!( - "resolved {} to {} (the kernel reports the second, so a rule for \ - the first would never match)", - from.display(), - to.display() - )), - Self::RewrittenButMissing { from, to } => Some(format!( - "{} resolves to {}, but nothing is installed there yet; the rule \ - is stored against the resolved path and will match once it is - \ - unless the program installs as a symlink, which would need the \ - rule rewritten", - from.display(), - to.display() - )), Self::Missing(p) => Some(format!( "{} does not exist; the rule is stored as written, and if the \ path turns out to be a symlink once the program is installed \ it will need rewriting to the real path", p.display() )), - Self::Relative(p) => Some(format!( - "{} is not an absolute path; /proc reports absolute paths, so \ - this rule can never match", - p.display() - )), + _ => None, } } } @@ -257,15 +234,6 @@ fn resolve_via_ancestor(path: &Path) -> Option { } } -/// Diagnostic resolution in place. Policy writes must use [`resolve_policy`]. -/// Returns `None` when the scope names no executable. -pub fn resolve_scope(scope: &mut crate::RuleScope) -> Option { - let current = scope.exe_path.as_deref()?; - let outcome = resolve(current); - scope.exe_path = Some(outcome.path().to_path_buf()); - Some(outcome) -} - /// Whether a *directory* on the way to an executable is sealed against /// non-root replacement. /// @@ -409,7 +377,7 @@ mod tests { "got {outcome:?}" ); assert_eq!( - outcome.path(), + outcome.into_path(), std::fs::canonicalize(realdir.join("curl")).expect("canon") ); } @@ -431,9 +399,9 @@ mod tests { let p = PathBuf::from("/nonexistent-8f2c1a/curl"); let outcome = resolve(&p); assert_eq!(outcome, Resolved::Missing(p.clone())); - assert_eq!(outcome.path(), p); assert!(outcome.is_inert()); assert!(outcome.note().expect("a note").contains("does not exist")); + assert_eq!(outcome.into_path(), p); } #[test] @@ -495,13 +463,11 @@ mod tests { ); // Still stored against the resolved directory, which is the point. assert_eq!( - outcome.path(), + outcome.into_path(), std::fs::canonicalize(&realdir) .expect("canon") .join("not-installed") ); - let note = outcome.note().expect("a note"); - assert!(note.contains("nothing is installed there yet"), "{note}"); } #[test] @@ -549,38 +515,17 @@ mod tests { !outcome.is_inert(), "a real file must not be reported as missing: {outcome:?}" ); - assert_eq!(outcome.path(), std::fs::canonicalize(&f).expect("canon")); + assert_eq!( + outcome.into_path(), + std::fs::canonicalize(&f).expect("canon") + ); } #[test] - fn a_relative_path_can_never_match_and_says_so() { + fn a_relative_path_can_never_match() { let outcome = resolve(Path::new("bin/curl")); assert!(matches!(outcome, Resolved::Relative(_))); assert!(outcome.is_inert()); - assert!(outcome.note().expect("a note").contains("absolute")); - } - - #[test] - fn resolving_a_scope_updates_it_in_place() { - let dir = tempfile::tempdir().expect("tempdir"); - let real = dir.path().join("prog"); - std::fs::write(&real, b"x").expect("write"); - let link = dir.path().join("prog-link"); - std::os::unix::fs::symlink(&real, &link).expect("symlink"); - - let mut scope = crate::RuleScope::any(); - scope.exe_path = Some(link); - let outcome = resolve_scope(&mut scope).expect("the scope names an exe"); - assert!(matches!(outcome, Resolved::Rewritten { .. })); - assert_eq!( - scope.exe_path.as_deref(), - Some(std::fs::canonicalize(&real).expect("canon").as_path()) - ); - - // A scope with no exe is not an error and is not touched. - let mut empty = crate::RuleScope::any(); - assert!(resolve_scope(&mut empty).is_none()); - assert_eq!(empty.exe_path, None); } } diff --git a/crates/cfc-core/src/lib.rs b/crates/cfc-core/src/lib.rs index 62f947c..f24d6f5 100644 --- a/crates/cfc-core/src/lib.rs +++ b/crates/cfc-core/src/lib.rs @@ -9,23 +9,6 @@ pub mod rule; pub mod verdict; pub use connection::{Connection, Direction, Protocol}; -pub use exe_path::Resolved as ResolvedExe; pub use process::{Process, Provenance, UNKNOWN_EXE}; pub use rule::{Action, Duration, Rule, RuleScope, RuleSet}; pub use verdict::{Verdict, VerdictSource}; - -use thiserror::Error; - -#[derive(Debug, Error)] -pub enum CoreError { - #[error("invalid rule: {0}")] - InvalidRule(String), - - #[error("invalid address: {0}")] - InvalidAddress(String), - - #[error("serde: {0}")] - Serde(#[from] serde_json::Error), -} - -pub type Result = std::result::Result; diff --git a/crates/cfc-core/src/process.rs b/crates/cfc-core/src/process.rs index 2d91da5..c2122d5 100644 --- a/crates/cfc-core/src/process.rs +++ b/crates/cfc-core/src/process.rs @@ -101,13 +101,6 @@ impl Process { provenance: Provenance::Unknown, } } - - pub fn display_name(&self) -> String { - self.exe - .file_name() - .map(|n| n.to_string_lossy().into_owned()) - .unwrap_or_else(|| format!("pid:{}", self.pid)) - } } #[cfg(test)] diff --git a/crates/cfc-core/src/verdict.rs b/crates/cfc-core/src/verdict.rs index 958d6de..e2f48f8 100644 --- a/crates/cfc-core/src/verdict.rs +++ b/crates/cfc-core/src/verdict.rs @@ -34,10 +34,6 @@ impl Verdict { } } - pub fn allow_from_rule(rule_id: uuid::Uuid) -> Self { - Self::from_rule(crate::Action::Allow, rule_id) - } - pub fn deny_from_rule(rule_id: uuid::Uuid) -> Self { Self::from_rule(crate::Action::Deny, rule_id) } diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index 773fe0b..0aecd29 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -314,29 +314,6 @@ impl Engine { self.inner.rules.read().rules.len() } - /// The distinct executables enabled rules name, under one read lock. - /// - /// `snapshot()` would answer this too, and answer it expensively: it deep - /// clones every rule - names, `PathBuf`s, every `Option` in every - /// scope - and merges hit counts on the way, none of which the caller - /// wants. This clones the paths it is actually asked for and nothing else. - /// - /// Deliberately does not evaluate anything while holding the lock: the - /// caller re-enters through `process_wide_action`, which takes its own read - /// lock, and nesting reads on a `std::sync::RwLock` can deadlock against a - /// waiting writer. - pub fn enabled_exe_paths(&self) -> std::collections::BTreeSet { - let now_unix_ms = chrono::Utc::now().timestamp_millis(); - self.inner - .rules - .read() - .rules - .iter() - .filter(|r| r.enabled && !r.is_expired(now_unix_ms)) - .filter_map(|r| r.scope.exe_path.clone()) - .collect() - } - /// Executables safe to compile into the kernel's table, and the ones that /// are not. /// @@ -348,6 +325,11 @@ impl Engine { /// A synthetic process with no uid cannot decide a uid-scoped rule. /// Excluding every executable such a rule could touch preserves the /// per-user decision in the packet path, where the uid is available. + /// + /// Returns owned paths and evaluates nothing under the lock: the caller + /// re-enters through `process_wide_action`, which takes its own read lock, + /// and nesting reads on a `std::sync::RwLock` can deadlock against a + /// waiting writer. pub fn compilable_exe_paths(&self) -> Option> { let now_unix_ms = chrono::Utc::now().timestamp_millis(); let rules = self.inner.rules.read(); diff --git a/crates/cfc-daemon/src/provenance.rs b/crates/cfc-daemon/src/provenance.rs index 7203778..80e3f97 100644 --- a/crates/cfc-daemon/src/provenance.rs +++ b/crates/cfc-daemon/src/provenance.rs @@ -1910,7 +1910,8 @@ mod tests { let curl = Path::new("/usr/bin/curl"); // Hash the real file the way process_resolve does, from the bytes // on disk. - let running = sha256_of(curl); + let running = + crate::process_resolve::sha256_file(curl, cfc_core::rule::SHA256_MAX_LEN).unwrap(); println!("/usr/bin/curl running sha256 = {running}"); // Cold: this call also builds the whole path index. @@ -1959,29 +1960,4 @@ mod tests { "identical bytes, but no package owns that path" ); } - - #[cfg(test)] - fn sha256_of(path: &Path) -> String { - use sha2::{Digest, Sha256}; - use std::io::Read as _; - let mut f = std::fs::File::open(path).unwrap(); - let mut h = Sha256::new(); - // Same shape as sha256_file in process_resolve: io::copy and `{:x}` - // both stop compiling under RustCrypto 0.11. This one only fails under - // --all-targets, which is why it outlived the other two. - let mut buf = [0u8; 64 * 1024]; - loop { - let n = f.read(&mut buf).unwrap(); - if n == 0 { - break; - } - h.update(&buf[..n]); - } - let mut out = String::with_capacity(64); - for byte in h.finalize() { - use std::fmt::Write as _; - let _ = write!(out, "{byte:02x}"); - } - out - } } From 7f7a6f317f599c55d6a003c727b76496a680bf5a Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 094/125] build: drop dependencies no crate uses once_cell, parking_lot and thiserror in cfc-core, prost and tracing in the CLI and client, the client's uuid/ipnet/chrono, eight in the GUI, prost and once_cell in the daemon and serde in cfc-proto were declared and never used. serde_json and the CLI's tokio-stream are test-only and move to dev-dependencies. The workspace once_cell entry goes with its last user. Features stay the same: the crates that use them still declare them. --- Cargo.lock | 21 --------------------- Cargo.toml | 1 - crates/cfc-cli/Cargo.toml | 7 ++++--- crates/cfc-client/Cargo.toml | 5 ----- crates/cfc-core/Cargo.toml | 5 +---- crates/cfc-daemon/Cargo.toml | 2 -- crates/cfc-proto/Cargo.toml | 1 - crates/cfc-ui/Cargo.toml | 8 -------- 8 files changed, 5 insertions(+), 45 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 722ceb2..ff2da68 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -704,7 +704,6 @@ dependencies = [ "ipnet", "libc", "owo-colors", - "prost", "serde", "serde_json", "sha2", @@ -712,7 +711,6 @@ dependencies = [ "tokio", "tokio-stream", "tonic", - "tracing", "tracing-subscriber", "uuid", ] @@ -724,17 +722,12 @@ dependencies = [ "anyhow", "cfc-core", "cfc-proto", - "chrono", "hyper-util", - "ipnet", - "prost", "thiserror 2.0.20", "tokio", "tokio-stream", "tonic", "tower", - "tracing", - "uuid", ] [[package]] @@ -743,12 +736,9 @@ version = "0.7.0" dependencies = [ "chrono", "ipnet", - "once_cell", - "parking_lot", "serde", "serde_json", "tempfile", - "thiserror 2.0.20", "uuid", ] @@ -771,11 +761,9 @@ dependencies = [ "libc", "nfq", "nix", - "once_cell", "parking_lot", "procfs", "proptest", - "prost", "rusqlite", "serde", "serde_json", @@ -800,7 +788,6 @@ name = "cfc-proto" version = "0.7.0" dependencies = [ "prost", - "serde", "tonic", "tonic-build", "tonic-prost", @@ -828,24 +815,16 @@ dependencies = [ name = "cfc-ui" version = "0.7.0" dependencies = [ - "anyhow", "cfc-client", "cfc-core", - "cfc-proto", "chrono", "colony-ui", "futures", "iced", "ipnet", - "prost", - "serde", - "thiserror 2.0.20", "tokio", - "tokio-stream", - "tonic", "tracing", "tracing-subscriber", - "uuid", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index 966e535..6d88ad7 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -96,7 +96,6 @@ clap = { version = "4.6", features = ["derive"] } owo-colors = { version = "4.2", features = ["supports-colors"] } # Utility -once_cell = "1.21" parking_lot = "0.12.5" chrono = { version = "0.4.44", features = ["serde"] } uuid = { version = "1.23", features = ["v4", "serde"] } diff --git a/crates/cfc-cli/Cargo.toml b/crates/cfc-cli/Cargo.toml index a57defb..2c3486c 100644 --- a/crates/cfc-cli/Cargo.toml +++ b/crates/cfc-cli/Cargo.toml @@ -20,8 +20,6 @@ cfc-client = { path = "../cfc-client" } tokio = { workspace = true } tonic = { workspace = true } -prost = { workspace = true } -tokio-stream = { workspace = true } futures = { workspace = true } clap = { workspace = true } # Completion + man generation live behind hidden subcommands so packaging @@ -33,7 +31,6 @@ thiserror = { workspace = true } # Raw-mode single-keypress reads for `cfc prompts`. libc is already in the # dependency graph; a TTY crate would be a new one. libc = { workspace = true } -tracing = { workspace = true } tracing-subscriber = { workspace = true } chrono = { workspace = true } serde = { workspace = true } @@ -41,3 +38,7 @@ serde_json = { workspace = true } ipnet = { workspace = true } uuid = { workspace = true } owo-colors = { workspace = true } + +[dev-dependencies] +# tests/cli_e2e.rs serves the daemon side over a Unix socket. +tokio-stream = { workspace = true } diff --git a/crates/cfc-client/Cargo.toml b/crates/cfc-client/Cargo.toml index b070622..714e07a 100644 --- a/crates/cfc-client/Cargo.toml +++ b/crates/cfc-client/Cargo.toml @@ -13,14 +13,9 @@ cfc-core = { path = "../cfc-core" } cfc-proto = { path = "../cfc-proto" } tonic = { workspace = true } -prost = { workspace = true } tokio = { workspace = true } tokio-stream = { workspace = true } tower = { workspace = true } hyper-util = { workspace = true } anyhow = { workspace = true } thiserror = { workspace = true } -tracing = { workspace = true } -uuid = { workspace = true } -ipnet = { workspace = true } -chrono = { workspace = true } diff --git a/crates/cfc-core/Cargo.toml b/crates/cfc-core/Cargo.toml index 5faaccc..0e3c5d9 100644 --- a/crates/cfc-core/Cargo.toml +++ b/crates/cfc-core/Cargo.toml @@ -10,13 +10,10 @@ description = "Colony Firewall Control - shared types and rule engine" [dependencies] serde = { workspace = true } -serde_json = { workspace = true } -thiserror = { workspace = true } chrono = { workspace = true } uuid = { workspace = true } ipnet = { workspace = true } -once_cell = { workspace = true } -parking_lot = { workspace = true } [dev-dependencies] +serde_json = { workspace = true } tempfile = "3" diff --git a/crates/cfc-daemon/Cargo.toml b/crates/cfc-daemon/Cargo.toml index 04d9944..e0874de 100644 --- a/crates/cfc-daemon/Cargo.toml +++ b/crates/cfc-daemon/Cargo.toml @@ -59,7 +59,6 @@ tokio = { workspace = true } tokio-stream = { workspace = true } futures = { workspace = true } tonic = { workspace = true } -prost = { workspace = true } rusqlite = { workspace = true } serde = { workspace = true } @@ -81,7 +80,6 @@ dns-lookup = { workspace = true } # workspace-style) on purpose -- only the daemon reads a package database. flate2 = "1.1" -once_cell = { workspace = true } parking_lot = { workspace = true } chrono = { workspace = true } uuid = { workspace = true } diff --git a/crates/cfc-proto/Cargo.toml b/crates/cfc-proto/Cargo.toml index 51f8540..d4aa974 100644 --- a/crates/cfc-proto/Cargo.toml +++ b/crates/cfc-proto/Cargo.toml @@ -12,7 +12,6 @@ description = "Colony Firewall Control - gRPC IPC schema" tonic = { workspace = true } tonic-prost = { workspace = true } prost = { workspace = true } -serde = { workspace = true } [build-dependencies] tonic-build = { workspace = true } diff --git a/crates/cfc-ui/Cargo.toml b/crates/cfc-ui/Cargo.toml index 66f2ccd..357e823 100644 --- a/crates/cfc-ui/Cargo.toml +++ b/crates/cfc-ui/Cargo.toml @@ -15,21 +15,13 @@ path = "src/main.rs" [dependencies] colony-ui = "0.1.5" cfc-core = { path = "../cfc-core" } -cfc-proto = { path = "../cfc-proto" } cfc-client = { path = "../cfc-client" } iced = { workspace = true } tokio = { workspace = true } -tokio-stream = { workspace = true } -tonic = { workspace = true } -prost = { workspace = true } futures = { workspace = true } -serde = { workspace = true } -anyhow = { workspace = true } -thiserror = { workspace = true } tracing = { workspace = true } tracing-subscriber = { workspace = true } chrono = { workspace = true } -uuid = { workspace = true } ipnet = { workspace = true } From 71b3e039733c850a6dc3b2a414e903b71ddac28d Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 095/125] refactor(cli): drop repeated cfg gates and forwarding wrappers The confinement filter and native modules are already compiled only on x86_64, so the gates inside them, the non-x86_64 sealed_filter stub, the nested block in gate() and the non-Linux verify() branch could never change anything. proto_for takes the body of allow_rule, its only caller; clap uses parse_duration directly; --socket is a PathBuf because its default means it is always set. --- crates/cfc-cli/src/confinement/filter.rs | 20 +-- crates/cfc-cli/src/confinement/mod.rs | 151 +++++++++++------------ crates/cfc-cli/src/confinement/native.rs | 19 +-- crates/cfc-cli/src/events.rs | 2 +- crates/cfc-cli/src/humantime.rs | 5 - crates/cfc-cli/src/main.rs | 16 +-- crates/cfc-cli/src/rules.rs | 78 +++++------- 7 files changed, 117 insertions(+), 174 deletions(-) diff --git a/crates/cfc-cli/src/confinement/filter.rs b/crates/cfc-cli/src/confinement/filter.rs index 30cb8f7..424f669 100644 --- a/crates/cfc-cli/src/confinement/filter.rs +++ b/crates/cfc-cli/src/confinement/filter.rs @@ -6,7 +6,6 @@ use std::fs::File; /// A sealed, rewound native cBPF file for bubblewrap's `--seccomp` option. -#[cfg(target_arch = "x86_64")] pub fn sealed_filter() -> anyhow::Result { use anyhow::Context; use std::io::{Seek, Write}; @@ -45,23 +44,12 @@ pub fn sealed_filter() -> anyhow::Result { Ok(file) } -#[cfg(not(target_arch = "x86_64"))] -pub fn sealed_filter() -> anyhow::Result { - anyhow::bail!("application confinement currently requires native x86_64") -} - -#[cfg(target_arch = "x86_64")] const ALLOW: u32 = 0x7fff_0000; -#[cfg(target_arch = "x86_64")] const KILL: u32 = 0x8000_0000; -#[cfg(target_arch = "x86_64")] const ENOSYS: u32 = 0x0005_0000 | libc::ENOSYS as u32; -#[cfg(target_arch = "x86_64")] const DENY: u32 = 0x0005_0000 | libc::EPERM as u32; -#[cfg(target_arch = "x86_64")] const NATIVE_ARCH: u32 = 0xc000_003e; -#[cfg(target_arch = "x86_64")] fn filter_program() -> Vec { let mut program = vec![ statement(0x20, 4), // seccomp_data.arch @@ -345,7 +333,6 @@ fn filter_program() -> Vec { program } -#[cfg(target_arch = "x86_64")] fn statement(code: u16, k: u32) -> libc::sock_filter { libc::sock_filter { code, @@ -355,12 +342,10 @@ fn statement(code: u16, k: u32) -> libc::sock_filter { } } -#[cfg(target_arch = "x86_64")] fn jump(code: u16, k: u32, jt: u8, jf: u8) -> libc::sock_filter { libc::sock_filter { code, jt, jf, k } } -#[cfg(target_arch = "x86_64")] fn conditional( program: &mut Vec, nr: libc::c_long, @@ -371,7 +356,6 @@ fn conditional( program.extend(body); } -#[cfg(target_arch = "x86_64")] fn argument_options(index: u32, options: &[u32]) -> Vec { // These syscall parameters are native int/unsigned int; the kernel uses // their low 32 bits. Clone flags below are an unsigned long, checked whole. @@ -384,7 +368,6 @@ fn argument_options(index: u32, options: &[u32]) -> Vec { body } -#[cfg(target_arch = "x86_64")] fn socket_policy() -> Vec { vec![ statement(0x20, 16), @@ -409,7 +392,6 @@ fn socket_policy() -> Vec { ] } -#[cfg(target_arch = "x86_64")] fn clone_policy() -> Vec { let ordinary = libc::CLONE_VM | libc::CLONE_FS @@ -434,7 +416,7 @@ fn clone_policy() -> Vec { ] } -#[cfg(all(test, target_arch = "x86_64"))] +#[cfg(test)] mod tests { use super::*; use std::io::Read; diff --git a/crates/cfc-cli/src/confinement/mod.rs b/crates/cfc-cli/src/confinement/mod.rs index 30ec639..b6a0014 100644 --- a/crates/cfc-cli/src/confinement/mod.rs +++ b/crates/cfc-cli/src/confinement/mod.rs @@ -553,84 +553,81 @@ pub(super) fn gate(id: &str) -> Result<()> { check_runtime(&manifest.runtime, &manifest.command)?; let bwrap = trusted_binary(Path::new("/usr/bin/bwrap"))?; let (uid, gid) = native::verify(&unit(id), &user(id), &manifest.allow)?; - #[cfg(target_arch = "x86_64")] - { - let seccomp = filter::sealed_filter()?; - drop_privileges(uid, gid)?; - let mut launch = Command::new(bwrap); - launch - .env_clear() - .args([ - "--unshare-user", - "--unshare-pid", - // Keep setup helpers outside the payload's PID view, including before seccomp. - "--as-pid-1", - "--unshare-ipc", - "--unshare-uts", - "--unshare-cgroup", - "--disable-userns", - "--assert-userns-disabled", - "--uid", - "65534", - "--gid", - "65534", - "--cap-drop", - "ALL", - "--new-session", - "--die-with-parent", - "--clearenv", - "--setenv", - "HOME", - "/home/cfc", - "--setenv", - "PATH", - "/usr/bin:/bin", - "--chdir", - "/home/cfc", - "--ro-bind", - ]) - .arg(&manifest.runtime) - .arg("/") - .args([ - "--proc", - "/proc", - "--dev", - "/dev", - "--tmpfs", - "/tmp", - "--tmpfs", - "/run", - "--tmpfs", - "/home", - "--dir", - "/home/cfc", - "--seccomp", - "3", - "--", - ]) - .args(&manifest.command) - .stdin(Stdio::null()) - .stdout(Stdio::null()) - .stderr(Stdio::null()); - let fd = seccomp.as_raw_fd(); - // pre_exec uses only async-signal-safe syscalls, and the gate has no runtime threads. - unsafe { - launch.pre_exec(move || { - if fd != 3 && libc::dup2(fd, 3) < 0 { - return Err(std::io::Error::last_os_error()); - } - if libc::fcntl(3, libc::F_SETFD, 0) < 0 { - return Err(std::io::Error::last_os_error()); - } - if libc::syscall(libc::SYS_close_range, 4u32, u32::MAX, 0u32) != 0 { - return Err(std::io::Error::last_os_error()); - } - Ok(()) - }); - } - let error = launch.exec(); - bail!("mandatory application isolation failed: {error}"); + let seccomp = filter::sealed_filter()?; + drop_privileges(uid, gid)?; + let mut launch = Command::new(bwrap); + launch + .env_clear() + .args([ + "--unshare-user", + "--unshare-pid", + // Keep setup helpers outside the payload's PID view, including before seccomp. + "--as-pid-1", + "--unshare-ipc", + "--unshare-uts", + "--unshare-cgroup", + "--disable-userns", + "--assert-userns-disabled", + "--uid", + "65534", + "--gid", + "65534", + "--cap-drop", + "ALL", + "--new-session", + "--die-with-parent", + "--clearenv", + "--setenv", + "HOME", + "/home/cfc", + "--setenv", + "PATH", + "/usr/bin:/bin", + "--chdir", + "/home/cfc", + "--ro-bind", + ]) + .arg(&manifest.runtime) + .arg("/") + .args([ + "--proc", + "/proc", + "--dev", + "/dev", + "--tmpfs", + "/tmp", + "--tmpfs", + "/run", + "--tmpfs", + "/home", + "--dir", + "/home/cfc", + "--seccomp", + "3", + "--", + ]) + .args(&manifest.command) + .stdin(Stdio::null()) + .stdout(Stdio::null()) + .stderr(Stdio::null()); + let fd = seccomp.as_raw_fd(); + // pre_exec uses only async-signal-safe syscalls, and the gate has no runtime threads. + unsafe { + launch.pre_exec(move || { + if fd != 3 && libc::dup2(fd, 3) < 0 { + return Err(std::io::Error::last_os_error()); + } + if libc::fcntl(3, libc::F_SETFD, 0) < 0 { + return Err(std::io::Error::last_os_error()); + } + if libc::syscall(libc::SYS_close_range, 4u32, u32::MAX, 0u32) != 0 { + return Err(std::io::Error::last_os_error()); + } + Ok(()) + }); } + let error = launch.exec(); + bail!("mandatory application isolation failed: {error}"); } #[cfg(not(target_arch = "x86_64"))] diff --git a/crates/cfc-cli/src/confinement/native.rs b/crates/cfc-cli/src/confinement/native.rs index c7a1d95..592bd25 100644 --- a/crates/cfc-cli/src/confinement/native.rs +++ b/crates/cfc-cli/src/confinement/native.rs @@ -77,19 +77,8 @@ fn program_set(direct: &[u32], effective: &[u32]) -> Result<()> { Ok(()) } -pub(super) fn verify(unit: &str, user: &str, allow: &[IpAddr]) -> Result<(u32, u32)> { - #[cfg(all(target_os = "linux", target_arch = "x86_64"))] - { - platform::verify(unit, user, allow) - } - #[cfg(not(all(target_os = "linux", target_arch = "x86_64")))] - { - let _ = (unit, user, allow); - bail!("application confinement requires x86_64 Linux") - } -} +pub(super) use platform::verify; -#[cfg(all(target_os = "linux", target_arch = "x86_64"))] mod platform { use super::*; use serde_json::{json, Map}; @@ -1044,7 +1033,11 @@ mod platform { Ok(()) } - pub(super) fn verify(unit: &str, user: &str, allow: &[IpAddr]) -> Result<(u32, u32)> { + pub(in crate::confinement) fn verify( + unit: &str, + user: &str, + allow: &[IpAddr], + ) -> Result<(u32, u32)> { ensure!( unsafe { libc::getuid() } == 0 && unsafe { libc::geteuid() } == 0, "native gate requires real and effective root" diff --git a/crates/cfc-cli/src/events.rs b/crates/cfc-cli/src/events.rs index cf47cce..2feac44 100644 --- a/crates/cfc-cli/src/events.rs +++ b/crates/cfc-cli/src/events.rs @@ -26,7 +26,7 @@ pub struct LogArgs { pub action: Option, /// Only records newer than this, e.g. 2h, 30m, 1d. - #[arg(long, value_parser = crate::humantime::parse_duration_arg)] + #[arg(long, value_parser = crate::humantime::parse_duration)] pub since: Option, } diff --git a/crates/cfc-cli/src/humantime.rs b/crates/cfc-cli/src/humantime.rs index 84f90ab..67e92d4 100644 --- a/crates/cfc-cli/src/humantime.rs +++ b/crates/cfc-cli/src/humantime.rs @@ -68,11 +68,6 @@ pub fn parse_duration(input: &str) -> Result { Ok(Duration::from_secs(total)) } -/// clap value parser wrapper. -pub fn parse_duration_arg(s: &str) -> Result { - parse_duration(s) -} - /// Renders a number of seconds the way the status line wants it: `2h 5m`, /// `45s`, `0s`. pub fn format_secs(total: i64) -> String { diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index dc528ac..0217764 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -40,7 +40,7 @@ rule's name."; )] struct Cli { #[arg(long, global = true, default_value = cfc_proto::DEFAULT_SOCKET_PATH)] - socket: Option, + socket: PathBuf, /// Output format. `json` is machine-readable; streaming commands emit /// NDJSON (one object per line). @@ -63,12 +63,6 @@ impl Cli { self.output.unwrap_or(OutputFormat::Human) } } - - fn socket(&self) -> PathBuf { - self.socket - .clone() - .unwrap_or_else(|| PathBuf::from(cfc_proto::DEFAULT_SOCKET_PATH)) - } } #[derive(Debug, Subcommand)] @@ -118,7 +112,7 @@ enum Command { /// How long to stay paused, e.g. 30m, 2h. Omitted means the /// daemon's configured default; the daemon clamps the maximum. #[arg(long = "for", value_name = "DURATION", - value_parser = humantime::parse_duration_arg)] + value_parser = humantime::parse_duration)] duration: Option, }, /// Resume normal filtering immediately. @@ -311,7 +305,7 @@ async fn cli_main() { async fn run(cli: Cli) -> CliResult { let format = cli.format(); - let socket = cli.socket(); + let socket = cli.socket; match cli.cmd { Command::Applications { cmd } => confinement::run(cmd, format).await.map_err(Into::into), @@ -636,7 +630,7 @@ mod tests { fn global_flags_work_after_the_subcommand() { let cli = Cli::parse_from(["cfc", "rules", "list", "--json", "--socket", "/tmp/x.sock"]); assert!(cli.format().is_json()); - assert_eq!(cli.socket(), PathBuf::from("/tmp/x.sock")); + assert_eq!(cli.socket, PathBuf::from("/tmp/x.sock")); } #[test] @@ -684,7 +678,7 @@ mod tests { #[test] fn socket_defaults_to_the_shared_constant() { let cli = Cli::parse_from(["cfc", "status"]); - assert_eq!(cli.socket(), PathBuf::from(cfc_proto::DEFAULT_SOCKET_PATH)); + assert_eq!(cli.socket, PathBuf::from(cfc_proto::DEFAULT_SOCKET_PATH)); } #[test] diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 89d5d3b..a1e87eb 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -1982,14 +1982,36 @@ fn plan(bundle: &Bundle) -> Planned { /// install path uses. When this was inlined, the call site passed `None` for /// two of the fields and nothing noticed until a rule was read back off disk. fn proto_for(spec: &BundleRule, exe: &str) -> proto::RuleInfo { - allow_rule( - spec.name, - exe, - spec.dst_port, - spec.protocol, - spec.direction, - spec.src_net, - ) + // `direction`/`src_net` are what an inbound bundle entry needs; every + // outbound one leaves them unset. + proto::RuleInfo { + duration_seconds: 0, + id: String::new(), + name: spec.name.to_string(), + enabled: true, + action: proto::Action::Allow as i32, + duration: proto::Duration::Always as i32, + scope: Some(proto::RuleScope { + exe_path: exe.to_string(), + exe_sha256: String::new(), + parent_exe: String::new(), + uid: 0, + has_uid: false, + dst_host: String::new(), + dst_net: String::new(), + dst_port: spec.dst_port.map(u32::from).unwrap_or(0), + has_dst_port: spec.dst_port.is_some(), + protocol: spec.protocol.map(|p| p as i32).unwrap_or(0), + has_protocol: spec.protocol.is_some(), + direction: spec.direction.map(|d| d as i32).unwrap_or(0), + has_direction: spec.direction.is_some(), + src_net: spec.src_net.unwrap_or_default().to_string(), + src_port: 0, + has_src_port: false, + }), + created_at_unix_ms: 0, + hit_count: 0, + } } /// Whether `rule` grants exactly what `wanted` would: same action, duration and @@ -2034,46 +2056,6 @@ fn bundle_rule_id(bundle: &str, name: &str) -> String { uuid::Uuid::from_bytes(bytes).to_string() } -/// `direction`/`src_net` are what an inbound bundle entry needs; every -/// outbound one leaves them unset. -fn allow_rule( - name: &str, - exe: &str, - port: Option, - proto_: Option, - direction: Option, - src_net: Option<&str>, -) -> proto::RuleInfo { - proto::RuleInfo { - duration_seconds: 0, - id: String::new(), - name: name.to_string(), - enabled: true, - action: proto::Action::Allow as i32, - duration: proto::Duration::Always as i32, - scope: Some(proto::RuleScope { - exe_path: exe.to_string(), - exe_sha256: String::new(), - parent_exe: String::new(), - uid: 0, - has_uid: false, - dst_host: String::new(), - dst_net: String::new(), - dst_port: port.map(u32::from).unwrap_or(0), - has_dst_port: port.is_some(), - protocol: proto_.map(|p| p as i32).unwrap_or(0), - has_protocol: proto_.is_some(), - direction: direction.map(|d| d as i32).unwrap_or(0), - has_direction: direction.is_some(), - src_net: src_net.unwrap_or_default().to_string(), - src_port: 0, - has_src_port: false, - }), - created_at_unix_ms: 0, - hit_count: 0, - } -} - /// `cfc rules bundle list` pub async fn bundle_list(client: &mut Client, format: OutputFormat) -> CliResult { let existing: std::collections::HashSet = client From 40b6f20b4a9fe6c2dc86ade750342b50d7f6939e Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 096/125] refactor(ui): carry the edited rule's scope whole CarriedScope copied nine RuleScope fields in by hand and out again on save, the shape that once dropped direction and the source scope and turned inbound rules outbound. The editor now keeps the rule's RuleScope and saves the visible fields over it with struct update, so a field added to the proto later is carried too. LiveEntry wrapped a single event, and PromptCard's deadline was a copy of the event's. --- crates/cfc-ui/src/main.rs | 191 ++++++++++------------------- crates/cfc-ui/src/views/live.rs | 20 +-- crates/cfc-ui/src/views/prompts.rs | 4 +- crates/cfc-ui/src/views/rules.rs | 4 +- 4 files changed, 78 insertions(+), 141 deletions(-) diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index a8afddf..b215f53 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -67,7 +67,7 @@ pub struct App { pub tab: Tab, pub daemon: DaemonState, pub rules: Vec, - pub live: VecDeque, + pub live: VecDeque, pub prompts: Vec, /// The card the A/D keys answer and when they may: `(prompt_id, /// armed_at_ms)`. See [`App::sync_key_target`]. @@ -83,7 +83,7 @@ pub struct App { pub live_verdict: VerdictFilter, /// Snapshot rendered while the feed is paused. The buffer behind it /// keeps filling, so nothing is lost. - pub live_frozen: Option>, + pub live_frozen: Option>, pub live_new: usize, pub session: SessionStats, /// Consecutive `StatusLoaded(Err)` since the last success. @@ -127,97 +127,63 @@ pub struct RuleEditor { pub created_at_unix_ms: i64, pub hit_count: u64, pub enabled: bool, - /// Scope predicates this editor has no widget for, carried through - /// untouched. + /// The edited rule's scope, so the predicates this editor has no widget + /// for are carried through untouched. /// /// `cfc rules add --uid`, an imported opensnitch ruleset, or a /// checksum-pinned rule can all set these. Rebuilding the scope from /// the visible fields alone would silently *widen* such a rule on /// save: a deny scoped to one uid would start matching every user, - /// and a sha256-pinned allow would lose its binary pin. - pub carried_scope: CarriedScope, + /// a sha256-pinned allow would lose its binary pin, and an inbound rule + /// (unset direction means outbound) would turn outbound with its source + /// scope gone. Its visible fields are stale once the form is edited, so + /// only [`hidden_scope_is_set`] and [`hidden_scope_summary`] read it, and + /// saving overwrites them from the form. + pub carried_scope: proto::RuleScope, } -/// Scope predicates preserved verbatim across an edit (see -/// [`RuleEditor::carried_scope`]). -#[derive(Debug, Clone, Default, PartialEq, Eq)] -pub struct CarriedScope { - pub exe_sha256: String, - pub parent_exe: String, - pub uid: u32, - pub has_uid: bool, - // The flow-side predicates the editor has no widgets for. Before they - // were carried, saving any edit rebuilt them as unset - and an unset - // direction means outbound, so renaming an inbound rule silently turned - // it into an outbound one with its source scope gone. - pub direction: i32, - pub has_direction: bool, - pub src_net: String, - pub src_port: u32, - pub has_src_port: bool, +/// True when `scope` carries a constraint the editor cannot show, so the view +/// can tell the user rather than let them assume the visible fields are the +/// whole rule. +pub fn hidden_scope_is_set(scope: &proto::RuleScope) -> bool { + scope.has_uid + || !scope.exe_sha256.is_empty() + || !scope.parent_exe.is_empty() + || scope.has_direction + || !scope.src_net.is_empty() + || scope.has_src_port } -impl CarriedScope { - fn from_scope(scope: Option<&proto::RuleScope>) -> Self { - match scope { - Some(s) => Self { - exe_sha256: s.exe_sha256.clone(), - parent_exe: s.parent_exe.clone(), - uid: s.uid, - has_uid: s.has_uid, - direction: s.direction, - has_direction: s.has_direction, - src_net: s.src_net.clone(), - src_port: s.src_port, - has_src_port: s.has_src_port, - }, - None => Self::default(), - } +/// One-line human summary of the hidden predicates, for that notice. +pub fn hidden_scope_summary(scope: &proto::RuleScope) -> String { + let mut parts = Vec::new(); + if scope.has_direction { + parts.push( + match proto::Direction::try_from(scope.direction) { + Ok(proto::Direction::Inbound) => "inbound", + Ok(proto::Direction::Outbound) => "outbound", + _ => "direction ?", + } + .to_string(), + ); } - - /// True when the rule carries a constraint the editor cannot show, so - /// the view can tell the user rather than let them assume the visible - /// fields are the whole rule. - pub fn is_set(&self) -> bool { - self.has_uid - || !self.exe_sha256.is_empty() - || !self.parent_exe.is_empty() - || self.has_direction - || !self.src_net.is_empty() - || self.has_src_port - } - - /// One-line human summary of the hidden predicates, for that notice. - pub fn summary(&self) -> String { - let mut parts = Vec::new(); - if self.has_direction { - parts.push( - match proto::Direction::try_from(self.direction) { - Ok(proto::Direction::Inbound) => "inbound", - Ok(proto::Direction::Outbound) => "outbound", - _ => "direction ?", - } - .to_string(), - ); - } - if !self.src_net.is_empty() { - parts.push(format!("from {}", self.src_net)); - } - if self.has_src_port { - parts.push(format!("src port {}", self.src_port)); - } - if self.has_uid { - parts.push(format!("uid {}", self.uid)); - } - if !self.parent_exe.is_empty() { - parts.push(format!("parent {}", self.parent_exe)); - } - if !self.exe_sha256.is_empty() { - let short: String = self.exe_sha256.chars().take(12).collect(); - parts.push(format!("sha256 {short}...")); - } - parts.join(", ") + if !scope.src_net.is_empty() { + parts.push(format!("from {}", scope.src_net)); + } + if scope.has_src_port { + parts.push(format!("src port {}", scope.src_port)); } + if scope.has_uid { + parts.push(format!("uid {}", scope.uid)); + } + if !scope.parent_exe.is_empty() { + parts.push(format!("parent {}", scope.parent_exe)); + } + if !scope.exe_sha256.is_empty() { + let short: String = scope.exe_sha256.chars().take(12).collect(); + parts.push(format!("sha256 {short}...")); + } + parts.join(", ") } impl Default for RuleEditor { @@ -239,7 +205,7 @@ impl Default for RuleEditor { created_at_unix_ms: 0, hit_count: 0, enabled: true, - carried_scope: CarriedScope::default(), + carried_scope: proto::RuleScope::default(), } } } @@ -271,7 +237,7 @@ impl RuleEditor { created_at_unix_ms: rule.created_at_unix_ms, hit_count: rule.hit_count, enabled: rule.enabled, - carried_scope: CarriedScope::from_scope(scope), + carried_scope: scope.cloned().unwrap_or_default(), } } @@ -351,27 +317,21 @@ impl RuleEditor { dst_port.to_string() }, protocol, - carried_scope: CarriedScope { + carried_scope: proto::RuleScope { direction: if inbound { direction } else { 0 }, has_direction: inbound, - ..CarriedScope::default() + ..Default::default() }, ..Self::default() } } } -#[derive(Debug, Clone)] -pub struct LiveEntry { - pub event: proto::ConnectionEvent, -} - #[derive(Debug, Clone)] pub struct PromptCard { + /// `deadline_unix_ms` is the wall clock at which the daemon answers this + /// prompt itself; 0 means it attached no deadline. pub event: proto::PromptEvent, - /// Wall clock at which the daemon answers this prompt itself. 0 means - /// the daemon attached no deadline. - pub deadline_unix_ms: i64, /// Wall clock before which the verdict buttons stay disabled. pub armed_at_ms: i64, /// A verdict for it is on its way to the daemon. The card stays until @@ -383,7 +343,6 @@ pub struct PromptCard { impl PromptCard { fn new(event: proto::PromptEvent, now_ms: i64) -> Self { Self { - deadline_unix_ms: event.deadline_unix_ms, event, armed_at_ms: now_ms.saturating_add(PROMPT_ARM_MS), submitting: false, @@ -673,7 +632,7 @@ impl App { let now = self.now_ms; let mut expired: Vec = Vec::new(); self.prompts.retain(|p| { - if format::is_expired(p.deadline_unix_ms, now) { + if format::is_expired(p.event.deadline_unix_ms, now) { expired.push(prompt_label(&p.event)); false } else { @@ -871,7 +830,7 @@ impl App { Message::LiveEvent(ev) => { self.stream_trouble = false; self.session.record(&ev); - self.live.push_front(LiveEntry { event: ev }); + self.live.push_front(ev); while self.live.len() > LIVE_CAP { self.live.pop_back(); } @@ -1792,7 +1751,7 @@ fn build_rule_from_editor(ed: &RuleEditor) -> Result { && dst_net.is_empty() && dst_port.is_none() && ed.protocol.is_none() - && !ed.carried_scope.is_set(); + && !hidden_scope_is_set(&ed.carried_scope); if scope_empty { return Err( "rule must restrict at least one of: exe, dst-host, dst-net, dst-port, protocol".into(), @@ -1813,30 +1772,18 @@ fn build_rule_from_editor(ed: &RuleEditor) -> Result { .into_owned() }; + // Every field this form shows is rebuilt from it; everything else is + // carried, so a predicate the editor cannot show is never dropped and the + // rule never widens on save. let scope = proto::RuleScope { exe_path: exe, - // Not editable here, so preserved rather than dropped: rebuilding - // the scope from the visible fields alone would widen the rule. - exe_sha256: ed.carried_scope.exe_sha256.clone(), - parent_exe: ed.carried_scope.parent_exe.clone(), - uid: ed.carried_scope.uid, - has_uid: ed.carried_scope.has_uid, dst_host: ed.dst_host.trim().to_string(), dst_net: dst_net.to_string(), dst_port: dst_port.map(u32::from).unwrap_or(0), has_dst_port: dst_port.is_some(), protocol: ed.protocol.map(|p| p as i32).unwrap_or(0), has_protocol: ed.protocol.is_some(), - // Carried like exe_sha256 above, and for the same reason - these - // three used to be rebuilt as unset here, two lines under the comment - // explaining why that must not happen. Unset direction means - // outbound, so the visible casualty was every inbound rule touched by - // this editor. - direction: ed.carried_scope.direction, - has_direction: ed.carried_scope.has_direction, - src_net: ed.carried_scope.src_net.clone(), - src_port: ed.carried_scope.src_port, - has_src_port: ed.carried_scope.has_src_port, + ..ed.carried_scope.clone() }; Ok(proto::RuleInfo { @@ -2158,7 +2105,7 @@ mod tests { ..existing_rule() }; let ed = RuleEditor::from_existing(&uid_only); - assert!(ed.carried_scope.is_set()); + assert!(hidden_scope_is_set(&ed.carried_scope)); assert!(build_rule_from_editor(&ed).is_ok()); // A genuinely empty scope is still refused. @@ -2172,11 +2119,11 @@ mod tests { #[test] fn hidden_scope_summary_names_each_predicate() { let ed = RuleEditor::from_existing(&rule_with_hidden_scope()); - let summary = ed.carried_scope.summary(); + let summary = hidden_scope_summary(&ed.carried_scope); assert!(summary.contains("uid 1000"), "{summary}"); assert!(summary.contains("/usr/bin/bash"), "{summary}"); assert!(summary.contains("sha256 abc123def456"), "{summary}"); - assert!(!CarriedScope::default().is_set()); + assert!(!hidden_scope_is_set(&proto::RuleScope::default())); } #[test] @@ -2500,14 +2447,4 @@ mod tests { assert!(!card.armed(5_000 + PROMPT_ARM_MS - 1)); assert!(card.armed(5_000 + PROMPT_ARM_MS)); } - - #[test] - fn prompt_card_captures_the_daemon_deadline() { - let ev = proto::PromptEvent { - prompt_id: "7".into(), - deadline_unix_ms: 1_700_000_000_000, - ..Default::default() - }; - assert_eq!(PromptCard::new(ev, 0).deadline_unix_ms, 1_700_000_000_000); - } } diff --git a/crates/cfc-ui/src/views/live.rs b/crates/cfc-ui/src/views/live.rs index 97a1114..5b83a33 100644 --- a/crates/cfc-ui/src/views/live.rs +++ b/crates/cfc-ui/src/views/live.rs @@ -10,7 +10,7 @@ use iced::widget::{ use iced::{Element, Length}; use std::collections::VecDeque; -use crate::{format, LiveEntry, Message}; +use crate::{format, Message}; #[derive(Debug, Default, Clone, Copy, PartialEq, Eq)] pub enum VerdictFilter { @@ -51,9 +51,9 @@ impl std::fmt::Display for VerdictFilter { } pub struct ListArgs<'a> { - pub live: &'a VecDeque, + pub live: &'a VecDeque, /// Snapshot rendered instead of `live` while the feed is paused. - pub frozen: Option<&'a [LiveEntry]>, + pub frozen: Option<&'a [proto::ConnectionEvent]>, pub filter: &'a str, pub verdict: VerdictFilter, pub new_while_paused: usize, @@ -104,14 +104,15 @@ pub fn view(args: ListArgs<'_>) -> Element<'_, Message> { } = args; let paused = frozen.is_some(); - let source: Box> = match frozen { + let source: Box> = match frozen { Some(f) => Box::new(f.iter()), None => Box::new(live.iter()), }; - let shown: Vec<&LiveEntry> = source - .filter(|e| matches(&e.event, filter, verdict)) - .collect(); - let total = frozen.map(<[LiveEntry]>::len).unwrap_or(live.len()); + let shown: Vec<&proto::ConnectionEvent> = + source.filter(|e| matches(e, filter, verdict)).collect(); + let total = frozen + .map(<[proto::ConnectionEvent]>::len) + .unwrap_or(live.len()); let pause_label = if paused { if new_while_paused > 0 { @@ -189,8 +190,7 @@ pub fn view(args: ListArgs<'_>) -> Element<'_, Message> { .into() } -fn live_row(e: &LiveEntry) -> Element<'_, Message> { - let ev = &e.event; +fn live_row(ev: &proto::ConnectionEvent) -> Element<'_, Message> { let conn = ev.connection.as_ref(); let proc = ev.process.as_ref(); diff --git a/crates/cfc-ui/src/views/prompts.rs b/crates/cfc-ui/src/views/prompts.rs index 1f52c58..470135f 100644 --- a/crates/cfc-ui/src/views/prompts.rs +++ b/crates/cfc-ui/src/views/prompts.rs @@ -555,13 +555,13 @@ fn countdown_row<'a>( timeout_secs: u32, now_ms: i64, ) -> Element<'a, Message> { - let Some(left) = format::remaining_secs(card.deadline_unix_ms, now_ms) else { + let Some(left) = format::remaining_secs(card.event.deadline_unix_ms, now_ms) else { return text("no deadline - waiting for your answer") .size(10) .into(); }; - let fraction = format::countdown_fraction(card.deadline_unix_ms, now_ms, timeout_secs); + let fraction = format::countdown_fraction(card.event.deadline_unix_ms, now_ms, timeout_secs); let style = if left <= URGENT_SECS { crate::theme::countdown_bar_urgent } else { diff --git a/crates/cfc-ui/src/views/rules.rs b/crates/cfc-ui/src/views/rules.rs index ec1cb5e..e8ffc32 100644 --- a/crates/cfc-ui/src/views/rules.rs +++ b/crates/cfc-ui/src/views/rules.rs @@ -407,11 +407,11 @@ fn editor_view(ed: &RuleEditor) -> Element<'_, Message> { // This rule restricts more than the fields above can show. Saving keeps // those predicates, but the user should know they are there rather than // read the visible fields as the whole rule. - let carried: Element<'_, Message> = if ed.carried_scope.is_set() { + let carried: Element<'_, Message> = if crate::hidden_scope_is_set(&ed.carried_scope) { container( text(format!( "also restricted to {} - kept on save, edit with cfc", - ed.carried_scope.summary() + crate::hidden_scope_summary(&ed.carried_scope) )) .size(11), ) From bc8560cfb404de910ea23966678ccf83cd65d8d7 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 097/125] refactor(deploy): drop the log directory the daemon never writes The daemon logs to the journal, yet the unit created /var/log/colony-firewall and kept it writable, and the SELinux module labelled and granted it. ReadWritePaths keeps only /sys/fs/bpf: StateDirectory= and RuntimeDirectory= already exempt their directories from ProtectSystem=strict. One less writable path for a root daemon. --- docs/HARDENING.md | 2 +- docs/TROUBLESHOOTING.md | 2 +- packaging/selinux/README.md | 2 +- packaging/selinux/TESTING.md | 2 +- packaging/selinux/colony_firewall.fc | 1 - packaging/selinux/colony_firewall.if | 24 ------------------------ packaging/selinux/colony_firewall.te | 8 -------- systemd/colony-firewalld.service | 5 +++-- 8 files changed, 7 insertions(+), 39 deletions(-) diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 655aa64..807bcb7 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -439,7 +439,7 @@ to shrink what a code-execution bug could reach: | `SystemCallFilter=bpf perf_event_open` | The two syscalls the eBPF layer needs, named individually | | `SystemCallArchitectures=native` | Closes the 32-bit-syscall bypass of that filter | | `MemoryDenyWriteExecute` | Nothing here JITs; no W+X memory | -| `ProtectSystem=strict`, `ProtectHome`, `ReadWritePaths` | Read-only filesystem apart from the state, runtime and log directories | +| `ProtectSystem=strict`, `ProtectHome`, `ReadWritePaths` | Read-only filesystem apart from the state and runtime directories and the bpffs pin directory | | `PrivateDevices` | Private `/dev` with only pseudo devices: uid 0 cannot open the block devices and write underneath `ProtectSystem` | | `RestrictAddressFamilies` | AF_UNIX, AF_INET, AF_INET6, AF_NETLINK only; no packet sockets | | `RestrictNamespaces`, `LockPersonality`, `RestrictRealtime`, `RestrictSUIDSGID` | Namespace and personality lockdown | diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 9019bcd..98d83e9 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -210,7 +210,7 @@ journalctl -u colony-firewalld -b -g 'opening rule store|durable storage require - `durable storage requires WAL` or `synchronous=FULL`: the path is on a filesystem that cannot hold a WAL journal (a network share, for one), or - outside the unit's `ReadWritePaths`. Keep `[storage] path` on local disk + outside the directories the unit can write. Keep `[storage] path` on local disk under `/var/lib/colony-firewall`. - `newer than this daemon supports`: the package was downgraded. Reinstall the newer one, or restore a backup of `rules.db` that the older version diff --git a/packaging/selinux/README.md b/packaging/selinux/README.md index a7a1197..74b77fb 100644 --- a/packaging/selinux/README.md +++ b/packaging/selinux/README.md @@ -12,7 +12,7 @@ are built around. So: a real module. make -f /usr/share/selinux/devel/Makefile colony_firewall.pp sudo semodule -i colony_firewall.pp sudo restorecon -RvF /usr/bin/colony-firewalld /etc/colony-firewall \ - /var/lib/colony-firewall /var/log/colony-firewall /run/colony-firewall + /var/lib/colony-firewall /run/colony-firewall ``` `selinux-policy-devel` provides that Makefile. The `.spec` in `packaging/rpm` diff --git a/packaging/selinux/TESTING.md b/packaging/selinux/TESTING.md index cf51101..9cc3816 100644 --- a/packaging/selinux/TESTING.md +++ b/packaging/selinux/TESTING.md @@ -70,7 +70,7 @@ this directory): make -f /usr/share/selinux/devel/Makefile colony_firewall.pp sudo semodule -i colony_firewall.pp sudo restorecon -RvF /usr/bin/colony-firewalld /etc/colony-firewall \ - /var/lib/colony-firewall /var/log/colony-firewall /run/colony-firewall + /var/lib/colony-firewall /run/colony-firewall ``` Verify the label took - this is the single most common way a policy "fails" diff --git a/packaging/selinux/colony_firewall.fc b/packaging/selinux/colony_firewall.fc index 414c5f6..158d073 100644 --- a/packaging/selinux/colony_firewall.fc +++ b/packaging/selinux/colony_firewall.fc @@ -14,7 +14,6 @@ /etc/colony-firewall(/.*)? gen_context(system_u:object_r:colony_firewall_conf_t,s0) /var/lib/colony-firewall(/.*)? gen_context(system_u:object_r:colony_firewall_var_lib_t,s0) -/var/log/colony-firewall(/.*)? gen_context(system_u:object_r:colony_firewall_log_t,s0) /run/colony-firewall(/.*)? gen_context(system_u:object_r:colony_firewall_runtime_t,s0) /usr/lib/systemd/system/colony-firewalld\.service -- gen_context(system_u:object_r:colony_firewall_unit_file_t,s0) diff --git a/packaging/selinux/colony_firewall.if b/packaging/selinux/colony_firewall.if index 568b640..38dffd0 100644 --- a/packaging/selinux/colony_firewall.if +++ b/packaging/selinux/colony_firewall.if @@ -43,26 +43,6 @@ interface(`colony_firewall_stream_connect',` stream_connect_pattern($1, colony_firewall_runtime_t, colony_firewall_runtime_t, colony_firewalld_t) ') -######################################## -## -## Read the Colony Firewall Control log directory. -## -## -## -## Domain allowed access. -## -## -# -interface(`colony_firewall_read_log',` - gen_require(` - type colony_firewall_log_t; - ') - - logging_search_logs($1) - list_dirs_pattern($1, colony_firewall_log_t, colony_firewall_log_t) - read_files_pattern($1, colony_firewall_log_t, colony_firewall_log_t) -') - ######################################## ## ## All of the rights needed to administer Colony Firewall Control. @@ -86,7 +66,6 @@ interface(`colony_firewall_admin',` type colony_firewall_conf_t; type colony_firewall_var_lib_t; type colony_firewall_runtime_t; - type colony_firewall_log_t; type colony_firewall_unit_file_t; type var_run_t; ') @@ -105,7 +84,4 @@ interface(`colony_firewall_admin',` allow $1 var_run_t:dir search; admin_pattern($1, colony_firewall_runtime_t) - - logging_search_logs($1) - admin_pattern($1, colony_firewall_log_t) ') diff --git a/packaging/selinux/colony_firewall.te b/packaging/selinux/colony_firewall.te index 081db82..bdf0f94 100644 --- a/packaging/selinux/colony_firewall.te +++ b/packaging/selinux/colony_firewall.te @@ -51,10 +51,6 @@ files_type(colony_firewall_var_lib_t) type colony_firewall_runtime_t; files_pid_file(colony_firewall_runtime_t) -# /var/log/colony-firewall -type colony_firewall_log_t; -logging_log_file(colony_firewall_log_t) - type colony_firewall_unit_file_t; systemd_unit_file(colony_firewall_unit_file_t) @@ -271,10 +267,6 @@ manage_files_pattern(colony_firewalld_t, colony_firewall_var_lib_t, colony_firew allow colony_firewalld_t colony_firewall_var_lib_t:file map; files_var_lib_filetrans(colony_firewalld_t, colony_firewall_var_lib_t, { dir file }) -manage_dirs_pattern(colony_firewalld_t, colony_firewall_log_t, colony_firewall_log_t) -manage_files_pattern(colony_firewalld_t, colony_firewall_log_t, colony_firewall_log_t) -logging_log_filetrans(colony_firewalld_t, colony_firewall_log_t, { dir file }) - manage_dirs_pattern(colony_firewalld_t, colony_firewall_runtime_t, colony_firewall_runtime_t) manage_files_pattern(colony_firewalld_t, colony_firewall_runtime_t, colony_firewall_runtime_t) manage_sock_files_pattern(colony_firewalld_t, colony_firewall_runtime_t, colony_firewall_runtime_t) diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index 5954670..9a193fe 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -101,12 +101,13 @@ ProtectHome=true # process-lifetime links, and `kill -9` on this unit lifts every in-kernel deny # it holds - which is the one thing that layer exists to prevent. The daemon # only ever creates /sys/fs/bpf/colony-firewall/; it writes nothing else there. -ReadWritePaths=/var/lib/colony-firewall /run/colony-firewall /var/log/colony-firewall /sys/fs/bpf +# StateDirectory= and RuntimeDirectory= below are writable without a line here. +# The daemon logs to the journal and owns no log directory. +ReadWritePaths=/sys/fs/bpf RuntimeDirectory=colony-firewall RuntimeDirectoryMode=0755 StateDirectory=colony-firewall StateDirectoryMode=0750 -LogsDirectory=colony-firewall # Sandboxing PrivateTmp=true From c1b3ae0c37869afcce6b30e62d7fe4fd873abd21 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 098/125] ci: drop the duplicate version checks check-versions.sh compares Cargo.toml, the PKGBUILD and the spec on every push and again at tag time, so rhel.yml's path-filtered version-agrees job repeated it with a weaker grep. The release pkgbuild job needs build, which already refused a tag other than v and ran that script. --- .github/workflows/release.yml | 14 +++----------- .github/workflows/rhel.yml | 18 ------------------ 2 files changed, 3 insertions(+), 29 deletions(-) diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 4b61c1d..64abd5a 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -261,7 +261,9 @@ jobs: pkgbuild: if: github.ref_type == 'tag' # Package only after the main build succeeds. Publishing runs separately - # with write permission after both builds complete. + # with write permission after both builds complete. build has already + # refused a tag other than v and run check-versions.sh, + # which holds pkgver to that version, so pkgver matches the tag here. needs: build runs-on: ubuntu-latest container: archlinux:base-devel @@ -289,16 +291,6 @@ jobs: useradd -m builder chown -R builder: . - - name: Verify pkgver matches the tag - env: - REF_NAME: ${{ github.ref_name }} - run: | - PKGVER="$(sed -n 's/^pkgver=//p' pkg/PKGBUILD | head -n1)" - if [ "${REF_NAME}" != "v${PKGVER}" ]; then - echo "::error::pkg/PKGBUILD pkgver=${PKGVER} does not match tag ${REF_NAME}" - exit 1 - fi - - name: updpkgsums - replace the SKIP placeholder with real checksums working-directory: pkg run: | diff --git a/.github/workflows/rhel.yml b/.github/workflows/rhel.yml index 46e2aff..cbaa4a2 100644 --- a/.github/workflows/rhel.yml +++ b/.github/workflows/rhel.yml @@ -319,21 +319,3 @@ jobs: /home/builder/rpmbuild/RPMS/**/*.rpm /home/builder/rpmbuild/SRPMS/*.rpm if-no-files-found: error - - version-agrees: - name: version agrees with Cargo.toml - runs-on: ubuntu-24.04 - steps: - - uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # v7.0.1 - - name: Compare - # Three files now carry the version - Cargo.toml, the PKGBUILD and the - # spec - and a release where they disagree ships a package whose name - # lies about what is inside it. - run: | - set -eu - cargo="$(grep -m1 '^version' Cargo.toml | cut -d'"' -f2)" - pkgbuild="$(grep -m1 '^pkgver=' pkg/PKGBUILD | cut -d= -f2)" - spec="$(grep -m1 '^Version:' packaging/rpm/colony-firewall-control.spec | awk '{print $2}')" - echo "Cargo.toml=$cargo PKGBUILD=$pkgbuild spec=$spec" - test "$cargo" = "$pkgbuild" || { echo "::error::PKGBUILD disagrees with Cargo.toml"; exit 1; } - test "$cargo" = "$spec" || { echo "::error::spec disagrees with Cargo.toml"; exit 1; } From 142a5d30f09a8c1263c30d100d275e80fea31ecd Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 099/125] refactor(ebpf): remove the unattached sendmsg programs Nothing has attached cfc_sendmsg4/6 since the Fast Allow userspace path went. They own no map and no event layout, so the ABI is unchanged. The startup cleanup of a 0.4-0.6 daemon's sendmsg links works on their pin paths and does not need the programs in the object. --- crates/cfc-ebpf/src/main.rs | 70 +++++----------------------- crates/cfc-ebpf/verifier-budget.toml | 23 --------- 2 files changed, 12 insertions(+), 81 deletions(-) diff --git a/crates/cfc-ebpf/src/main.rs b/crates/cfc-ebpf/src/main.rs index f09538c..a832330 100644 --- a/crates/cfc-ebpf/src/main.rs +++ b/crates/cfc-ebpf/src/main.rs @@ -1,6 +1,6 @@ //! Colony Firewall Control - kernel-side eBPF programs. //! -//! Nine programs, of which at most seven are attached at once - the two +//! Seven programs, of which at most five are attached at once - the two //! `_basic` connect variants are the fallback for a kernel that will not //! verify the cookie ones, never loaded alongside them (see `README.md` for //! attach points and capability requirements): @@ -11,7 +11,6 @@ //! | `tracepoint/sched/sched_process_exit` | evict dead pids from `PROCS` + `EXIT_EVENTS` | //! | `cgroup_skb/ingress` | copy DNS response payloads into `DNS_PACKETS` | //! | `cgroup/connect4`, `cgroup/connect6` | refuse `connect()` for already-denied pids, and mark the sockets of fast-allowed ones | -//! | `cgroup/sendmsg4`, `cgroup/sendmsg6` | the mark decision again, for UDP sends that carry a destination | //! //! The first three *observe*. The rest **decide**, and are the only part of //! CFC that enforces without a userspace round trip: their link is pinned to @@ -140,7 +139,7 @@ static DENY_EVENTS: RingBuf = RingBuf::with_byte_size(64 * 1024, 0); /// Processes the daemon has ruled allowed *process-wide* - no destination, port /// or protocol in the rule, and a rule that lasts (`always` / until restart). -/// Read on the `connect()` and `sendmsg()` paths, where a hit marks the socket +/// Read on the `connect()` path, where a hit marks the socket /// so nftables accepts its packets ahead of the queue. /// /// A set, not a verdict map: the deny map next door is consulted first, and an @@ -174,7 +173,7 @@ static FAST_ALLOW_UNTIL: Array = Array::with_max_entries(1, 0); #[map] static FAST_ALLOW_MARK: Array = Array::with_max_entries(1, 0); -/// One record per fast-allowed `connect()`/`sendmsg()`, the `ConnectReport` +/// One record per fast-allowed `connect()`, the `ConnectReport` /// shape shared with `DENY_EVENTS`. A marked flow never reaches NFQUEUE, so /// without this the live feed, the rule hit counts and the "enforcing" heuristic /// would all go quiet on exactly the traffic the firewall handles best. @@ -1109,7 +1108,7 @@ const SOL_SOCKET: i32 = 1; const SO_MARK: i32 = 36; /// The fast path: decide, on every flow-initiating call, whether this socket -/// carries the daemon's mark. Shared by the connect and sendmsg hooks. +/// carries the daemon's mark. Called by the cookie connect hooks. /// /// **Re-decided at every hook that runs, never left in place.** `SO_MARK` /// lives on the socket for as long as the socket does, and the socket may @@ -1120,9 +1119,8 @@ const SO_MARK: i32 = 36; /// every time one of these hooks runs. /// /// Which is not the same as "at every flow start", and the difference is why -/// only TCP is ever marked. `cgroup/connect{4,6}` runs at `connect()`; -/// `cgroup/sendmsg{4,6}` runs for a send that carries a destination. A -/// **connected UDP socket** passes neither again, so a mark given to one could +/// only TCP is ever marked. `cgroup/connect{4,6}` runs at `connect()`, and a +/// **connected UDP socket** never passes it again, so a mark given to one could /// never be taken back - and unreplied UDP is conntrack-NEW on every datagram /// (see `nfqueue.rs`), so that mark would carry every datagram past the queue /// for as long as the socket lived. The body below is the allowlist that @@ -1154,13 +1152,12 @@ fn mark_decision(ctx: &SockAddrContext, tgid: u32, family: u8) { // Two narrower guards were tried before this one and both leaked, which is // why this is an allowlist and not a list of protocols to exclude: // - // * refusing UDP only at `connect()` left the sendmsg hooks free to mark a - // socket that was *already* connected - `sendto` with an explicit + // * refusing UDP only at `connect()` left the sendmsg hooks this object + // carried until 0.7 free to mark a socket that was *already* connected - `sendto` with an explicit // address is legal on a connected UDP socket, and `udp_sendmsg` runs the // hook whenever `msg_name` is supplied. Closed one door, left the other. // * naming UDP at all only covers what someone thought to name. UDP-Lite, - // DCCP and SCTP connect the same way and have no sendmsg hook here - // either. + // DCCP and SCTP connect the same way. // // The cost is small and lands where it does least harm. The ruleset queues // `ct state new`; a UDP peer that answers makes the flow @@ -1213,10 +1210,9 @@ fn mark_decision(ctx: &SockAddrContext, tgid: u32, family: u8) { // order, which is not cosmetic. // // `&&` short-circuits, so the cheaper test goes first: reading a context - // field against comparing a hash-map key. The sendmsg hooks see only UDP - // and now never grant, so with the map lookup first every `sendto` on the - // machine - every DNS query - paid a hash lookup to reach a conclusion the - // protocol alone settles. Reversed, they pay a load and a compare. + // field against comparing a hash-map key. With the map lookup first, every + // UDP `connect()` on the machine paid a hash lookup to reach a conclusion + // the protocol alone settles. Reversed, it pays a load and a compare. // // The protocol test folds in here instead of standing beside `want` as its // own flag. Two booleans live across the tail made the verifier walk the @@ -1313,48 +1309,6 @@ pub fn cfc_connect6_basic(ctx: SockAddrContext) -> i32 { connect_verdict(&ctx, 6, false) } -/// The mark decision for UDP that never calls `connect()`. -/// -/// `sendto()` on an unconnected socket starts a new flow without ever passing -/// the connect hooks. These hooks run the same `mark_decision` there. No -/// refusal here - the in-kernel deny is a `connect()` thing, and unconnected -/// UDP has always been the packet path's to refuse. -/// -/// **Not "on every datagram send"**, which is what this comment used to claim -/// and what the design was reviewed against. `cgroup/sendmsg{4,6}` runs for a -/// send that carries a destination; a `send()` or `write()` on a socket that -/// has already been `connect()`ed does not pass it, so such a socket would -/// keep whatever mark it was given at `connect()` for as long as it is open - -/// past a revocation, past the deadline, past the daemon's death. -/// -/// That is why `mark_decision` never marks a UDP socket at all, from either -/// hook - see the allowlist there. What these hooks still do for UDP is the -/// other half of the decision: a socket that carries our mark without having -/// been granted it - forged by a process that learned the value and was then -/// revoked - has it stripped on its next addressed send. Defence in depth, per -/// datagram, on the one socket type the connect hook cannot reach again. Two -/// earlier versions of this paragraph described designs that had already been -/// replaced; the test in `cargo xtask ebpf-check` that the kernel never writes -/// the grant map is what this one rests on. -/// -/// No `_basic` twins: these exist only for the fast path, which the basic -/// variants do not have, so on a kernel that verifies only the basic connect -/// programs these are simply not attached. -#[cgroup_sock_addr(sendmsg4)] -pub fn cfc_sendmsg4(ctx: SockAddrContext) -> i32 { - let tgid = (bpf_get_current_pid_tgid() >> 32) as u32; - mark_decision(&ctx, tgid, 4); - CONNECT_PROCEED -} - -/// See [`cfc_sendmsg4`]. -#[cgroup_sock_addr(sendmsg6)] -pub fn cfc_sendmsg6(ctx: SockAddrContext) -> i32 { - let tgid = (bpf_get_current_pid_tgid() >> 32) as u32; - mark_decision(&ctx, tgid, 6); - CONNECT_PROCEED -} - // --------------------------------------------------------------------------- // Object metadata // --------------------------------------------------------------------------- diff --git a/crates/cfc-ebpf/verifier-budget.toml b/crates/cfc-ebpf/verifier-budget.toml index 5012246..4321846 100644 --- a/crates/cfc-ebpf/verifier-budget.toml +++ b/crates/cfc-ebpf/verifier-budget.toml @@ -114,29 +114,6 @@ max = 450 observed = 340 max = 450 -# The mark decision alone: no deny lookup, no cookie, no ring record unless a -# grant is honoured. Note what the matrix taught on the first run: 5.10 -# refuses these outright - `unknown func bpf_getsockopt` on a sendmsg hook, -# while the same helper verifies on the connect hooks of that kernel - and -# 6.12 onward accepts them. Neither fact switches the fast path off any more: -# a missing sendmsg hook is a caveat in the report (they can only strip a forged -# mark now), and a missing `group_dead` in the exit record (absent on 5.10 and -# 6.12, present on 6.18) selects the short deadline plus a per-beat sweep of the -# granted pids rather than a refusal. -# -# These two carry the protocol guard as well, and now pay for a decision they -# can no longer reach: with only TCP ever marked, nothing here ever sets a mark, -# and the only thing left to strip is one a process forged. They stay as -# defence in depth where the kernel verifies them; there is no restriction left -# for their removal to lift. -[cfc_sendmsg4] -observed = 267 -max = 350 - -[cfc_sendmsg6] -observed = 269 -max = 350 - # The no-cookie twins, for kernels whose verifier does not offer # bpf_get_socket_cookie to sock_addr programs. Identical enforcement, no # SOCK_PIDS write. `observed` is blank because 7.1.8 verifies the cookie From dc99d136e2c1f347b3a4eb37842aac947611253b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:49:17 +0200 Subject: [PATCH 100/125] docs(changelog): list the removals --- CHANGELOG.md | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 84e5c49..ec398da 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -132,6 +132,15 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). nftables set. When the eBPF layer loads, it also disarms the legacy pinned maps and removes the old sendmsg link pins; with the layer off, without the object or after a failed load, those stay until reboot. +- The `cfc_sendmsg4`/`cfc_sendmsg6` programs, which nothing had attached since + the Fast Allow userspace path went. The eBPF ABI is unchanged. +- `LogsDirectory=colony-firewall` and the `/var/log/colony-firewall` write + access in the unit and the SELinux module (the `colony_firewall_log_t` type + and the `colony_firewall_read_log` interface). The daemon logs to the + journal and never wrote there. An existing directory is left in place. +- Unused library items: `cfc_core::CoreError`, `cfc_core::Result`, + `cfc_core::ResolvedExe`, `exe_path::resolve_scope` and `Resolved::path`, + together with dependencies no crate used. ### Fixed From a322b295d1817c4e4478af4b549092805d9bd00c Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 10:57:20 +0200 Subject: [PATCH 101/125] fix(daemon): keep " (deleted)" for an image on another namespace's mount A process that stayed in the host user namespace could still run bytes a child namespace mounted at /usr/bin/curl, through /proc//root or a passed descriptor. Once they were deleted, the kernel reported "/usr/bin/curl (deleted)" for it and the suffix was dropped, handing it the host curl's path-only rules. The suffix is now dropped only when the image's mount, as the opened image reports it in fdinfo, appears in the process's own mountinfo. The daemon's own mountinfo would not do: its unit gives it a private mount namespace, so no host mount id appears there. --- CHANGELOG.md | 8 ++- crates/cfc-daemon/src/process_resolve.rs | 90 +++++++++++++++++++++--- docs/ARCHITECTURE.md | 3 +- docs/HARDENING.md | 18 ++--- 4 files changed, 99 insertions(+), 20 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index ec398da..3e6c735 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -193,9 +193,11 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - The same namespace could still borrow the host's path by deleting its bytes once running: the kernel's `" (deleted)"` suffix was dropped, and a deleted image names no file to compare with. The suffix is now dropped only - for a process in the daemon's user namespace, so a program in another one - (`unshare -U`, a rootless container) that runs across its own upgrade - matches its rules again only after a restart. + for a process in the daemon's user namespace whose image sits on a mount of + its own mount namespace, since a process can also run such bytes through + `/proc//root` or a passed descriptor. A program in another user + namespace (`unshare -U`, a rootless container) that runs across its own + upgrade matches its rules again only after a restart. - Executables over 64 MiB, such as Chromium, Electron apps and VS Code, have no digest, so every queued packet from them, each retransmit and parallel connection, opened its own prompt. On a root-sealed path they now share one diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index 36f25f8..fd640c3 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -258,12 +258,13 @@ fn resolve_inner( // Only an image with no link left is stripped (`policy_exe_path`): a file // literally named "curl (deleted)" must not pass for "curl". // - // Nor is it stripped for a process in another user namespace. Such a - // process can mount its own bytes at /usr/bin/curl, run them and remove - // them; the former path then names no file to compare the image with - // (`path_names_image`), and stripping would hand it the host curl's - // rules. In our user namespace only root, or a setuid helper root - // installed, sets up mounts. + // Nor is it stripped for a process in another user namespace, or for an + // image on a mount outside the process's own namespace (`finish`). A user + // can mount their own bytes at /usr/bin/curl in a namespace of their own, + // run them from there and remove them; the former path then names no file + // to compare the image with (`path_names_image`), and stripping would hand + // it the host curl's rules. In our user namespace only root, or a setuid + // helper root installed, sets up mounts. let exe_for_provenance = exe.clone(); let exe = policy_exe_path(exe, unlinked); @@ -867,7 +868,7 @@ impl MappedImage { } /// The image's path as `/proc` renders it, its digest, and whether it has - /// no link left. + /// no link left. `link` is `/proc//exe`. fn finish(self, link: &Path) -> Option<(PathBuf, Option, bool)> { let meta = self.file.metadata().ok()?; if image_key(&meta) != self.key { @@ -877,6 +878,12 @@ impl MappedImage { trace!(path = %self.path.display(), "exe path names another file here"); return None; } + // A process can run an image from another namespace's mount (through + // /proc//root or a passed descriptor), and the link then reads + // that mount's path. So only an image on a mount of the process's own + // namespace counts as unlinked. + let unlinked = + meta.nlink() == 0 && mounted_in(&self.file, &link.with_file_name("mountinfo")); if meta.len() > SHA256_MAX_LEN { trace!(len = meta.len(), "exe too large to hash; skipping"); } @@ -886,10 +893,26 @@ impl MappedImage { { return None; } - Some((self.path, sha256, meta.nlink() == 0)) + Some((self.path, sha256, unlinked)) } } +/// Whether `file` was opened through a mount of the namespace whose +/// `mountinfo` is given. The open file pins its mount, so the id cannot be +/// reused while it is checked. False when either cannot be read. +fn mounted_in(file: &fs::File, mountinfo: &Path) -> bool { + use std::os::fd::AsRawFd as _; + let fdinfo = fs::read_to_string(format!("/proc/self/fdinfo/{}", file.as_raw_fd())); + let Some(id) = fdinfo.ok().and_then(|info| { + info.lines() + .find_map(|l| l.strip_prefix("mnt_id:").map(|id| id.trim().to_owned())) + }) else { + return false; + }; + fs::read_to_string(mountinfo) + .is_ok_and(|mounts| mounts.lines().any(|l| l.split(' ').next() == Some(&id))) +} + /// Whether `path`, read in the daemon's own mount namespace, can stand for /// the mapped image `key` describes. /// @@ -1726,6 +1749,57 @@ mod tests { assert_eq!(exe.as_os_str(), deleted); } + #[test] + fn a_deleted_image_from_another_namespace_mount_keeps_its_suffix() { + // The process itself stays in this user and mount namespace but runs + // bytes a child namespace mounted, through /proc//root. + use std::io::{BufRead as _, BufReader}; + use std::process::{Command, Stdio}; + let Some(sleep) = ["/usr/bin/sleep", "/bin/sleep"] + .into_iter() + .map(Path::new) + .find(|p| p.exists()) + else { + return; + }; + let dir = tempfile::tempdir().unwrap(); + let mut holder = Command::new("unshare") + .args(["-rm", "sh", "-c"]) + .arg(r#"mount -t tmpfs tmpfs "$0" && cp "$1" "$0/sleep" && echo ready && read _"#) + .arg(dir.path()) + .arg(sleep) + .stdin(Stdio::piped()) + .stdout(Stdio::piped()) + .stderr(Stdio::null()) + .spawn() + .unwrap(); + let mut ready = String::new(); + let _ = BufReader::new(holder.stdout.take().unwrap()).read_line(&mut ready); + if ready.trim() != "ready" { + let _ = holder.wait(); + return; // no unprivileged user namespaces here + } + let image = dir.path().join("sleep"); + let via_child = PathBuf::from(format!("/proc/{}/root", holder.id())) + .join(image.strip_prefix("/").unwrap()); + let mut child = Command::new(&via_child).arg("30").spawn().unwrap(); + let link = format!("/proc/{}/exe", child.id()); + let deadline = Instant::now() + Duration::from_secs(5); + while fs::read_link(&link).ok().as_deref() != Some(&image) && Instant::now() < deadline { + std::thread::sleep(Duration::from_millis(10)); + } + fs::remove_file(&via_child).unwrap(); + let exe = resolve(child.id()).exe; + let _ = child.kill(); + let _ = child.wait(); + drop(holder.stdin.take()); + let _ = holder.wait(); + + let mut deleted = image.into_os_string(); + deleted.push(DELETED_SUFFIX); + assert_eq!(exe.as_os_str(), deleted); + } + #[test] fn a_path_naming_another_file_here_is_not_the_image() { let dir = tempfile::tempdir().unwrap(); diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index fa333fa..c0ea326 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -192,7 +192,8 @@ still agree before publishing executable identity. The link is rendered in the process's own mount namespace, so a path that names a different file in the daemon's view leaves the executable unknown. A deleted image's `" (deleted)"` suffix is dropped only for a process in the daemon's user -namespace; elsewhere the former path cannot be checked. A digest is cached only +namespace whose image sits on a mount of its own mount namespace; elsewhere +the former path cannot be checked. A digest is cached only when the image's ctime was at least 2 seconds old as hashing began. Userspace cannot set ctime and any write moves it, so a changed image misses the cache; an unchanged one is never rehashed per packet on the single worker. The diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 807bcb7..5703d59 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -243,14 +243,16 @@ is a separate launch mode. chosen arguments or with `LD_PRELOAD`, which a hash pin does not prevent. An image deleted while it runs is matched by the path it had (an upgraded program keeps its rules), which the daemon cannot check against anything. - Only a process in the daemon's own user namespace gets that path; one in - another user namespace (`unshare -U`, a rootless container) keeps the - `" (deleted)"` suffix, so its rules stop matching until it restarts. A - setuid mount helper such as setuid `bwrap` lets a user present a path - that way too. Digests and the - root-sealed test trust what the filesystem reports: on a FUSE filesystem - the daemon can read (`user_allow_other` in `/etc/fuse.conf`), the user who - mounted it controls both. An image the daemon cannot open at all, such as + Only a process in the daemon's own user namespace, running an image from a + mount of its own mount namespace, gets that path. One in another user + namespace (`unshare -U`, a rootless container), or one running an image + through another namespace's mount (`/proc//root`, a passed + descriptor), keeps the `" (deleted)"` suffix, so its rules stop matching + until it restarts. A setuid mount helper such as setuid `bwrap` lets a + user present a path that way too. Digests and the root-sealed test trust + what the filesystem reports: on a FUSE filesystem the daemon can read + (`user_allow_other` in `/etc/fuse.conf`), the user who mounted it controls + both. An image the daemon cannot open at all, such as an AppImage or anything on a FUSE mount without `allow_other`, has no executable identity, so an Allow scoped to its path never applies to it. - **Raw and packet sockets**: applications with `CAP_NET_RAW` can use AF_PACKET From 1c09a2180f7940706414211c52d06d221adb6d84 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:00:04 +0200 Subject: [PATCH 102/125] fix(clients): send a stored executable path back without rechecking it The daemon accepts a rule's stored path sent back unchanged, but the GUI rechecked every upsert and `cfc rules import` every path in the file, so toggling, editing or restoring a rule whose target became an alias still failed on the client. The GUI now checks only a typed or changed path, and import checks a path only when it is not already stored under the rule's id. --- CHANGELOG.md | 3 ++- crates/cfc-cli/src/rules.rs | 53 +++++++++++++++++++++++++++++++++++-- crates/cfc-ui/src/main.rs | 37 +++++++++++++++++++------- 3 files changed, 81 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3e6c735..299c6db 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -224,7 +224,8 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). - A rule whose stored executable path later became an alias (a legacy `/bin/curl`, or a target a package turned into a symlink) could not be disabled, renamed or re-imported, only deleted. A path sent back unchanged - is accepted; new and changed paths are still checked. + is accepted by the daemon, the GUI's toggle and editor, and + `cfc rules import`; new and changed paths are still checked. - A new timed rule took its creation date from the client, so a date in the future kept "allow for 90s" alive indefinitely. Dates are clamped to now. - Enabled legacy hostname rules refuse flows that are logged as the default diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index a1e87eb..40ad63a 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -627,12 +627,17 @@ pub async fn import( let rules: Vec = serde_json::from_str(&json).context("parsing JSON")?; // Parse and validate the complete file before the atomic server-side batch. + let stored = client.list_rules().await?; let mut pending = Vec::with_capacity(rules.len()); let mut problems = Vec::new(); let mut seen_ids: std::collections::HashMap = std::collections::HashMap::new(); for r in rules { match r.try_into_proto() { Ok(pb) => { + if let Err(e) = check_new_exe(&pb, &stored) { + problems.push(e); + continue; + } // Two rules sharing an id are not two rules: the second upsert // overwrites the first, so the file describes a state the // import cannot produce and the count printed at the end is @@ -711,6 +716,35 @@ pub async fn import( Ok(()) } +/// Validates an imported rule's executable path in the caller's namespace, +/// where `ProtectHome` and `PrivateTmp` hide nothing, before the daemon does +/// in its own. +/// +/// A rule that sends back the path already stored under its id is left alone, +/// as the daemon leaves it: that path may have become an alias since it was +/// written, and refusing it made a `--replace` restore of an export fail for +/// as long as such a rule existed. +fn check_new_exe(rule: &proto::RuleInfo, stored: &[proto::RuleInfo]) -> Result<(), String> { + let Some(exe) = rule + .scope + .as_ref() + .map(|scope| scope.exe_path.as_str()) + .filter(|exe| !exe.is_empty()) + else { + return Ok(()); + }; + let kept = !rule.id.is_empty() + && stored + .iter() + .any(|old| old.id == rule.id && old.scope.as_ref().is_some_and(|s| s.exe_path == exe)); + if kept { + return Ok(()); + } + cfc_core::exe_path::resolve_policy(std::path::Path::new(exe)) + .map(drop) + .map_err(|error| format!("rule `{}`: {error}", rule.name)) +} + // --------------------------------------------------------------------------- // Export / import format // --------------------------------------------------------------------------- @@ -933,8 +967,6 @@ impl ExportedRule { match on absolute executable paths, so it could never fire" )); } - cfc_core::exe_path::resolve_policy(std::path::Path::new(exe)) - .map_err(|error| format!("rule `{name}`: {error}"))?; } let exe_sha256 = match self.scope.exe_sha256.as_deref() { Some(h) => Some( @@ -2520,6 +2552,23 @@ mod json_tests { assert!(rule.try_into_proto().is_err()); } + // A legacy rule whose path became an alias blocked every restore. + #[test] + fn import_checks_only_a_new_or_changed_executable_path() { + // /proc/self/exe is a symlink on every Linux: an alias. + let mut rule = exported("allow"); + rule.id = "1f0a5c7e-0000-4000-8000-000000000001".into(); + rule.scope.exe_path = Some("/proc/self/exe".into()); + let pb = rule + .try_into_proto() + .expect("the daemon decides on aliases"); + assert!(check_new_exe(&pb, &[]).is_err()); + assert!(check_new_exe(&pb, std::slice::from_ref(&pb)).is_ok()); + let mut moved = pb.clone(); + moved.scope.as_mut().unwrap().exe_path = "/proc/self/cwd".into(); + assert!(check_new_exe(&moved, std::slice::from_ref(&pb)).is_err()); + } + // A restored backup used to start a timed Allow's lifetime again. #[test] fn a_timed_rule_keeps_its_deadline_through_export_and_import() { diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index b215f53..24e0543 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -137,8 +137,8 @@ pub struct RuleEditor { /// a sha256-pinned allow would lose its binary pin, and an inbound rule /// (unset direction means outbound) would turn outbound with its source /// scope gone. Its visible fields are stale once the form is edited, so - /// only [`hidden_scope_is_set`] and [`hidden_scope_summary`] read it, and - /// saving overwrites them from the form. + /// only [`hidden_scope_is_set`], [`hidden_scope_summary`] and the + /// unchanged-path check read it, and saving overwrites them from the form. pub carried_scope: proto::RuleScope, } @@ -1658,14 +1658,12 @@ async fn fetch_rules(path: PathBuf) -> Result, String> { } /// Saves `rule` and returns the log line describing what was stored. +/// +/// No executable validation here: the editor validates a path the user typed +/// or changed, and the enable toggle sends the stored path back unchanged, +/// which the daemon accepts as is. Checking it again refused to toggle a rule +/// whose target had since become an alias. async fn upsert_rule(path: PathBuf, rule: proto::RuleInfo) -> Result { - if let Some(scope) = rule - .scope - .as_ref() - .filter(|scope| !scope.exe_path.is_empty()) - { - cfc_core::exe_path::resolve_policy(std::path::Path::new(&scope.exe_path))?; - } let line = saved_rule_line(&rule); let mut client = Client::connect(&path).await.map_err(|e| e.to_string())?; client.upsert_rule(rule).await.map_err(|e| e.to_string())?; @@ -1762,9 +1760,14 @@ fn build_rule_from_editor(ed: &RuleEditor) -> Result { // and PrivateTmp hide exactly the paths a person commonly enters. // Not trimmed: a file name may end in a space, and saving a rule for // "/opt/app " must not quietly retarget it to "/opt/app". + // An edited rule's unchanged path is sent back as stored, as the daemon + // expects: it may have become an alias since, and refusing it here left + // such a rule impossible to rename or retarget anywhere but the CLI. let typed = ed.exe.as_str(); let exe = if typed.trim().is_empty() { String::new() + } else if ed.editing_id.is_some() && typed == ed.carried_scope.exe_path { + typed.to_string() } else { cfc_core::exe_path::resolve_policy(std::path::Path::new(typed))? .into_path() @@ -1926,6 +1929,22 @@ mod tests { assert_eq!(rule.scope.unwrap().exe_path, "/usr/bin/curl"); } + #[test] + fn editor_sends_an_unchanged_alias_back_but_refuses_a_typed_one() { + // /proc/self/exe is a symlink on every Linux, so it stands in for a + // stored path that has since become an alias. + let mut rule = existing_rule(); + rule.scope.as_mut().unwrap().exe_path = "/proc/self/exe".into(); + let mut ed = RuleEditor::from_existing(&rule); + ed.name = "renamed".into(); + let saved = build_rule_from_editor(&ed).expect("unchanged path is kept"); + assert_eq!(saved.scope.unwrap().exe_path, "/proc/self/exe"); + + let mut fresh = editor_with_scope(); + fresh.exe = "/proc/self/exe".into(); + assert!(build_rule_from_editor(&fresh).is_err()); + } + #[test] fn observed_seed_scopes_by_program_or_pins_the_endpoint_without_one() { let ed = RuleEditor::from_observed("/bin/x", "1.2.3.4", 443, 1, 0); From 3cc5318003ff7fbf08ddf677a223e6b613f2d6a9 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:00:04 +0200 Subject: [PATCH 103/125] fix(ebpf): drop the oldest exit candidates when the queue is full The recheck put survivors at the front and truncated the tail, which threw away the candidates that arrived during the tick. Leaders whose workers never exit could fill the queue for good and every later exit was forgotten. The oldest are now dropped instead, as documented. --- crates/cfc-daemon/src/ebpf/loader.rs | 28 +++++++++++++++++++++++++--- 1 file changed, 25 insertions(+), 3 deletions(-) diff --git a/crates/cfc-daemon/src/ebpf/loader.rs b/crates/cfc-daemon/src/ebpf/loader.rs index 1be3c69..01c54be 100644 --- a/crates/cfc-daemon/src/ebpf/loader.rs +++ b/crates/cfc-daemon/src/ebpf/loader.rs @@ -1303,14 +1303,24 @@ fn spawn_exit_recheck( }) .collect(); if !waiting.is_empty() { - let mut queue = pending.lock(); - queue.splice(0..0, waiting); - queue.truncate(MAX_PENDING_EXITS); + requeue(&mut pending.lock(), waiting); } } }) } +/// Puts the candidates still waiting back ahead of those that arrived during +/// the tick, then drops the oldest past the cap. +/// +/// Truncating the tail instead dropped the new arrivals, so a set of leaders +/// whose workers never exit filled the queue for good and every later exit +/// was forgotten. +fn requeue(queue: &mut Vec<(u32, Option)>, waiting: Vec<(u32, Option)>) { + queue.splice(0..0, waiting); + let excess = queue.len().saturating_sub(MAX_PENDING_EXITS); + queue.drain(..excess); +} + /// Takes a ring-buffer map out of the object and starts a task that drains it. /// /// `AsyncFd` is the pattern aya's own docs point at: `RingBuf` implements @@ -1378,6 +1388,18 @@ fn decode(bytes: &[u8]) -> Option { mod tests { use super::*; + #[test] + fn a_full_exit_queue_keeps_the_newest_candidates() { + let waiting: Vec<_> = (1..=MAX_PENDING_EXITS as u32) + .map(|pid| (pid, None)) + .collect(); + let mut queue = vec![(99_999, Some(7))]; + requeue(&mut queue, waiting); + assert_eq!(queue.len(), MAX_PENDING_EXITS); + assert_eq!(queue.first(), Some(&(2, None))); + assert_eq!(queue.last(), Some(&(99_999, Some(7)))); + } + #[test] fn a_live_thread_group_is_not_evicted() { assert!(!process_group_is_gone(std::process::id())); From d606d5d2fec79c13628cebc1540792ecff72334e Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:04:40 +0200 Subject: [PATCH 104/125] fix(ui): disarm a prompt card's buttons when it moves up A card's verdict buttons were armed one second after it arrived and never again, so when the card above was answered or expired, the card that moved into its place was live under the cursor. The second click of a double-click on "Always allow" for one program wrote an always-allow rule for the next one. Removing cards now re-arms every card below the removed one for a second, counted from the removal. --- CHANGELOG.md | 5 ++-- crates/cfc-ui/src/main.rs | 58 ++++++++++++++++++++++++++++++++++++--- 2 files changed, 57 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 299c6db..2ac2eb0 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -82,8 +82,9 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). for another program. They now answer the marked top card, only on the Prompts tab and without those modifiers. The keys are disarmed for one second whenever their target changes, and a card's buttons for one second - after it appears, so input already on its way when the window was raised - or a card moved does not answer it. + after it appears or moves up, so input already on its way when the window + was raised or a card moved, such as the second click of a double-click on + the card above, does not answer it. - GUI and tray: executable paths, command lines, working directories and DNS names were shown raw, so bidi and control characters could reorder or add lines to a prompt, and the tray's notification body was parsed as markup diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 24e0543..6a080b6 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -332,7 +332,8 @@ pub struct PromptCard { /// `deadline_unix_ms` is the wall clock at which the daemon answers this /// prompt itself; 0 means it attached no deadline. pub event: proto::PromptEvent, - /// Wall clock before which the verdict buttons stay disabled. + /// Wall clock before which the verdict buttons stay disabled: a second + /// after the card arrives, and again after it moves up. pub armed_at_ms: i64, /// A verdict for it is on its way to the daemon. The card stays until /// the daemon confirms, so a verdict that never arrived can be given @@ -631,7 +632,7 @@ impl App { let now = self.now_ms; let mut expired: Vec = Vec::new(); - self.prompts.retain(|p| { + self.retire_cards(|p| { if format::is_expired(p.event.deadline_unix_ms, now) { expired.push(prompt_label(&p.event)); false @@ -695,12 +696,35 @@ impl App { } } + /// Drops the cards `keep` rejects and disarms, for [`PROMPT_ARM_MS`], + /// every card that moved up to take a dropped card's place. + /// + /// A card's buttons are otherwise armed from its arrival, so the second + /// click of a double-click on one card, or a click on a card that + /// expired under the cursor, landed on the button of the program below + /// that just moved into the same spot. + fn retire_cards(&mut self, mut keep: impl FnMut(&PromptCard) -> bool) { + let rearm_at = self.now_ms.saturating_add(PROMPT_ARM_MS); + let mut moved = false; + self.prompts.retain_mut(|card| { + if !keep(card) { + moved = true; + return false; + } + if moved { + card.armed_at_ms = card.armed_at_ms.max(rearm_at); + } + true + }); + } + /// Settles the card a verdict was sent for: gone once the daemon has /// applied an answer, answerable again when nothing was applied. fn settle_card(&mut self, prompt_id: &str, applied: bool) { + // The arming window below counts from now, not from the last tick. + self.now_ms = now_ms(); if applied { - self.prompts - .retain(|card| card.event.prompt_id != prompt_id); + self.retire_cards(|card| card.event.prompt_id != prompt_id); } else if let Some(card) = self .prompts .iter_mut() @@ -2447,6 +2471,32 @@ mod tests { ); } + #[test] + fn a_card_that_moves_up_is_disarmed_again() { + let (mut app, _) = App::new(); + app.prompts.push(PromptCard::new( + prompt("top", "/usr/lib/firefox/firefox"), + 0, + )); + app.prompts.push(PromptCard::new( + prompt("below", "/home/u/.cache/x/updater"), + 0, + )); + assert!(app.prompts[1].armed(app.now_ms)); + let _ = app.update(Message::SubmitVerdict { + prompt_id: "top".into(), + action: proto::Action::Allow, + scope: None, + duration: proto::Duration::Always, + }); + let _ = app.update(Message::VerdictSubmitted(Ok(("top".into(), true, None)))); + // The second click of a double-click lands on the card that took + // the answered one's place. + assert_eq!(app.prompts[0].event.prompt_id, "below"); + assert!(!app.prompts[0].armed(app.now_ms)); + assert!(app.prompts[0].armed(app.now_ms + PROMPT_ARM_MS)); + } + #[test] fn pause_ignores_a_click_right_after_reconnecting() { let (mut app, _) = App::new(); From 6ab03070b1e8150c5ebcc12bb30cdc8affbb3976 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:04:55 +0200 Subject: [PATCH 105/125] fix(ui): disarm Pause after enforcement resumes Pause shares its header slot with Resume, and was only held disabled after a handshake. Once set_paused(false) was confirmed, Resume turned into an enabled Pause under the cursor, so a double-click on Resume paused enforcement again. Pause is now disabled for a second whenever it takes the slot: after a handshake, after a confirmed resume, and when a status poll shows a pause ended elsewhere. --- CHANGELOG.md | 5 +++-- crates/cfc-ui/src/main.rs | 40 +++++++++++++++++++++++++++++++-------- 2 files changed, 35 insertions(+), 10 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2ac2eb0..09965e7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -246,8 +246,9 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). expired was closed with the user's edits; it now stays open as a new rule. Footer errors are no longer pushed out by a burst of warnings. - GUI: Pause replaced Reconnect under the cursor as soon as the daemon came - back, so a double-click on Reconnect paused enforcement. Pause now stays - disabled for one second after connecting. + back, and Resume as soon as enforcement resumed, so a double-click on + either paused enforcement. Pause now stays disabled for one second after + connecting and after resuming. - Tray: on GNOME, three expired prompt bubbles held every actionable slot, so later prompts only reached the overflow bubble, which cannot answer them. Slots are freed once their prompt's deadline has passed, and the diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 6a080b6..eec2c07 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -91,9 +91,10 @@ pub struct App { pub status_ticks: u32, /// Failed reconnect attempts, feeding the backoff. pub retry_attempts: u32, - /// When the last handshake succeeded. Pause stays disabled for - /// [`PROMPT_ARM_MS`] after it (see [`App::pause_armed`]). - pub connected_at_ms: i64, + /// When Pause last took its header slot: a handshake (it replaces + /// Reconnect) or enforcement resuming (it replaces Resume). Pause stays + /// disabled for [`PROMPT_ARM_MS`] after it (see [`App::pause_armed`]). + pub pause_shown_at_ms: i64, pub retry_at_ms: Option, /// Set when a gRPC stream drops; the badge shows "reconnecting" instead /// of the footer being rewritten every two seconds. @@ -557,7 +558,7 @@ impl App { status_ticks: 0, retry_attempts: 0, retry_at_ms: None, - connected_at_ms: 0, + pause_shown_at_ms: 0, stream_trouble: false, now_ms: now_ms(), }; @@ -577,11 +578,12 @@ impl App { } /// Pause takes the place of Reconnect, at the right end of the header, - /// as soon as a handshake lands. The second click of a double-click on - /// Reconnect would otherwise switch enforcement off with no + /// as soon as a handshake lands, and the place of Resume as soon as + /// enforcement resumes. The second click of a double-click on Reconnect + /// or Resume would otherwise switch enforcement off with no /// confirmation. fn pause_armed(&self) -> bool { - self.now_ms >= self.connected_at_ms.saturating_add(PROMPT_ARM_MS) + self.now_ms >= self.pause_shown_at_ms.saturating_add(PROMPT_ARM_MS) } fn connect_task(&mut self) -> Task { @@ -747,7 +749,7 @@ impl App { } Message::HandshakeDone(Ok(data)) => { self.now_ms = now_ms(); - self.connected_at_ms = self.now_ms; + self.pause_shown_at_ms = self.now_ms; self.daemon = DaemonState::Connected; self.status = Some(data.status); self.rules = data.rules; @@ -801,6 +803,10 @@ impl App { Task::none() } Message::StatusLoaded(Ok(s)) => { + // A timed pause ran out, or another client resumed. + if self.status.as_ref().is_some_and(|old| old.paused) && !s.paused { + self.pause_shown_at_ms = self.now_ms; + } self.status = Some(s); self.status_failures = 0; self.stream_trouble = false; @@ -1056,6 +1062,10 @@ impl App { Task::perform(set_paused(socket, !current), Message::PausedSet) } Message::PausedSet(Ok((paused, resume_at_unix_ms))) => { + if !paused { + self.now_ms = now_ms(); + self.pause_shown_at_ms = self.now_ms; + } if let Some(s) = &mut self.status { s.paused = paused; s.resume_at_unix_ms = resume_at_unix_ms; @@ -2497,6 +2507,20 @@ mod tests { assert!(app.prompts[0].armed(app.now_ms + PROMPT_ARM_MS)); } + #[test] + fn pause_ignores_a_click_right_after_resuming() { + let (mut app, _) = App::new(); + app.status = Some(proto::StatusResponse { + paused: true, + ..Default::default() + }); + assert_eq!(app.update(Message::TogglePaused).units(), 1, "Resume"); + let _ = app.update(Message::PausedSet(Ok((false, 0)))); + assert_eq!(app.update(Message::TogglePaused).units(), 0); + app.now_ms += PROMPT_ARM_MS; + assert_eq!(app.update(Message::TogglePaused).units(), 1); + } + #[test] fn pause_ignores_a_click_right_after_reconnecting() { let (mut app, _) = App::new(); From bf1819529c8c6e17604106b8740abe0b148d9f26 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:05:28 +0200 Subject: [PATCH 106/125] fix(ui): seed an inbound "make rule" on the peer, not our own address An inbound flow has no program, so the seed pinned its destination, our own address, with the inbound direction. The daemon refuses dst_net on an inbound rule, and the editor has no source field, so the seed could not be saved, and clearing the address to get past the error left a rule open to every peer. The seed now carries the peer seen as src_net with the local port and protocol, never an executable or a destination address. "Customize" on a prompt goes through the same seed. --- CHANGELOG.md | 8 ++-- crates/cfc-ui/src/main.rs | 67 +++++++++++++++++++++++++-------- crates/cfc-ui/src/views/live.rs | 1 + 3 files changed, 57 insertions(+), 19 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 09965e7..2fd669a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,9 +23,11 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). scope `cfc rules add --exe --dst-port --protocol` builds, instead of pinning the one address seen, which left the app denied on its next address (#46). A row without an identified program pins the address and - never seeds ``; an inbound row keeps its direction. "Customize" - on a prompt seeds the same way. A saved rule logs the scope it stored, and - a rule the editor refuses is also reported in the footer. + never seeds ``; an inbound row keeps its direction and is scoped + on the peer seen and the local port, since the daemon refuses our own + address as an inbound destination. "Customize" on a prompt seeds the same + way. A saved rule logs the scope it stored, and a rule the editor refuses + is also reported in the footer. - GUI: a prompt arriving while others are pending no longer switches to the Prompts tab; only the first one does, and raises the window. - `cfc rules import-opensnitch` stops before changing anything when a source diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index eec2c07..0f113f9 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -254,11 +254,17 @@ impl RuleEditor { .as_ref() .map(|p| p.exe.as_str()) .unwrap_or_default(); - let (dst_ip, dst_port, protocol, direction) = match ev.connection.as_ref() { - Some(c) => (c.dst_ip.as_str(), c.dst_port, c.protocol, c.direction), - None => ("", 0, 0, 0), + let (src_ip, dst_ip, dst_port, protocol, direction) = match ev.connection.as_ref() { + Some(c) => ( + c.src_ip.as_str(), + c.dst_ip.as_str(), + c.dst_port, + c.protocol, + c.direction, + ), + None => ("", "", 0, 0, 0), }; - let mut editor = Self::from_observed(exe, dst_ip, dst_port, protocol, direction); + let mut editor = Self::from_observed(exe, src_ip, dst_ip, dst_port, protocol, direction); editor.prompt_id = Some(ev.prompt_id.clone()); editor.prompt_hash_required = ev.binds_to_hash; if ev.binds_to_hash { @@ -285,10 +291,17 @@ impl RuleEditor { /// Otherwise the placeholder for an unidentified process is never /// seeded, since a rule on it is refused, and the numeric endpoint is /// pinned instead so the rule does not cover the whole port. DNS names - /// stay diagnostic. An inbound flow keeps its direction: without it the - /// rule would be an outbound one to our own address. + /// stay diagnostic. + /// + /// An inbound flow keeps its direction (without it the rule would be an + /// outbound one to our own address) and pins the one peer seen as its + /// source. Inbound, the destination is this machine, which the daemon + /// refuses as a scope, and the editor has no source field: a seed on our + /// own address could not be saved, and clearing it opened the port to + /// every peer. pub fn from_observed( exe: &str, + src_ip: &str, dst_ip: &str, dst_port: u32, protocol: i32, @@ -297,8 +310,9 @@ impl RuleEditor { let protocol = proto::Protocol::try_from(protocol) .ok() .filter(|p| !matches!(p, proto::Protocol::Unspecified)); - let scopable = cfc_client::convert::exe_is_rule_scopable(exe); let inbound = direction == proto::Direction::Inbound as i32; + // Inbound flows are never attributed to a program. + let scopable = !inbound && cfc_client::convert::exe_is_rule_scopable(exe); Self { name: String::new(), exe: if scopable { @@ -307,7 +321,7 @@ impl RuleEditor { String::new() }, dst_host: String::new(), - dst_net: if scopable { + dst_net: if scopable || inbound { String::new() } else { format::host_cidr(dst_ip) @@ -321,6 +335,11 @@ impl RuleEditor { carried_scope: proto::RuleScope { direction: if inbound { direction } else { 0 }, has_direction: inbound, + src_net: if inbound { + format::host_cidr(src_ip) + } else { + String::new() + }, ..Default::default() }, ..Self::default() @@ -472,6 +491,7 @@ pub enum Message { /// Opens the rule editor pre-filled from an observed connection. MakeRuleFromEvent { exe: String, + src_ip: String, dst_ip: String, dst_port: u32, protocol: i32, @@ -1029,13 +1049,14 @@ impl App { } Message::MakeRuleFromEvent { exe, + src_ip, dst_ip, dst_port, protocol, direction, } => { self.editor = Some(RuleEditor::from_observed( - &exe, &dst_ip, dst_port, protocol, direction, + &exe, &src_ip, &dst_ip, dst_port, protocol, direction, )); self.tab = Tab::Rules; Task::none() @@ -1981,7 +2002,7 @@ mod tests { #[test] fn observed_seed_scopes_by_program_or_pins_the_endpoint_without_one() { - let ed = RuleEditor::from_observed("/bin/x", "1.2.3.4", 443, 1, 0); + let ed = RuleEditor::from_observed("/bin/x", "", "1.2.3.4", 443, 1, 0); assert_eq!(ed.exe, "/bin/x"); assert!(ed.dst_host.is_empty()); assert!( @@ -1991,18 +2012,31 @@ mod tests { assert_eq!(ed.dst_port, "443"); for exe in [cfc_client::convert::UNKNOWN_EXE, ""] { - let ed = RuleEditor::from_observed(exe, "2001:db8::1", 0, 0, 0); + let ed = RuleEditor::from_observed(exe, "", "2001:db8::1", 0, 0, 0); assert!(ed.exe.is_empty(), "{exe:?} is never seeded"); assert_eq!(ed.dst_net, "2001:db8::1/128"); assert!(ed.dst_port.is_empty()); assert!(ed.protocol.is_none()); } + // Inbound: the peer seen, never our own address, which the daemon + // refuses as an inbound scope. let inbound = proto::Direction::Inbound as i32; - let ed = - RuleEditor::from_observed(cfc_client::convert::UNKNOWN_EXE, "10.0.0.2", 22, 1, inbound); - assert!(ed.carried_scope.has_direction); - assert_eq!(ed.carried_scope.direction, inbound); + let ed = RuleEditor::from_observed( + cfc_client::convert::UNKNOWN_EXE, + "192.168.1.20", + "10.0.0.2", + 8384, + 1, + inbound, + ); + let scope = build_rule_from_editor(&ed).unwrap().scope.unwrap(); + assert!(scope.has_direction); + assert_eq!(scope.direction, inbound); + assert_eq!(scope.src_net, "192.168.1.20/32"); + assert!(scope.dst_net.is_empty()); + assert!(scope.exe_path.is_empty()); + assert_eq!(scope.dst_port, 8384); } /// Issue #46: "make rule" on a LIVE row must either send the rule or say @@ -2030,6 +2064,7 @@ mod tests { let (mut app, _) = App::new(); let _ = app.update(Message::MakeRuleFromEvent { exe: exe.into(), + src_ip: "10.0.0.2".into(), dst_ip: dst_ip.into(), dst_port, protocol: protocol as i32, @@ -2214,7 +2249,7 @@ mod tests { #[test] fn a_rule_seeded_from_an_observed_flow_is_new() { - let ed = RuleEditor::from_observed("/bin/x", "1.2.3.4", 443, 1, 0); + let ed = RuleEditor::from_observed("/bin/x", "", "1.2.3.4", 443, 1, 0); assert_eq!(ed.created_at_unix_ms, 0); assert_eq!(ed.hit_count, 0); assert!(ed.enabled); diff --git a/crates/cfc-ui/src/views/live.rs b/crates/cfc-ui/src/views/live.rs index 5b83a33..006b839 100644 --- a/crates/cfc-ui/src/views/live.rs +++ b/crates/cfc-ui/src/views/live.rs @@ -239,6 +239,7 @@ fn live_row(ev: &proto::ConnectionEvent) -> Element<'_, Message> { .padding([1, 6]) .on_press(Message::MakeRuleFromEvent { exe: proc.map(|p| p.exe.clone()).unwrap_or_default(), + src_ip: c.src_ip.clone(), dst_ip: c.dst_ip.clone(), dst_port: c.dst_port, protocol: c.protocol, From 46ab50fad1bc1988f8bbf71c4d00df44401f7565 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:09:29 +0200 Subject: [PATCH 107/125] fix(rules): recognise bundle rules as 0.3.0 through 0.7.0 resolved them Copies seeded before 0.7.0 pinned the first candidate that existed, a launcher included, from candidate lists later re-pathed for git, apt and the browsers, and for entries since dropped (Epiphany, npm, pip). None of them matched what the entry resolves to now, so `bundle add` stopped on the bundle's own rule and `bundle remove` left it in place, among them the "any Node program to 443" Allow. Keep the old candidate lists for the changed and dropped entries, and count a same-named rule as the bundle's own when it grants exactly what the entry grants now or granted then. The dropped names come from the same table, replacing RETIRED_BUNDLE_RULES, and `bundle add` names a legacy copy that pins an old path as it already did for current ones. --- CHANGELOG.md | 8 +- crates/cfc-cli/src/main.rs | 3 +- crates/cfc-cli/src/rules.rs | 271 +++++++++++++++++++++++++++++------- 3 files changed, 231 insertions(+), 51 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2fd669a..a25d605 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -264,9 +264,11 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). grants. SIGINT, SIGHUP, SIGQUIT and SIGTERM now stop the tree at any point, and its identity is printed before it starts. - `cfc rules bootstrap-defaults` and `bundle add` failed on hosts seeded - before 0.7.0, calling the bundle's own rules outside it. An identical - same-named rule now counts as present, and `bundle remove` removes it; a - different one still stops the command. + before 0.7.0, calling the bundle's own rules outside it. A same-named rule + identical to what the entry installs here now, or to what 0.3.0 through + 0.7.0 installed here (the old path, Epiphany, npm and pip included), counts + as present, and `bundle remove` removes it; a different one still stops + the command. - `cfc rules bundle remove` deleted a bundle rule the user had edited into a deny. It now keeps any of its rules that is no longer an allow. - OpenSnitch import passed `dest.ip` networks (`10.0.0.0/8/32`), bad CIDRs diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index 0217764..030fe9a 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -223,7 +223,8 @@ enum BundleCmd { /// /// Matches the ids the bundle gave its rules, never a name or a prefix, /// so a rule you wrote yourself is never caught by it. Rules seeded - /// before 0.7.0 are removed only while identical to the bundle's entry. + /// before 0.7.0 are removed only while identical to what the entry + /// installs now or installed then. /// A bundle rule you edited into a deny or reject is kept. Remove { /// Bundle name (see `cfc rules bundle list`). diff --git a/crates/cfc-cli/src/rules.rs b/crates/cfc-cli/src/rules.rs index 40ad63a..7585a65 100644 --- a/crates/cfc-cli/src/rules.rs +++ b/crates/cfc-cli/src/rules.rs @@ -1456,6 +1456,7 @@ fn apply_simple(s: &OsnSimple, scope: &mut proto::RuleScope) -> anyhow::Result<( /// binaries come first, and launchers are skipped (see [`is_launcher`]). /// Tools where only an interpreter connects (npm, pip) are not bundled at all: /// allowing `/usr/bin/node` to reach 443 would allow every Node program. +#[derive(Clone, Copy)] struct BundleRule { name: &'static str, /// Absolute paths to try, in order. First one that exists wins. @@ -1493,6 +1494,19 @@ impl BundleRule { .map(|p| cfc_core::exe_path::resolve(std::path::Path::new(p)).into_path()) .find(|p| !is_launcher(p)) } + + /// The path 0.3.0 through 0.7.0 pinned: the first candidate that exists, + /// launcher or not. + fn resolve_pre_0_7(&self) -> Option { + if self.exe_candidates.is_empty() { + return Some(PathBuf::new()); + } + self.exe_candidates + .iter() + .map(std::path::Path::new) + .find(|p| p.is_file()) + .map(|p| cfc_core::exe_path::resolve(p).into_path()) + } } /// True for a file that execs another image instead of connecting itself: a @@ -2055,29 +2069,134 @@ fn same_policy(rule: &proto::RuleInfo, wanted: &proto::RuleInfo) -> bool { && rule.scope == wanted.scope } +/// Entries whose candidates changed after 0.7.0, or that were dropped, as +/// 0.3.0 through 0.7.0 shipped them: `(bundle, entry, candidates, port)`, all +/// TCP. Epiphany fetches through the WebKit network process every WebKitGTK +/// app shares, and npm and pip connect as their interpreter, so those went. +const PRE_0_7_CHANGED: &[(&str, &str, &[&str], u16)] = &[ + ( + "updates", + "updates-apt-https", + &["/usr/bin/apt-get", "/usr/lib/apt/methods/https"], + 443, + ), + ( + "updates", + "updates-apt-http", + &["/usr/bin/apt-get", "/usr/lib/apt/methods/http"], + 80, + ), + ("dev", "dev-git-https", &["/usr/bin/git"], 443), + ("dev", "dev-git-ssh", &["/usr/bin/git"], 22), + ( + "dev", + "dev-npm-https", + &["/usr/bin/npm", "/usr/bin/node"], + 443, + ), + ( + "dev", + "dev-pip-https", + &["/usr/bin/pip", "/usr/bin/pip3"], + 443, + ), + ( + "web", + "web-firefox-https", + &["/usr/bin/firefox", "/usr/lib/firefox/firefox"], + 443, + ), + ( + "web", + "web-firefox-http", + &["/usr/bin/firefox", "/usr/lib/firefox/firefox"], + 80, + ), + ("web", "web-librewolf-https", &["/usr/bin/librewolf"], 443), + ("web", "web-librewolf-http", &["/usr/bin/librewolf"], 80), + ( + "web", + "web-chromium-https", + &["/usr/bin/chromium", "/usr/lib/chromium/chromium"], + 443, + ), + ( + "web", + "web-chromium-http", + &["/usr/bin/chromium", "/usr/lib/chromium/chromium"], + 80, + ), + ( + "web", + "web-chrome-https", + &["/usr/bin/google-chrome-stable"], + 443, + ), + ( + "web", + "web-chrome-http", + &["/usr/bin/google-chrome-stable"], + 80, + ), + ("web", "web-brave-https", &["/usr/bin/brave"], 443), + ("web", "web-brave-http", &["/usr/bin/brave"], 80), + ( + "web", + "web-vivaldi-https", + &["/usr/bin/vivaldi-stable"], + 443, + ), + ("web", "web-vivaldi-http", &["/usr/bin/vivaldi-stable"], 80), + ("web", "web-epiphany-https", &["/usr/bin/epiphany"], 443), + ("web", "web-epiphany-http", &["/usr/bin/epiphany"], 80), +]; + +/// Every entry of `bundle` as 0.3.0 through 0.7.0 shipped it, dropped ones +/// included. +fn pre_0_7_entries(bundle: &Bundle) -> Vec { + let changed = PRE_0_7_CHANGED.iter().filter(|(b, ..)| *b == bundle.name); + bundle + .rules + .iter() + .filter(|r| !PRE_0_7_CHANGED.iter().any(|(_, name, ..)| *name == r.name)) + .copied() + .chain(changed.map(|&(_, name, exe_candidates, port)| BundleRule { + name, + exe_candidates, + dst_port: Some(port), + protocol: Some(proto::Protocol::Tcp), + direction: None, + src_net: None, + })) + .collect() +} + /// Ids of the rules that are this bundle's own entries as seeded before 0.7.0 /// gave bundle rules deterministic ids: same name, and granting exactly what -/// the entry would install here. `bundle add` counts them as present and +/// the entry would install here now, or what 0.3.0 through 0.7.0 installed +/// here (see [`PRE_0_7_CHANGED`]). `bundle add` counts them as present and /// `bundle remove` removes them; an edited copy is neither. fn legacy_copies( bundle: &Bundle, - present: &[(&'static str, PathBuf)], existing: &[proto::RuleInfo], ) -> std::collections::HashSet { - let mut ids = std::collections::HashSet::new(); - for (rule_name, exe) in present { - let Some(spec) = bundle.rules.iter().find(|r| r.name == *rule_name) else { - continue; - }; - let wanted = proto_for(spec, &exe.to_string_lossy()); - ids.extend( - existing + let now = bundle.rules.iter().filter_map(|r| Some((*r, r.resolve()?))); + let then = pre_0_7_entries(bundle) + .into_iter() + .filter_map(|r| Some((r, r.resolve_pre_0_7()?))); + let wanted: Vec = now + .chain(then) + .map(|(spec, exe)| proto_for(&spec, &exe.to_string_lossy())) + .collect(); + existing + .iter() + .filter(|rule| { + wanted .iter() - .filter(|rule| rule.name == *rule_name && same_policy(rule, &wanted)) - .map(|rule| rule.id.clone()), - ); - } - ids + .any(|w| w.name == rule.name && same_policy(rule, w)) + }) + .map(|rule| rule.id.clone()) + .collect() } fn bundle_rule_id(bundle: &str, name: &str) -> String { @@ -2220,7 +2339,7 @@ pub async fn bundle_add( // A rule with an entry's name but another id was not installed by this // bundle. A copy seeded before 0.7.0 counts as present; any other one // stops the command before it changes anything. - let legacy = legacy_copies(&bundle, &planned.present, &existing); + let legacy = legacy_copies(&bundle, &existing); for (rule_name, _) in &planned.present { let id = bundle_rule_id(bundle.name, rule_name); if let Some(rule) = existing @@ -2238,7 +2357,10 @@ pub async fn bundle_add( for (rule_name, exe) in &planned.present { let id = bundle_rule_id(bundle.name, rule_name); - if let Some(rule) = existing.iter().find(|rule| rule.id == id) { + if let Some(rule) = existing + .iter() + .find(|rule| rule.id == id || (rule.name == *rule_name && legacy.contains(&rule.id))) + { let stored = rule.scope.as_ref().map_or("", |s| s.exe_path.as_str()); if !format.is_json() && std::path::Path::new(stored) != exe.as_path() { println!( @@ -2251,13 +2373,6 @@ pub async fn bundle_add( skipped_present += 1; continue; } - if existing - .iter() - .any(|rule| rule.name == *rule_name && legacy.contains(&rule.id)) - { - skipped_present += 1; - continue; - } let spec = &by_name[*rule_name]; if !dry_run { let mut rule = proto_for(spec, &exe.to_string_lossy()); @@ -2310,22 +2425,12 @@ pub async fn bundle_add( Ok(()) } -/// Entries a later version dropped because their rule could never fire: -/// Epiphany fetches through the WebKit network process every WebKitGTK app -/// shares, and npm and pip connect as their interpreter. `bundle remove` -/// still removes what older versions installed under these names. -const RETIRED_BUNDLE_RULES: &[(&str, &str)] = &[ - ("web", "web-epiphany-https"), - ("web", "web-epiphany-http"), - ("dev", "dev-npm-https"), - ("dev", "dev-pip-https"), -]; - /// `cfc rules bundle remove ` /// /// Removes the deterministic IDs this bundle gives its rules, and the copies /// of its entries seeded before 0.7.0 that still grant exactly what the entry -/// would (see [`legacy_copies`]). Any other rule, same name or not, is kept. +/// grants now or granted then (see [`legacy_copies`]), dropped entries +/// included. Any other rule, same name or not, is kept. pub async fn bundle_remove( client: &mut Client, name: &str, @@ -2333,20 +2438,16 @@ pub async fn bundle_remove( format: OutputFormat, ) -> CliResult { let bundle = find_bundle(name)?; - let retired = RETIRED_BUNDLE_RULES - .iter() - .filter(|(b, _)| *b == bundle.name) - .map(|(_, name)| *name); + // Dropped entries included: older versions installed them too. let owned: std::collections::HashSet = bundle .rules .iter() - .map(|r| r.name) - .chain(retired) - .map(|name| bundle_rule_id(bundle.name, name)) + .chain(&pre_0_7_entries(&bundle)) + .map(|r| bundle_rule_id(bundle.name, r.name)) .collect(); let existing = client.list_rules().await?; - let legacy = legacy_copies(&bundle, &plan(&bundle).present, &existing); + let legacy = legacy_copies(&bundle, &existing); let mut removed = Vec::new(); let mut kept = Vec::new(); for r in existing @@ -2825,8 +2926,7 @@ mod bundle_tests { fn a_pre_0_7_copy_of_a_bundle_rule_counts_only_while_unchanged() { let bundle = find_bundle("inbound").unwrap(); let entry = &bundle.rules[0]; - let present = [(entry.name, PathBuf::from("/usr/bin/sshd"))]; - let wanted = proto_for(entry, "/usr/bin/sshd"); + let wanted = proto_for(entry, ""); let mut legacy = wanted.clone(); legacy.id = "11111111-1111-4111-8111-111111111111".into(); legacy.enabled = false; @@ -2839,7 +2939,7 @@ mod bundle_tests { let mut elsewhere = legacy.clone(); elsewhere.id = "44444444-4444-4444-8444-444444444444".into(); elsewhere.scope.as_mut().unwrap().exe_path = "/usr/local/bin/sshd".into(); - let found = legacy_copies(&bundle, &present, &[legacy, deny, wider, elsewhere]); + let found = legacy_copies(&bundle, &[legacy, deny, wider, elsewhere]); assert_eq!( found, std::collections::HashSet::from(["11111111-1111-4111-8111-111111111111".to_owned()]) @@ -2880,6 +2980,83 @@ mod bundle_tests { std::fs::remove_dir_all(dir).unwrap(); } + // 0.6.0 pinned the first candidate that existed, a launcher included, and + // installed entries later versions re-pathed or dropped. Those copies are + // still the bundle's own. + #[test] + fn a_pre_0_7_copy_pinned_to_the_old_path_is_the_bundles_own() { + let dir = std::env::temp_dir().join(format!("cfc-bundle-{}", uuid::Uuid::new_v4())); + std::fs::create_dir(&dir).unwrap(); + let script = dir.join("browser"); + let real = dir.join("browser-bin"); + std::fs::write(&script, "#!/bin/sh\nexec browser-bin \"$@\"\n").unwrap(); + std::fs::write(&real, b"\x7fELF").unwrap(); + let leak = |p: &std::path::Path| -> &'static str { + Box::leak(p.to_str().unwrap().to_owned().into_boxed_str()) + }; + let entry = BundleRule { + name: "test-https", + exe_candidates: Box::leak(vec![leak(&script), leak(&real)].into_boxed_slice()), + dst_port: Some(443), + protocol: Some(proto::Protocol::Tcp), + direction: None, + src_net: None, + }; + let bundle = Bundle { + name: "test", + summary: "", + rules: vec![entry], + }; + let seeded = |id: &str, exe: &std::path::Path| { + let mut rule = proto_for( + &entry, + &cfc_core::exe_path::resolve(exe) + .into_path() + .to_string_lossy(), + ); + rule.id = id.into(); + rule + }; + let old = seeded("11111111-1111-4111-8111-111111111111", &script); + let new = seeded("22222222-2222-4222-8222-222222222222", &real); + let other = seeded("33333333-3333-4333-8333-333333333333", &dir); + assert_eq!( + legacy_copies(&bundle, &[old, new, other]), + std::collections::HashSet::from([ + "11111111-1111-4111-8111-111111111111".to_owned(), + "22222222-2222-4222-8222-222222222222".to_owned(), + ]) + ); + std::fs::remove_dir_all(dir).unwrap(); + + // The shipped history: a re-pathed entry keeps its scope apart from + // the path, and the dropped ones are still known by name. + let mut dropped = Vec::new(); + for bundle in bundles() { + for past in pre_0_7_entries(&bundle) { + match bundle.rules.iter().find(|r| r.name == past.name) { + Some(now) => assert_eq!( + (now.dst_port, now.protocol, now.direction, now.src_net), + (past.dst_port, past.protocol, past.direction, past.src_net), + "{}", + past.name + ), + None => dropped.push(past.name), + } + } + } + dropped.sort_unstable(); + assert_eq!( + dropped, + [ + "dev-npm-https", + "dev-pip-https", + "web-epiphany-http", + "web-epiphany-https" + ] + ); + } + /// The invariant the whole feature rests on. /// /// A bundle that installed a bare "allow tcp/443" outbound would re-open From 25110f15c721f56fa2765c92f56b160c2e38e445 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:36:24 +0200 Subject: [PATCH 108/125] feat(clients): seal the official app and tray, and surface daemon denials The GUI and tray now close every inherited descriptor above stderr and mark themselves non-dumpable as the first statement of main, so the daemon can tell an installed, untampered copy from a program that exec'd it with a socket already open. PERMISSION_DENIED becomes ClientError::Denied carrying the daemon's reason, and connect_interactive allows 150 s for requests that wait on an administrator password. --- Cargo.lock | 1 + crates/cfc-cli/src/error.rs | 10 + crates/cfc-client/Cargo.toml | 2 + crates/cfc-client/src/lib.rs | 260 ++++++++++++++++++++- crates/cfc-daemon/tests/ipc_integration.rs | 1 + crates/cfc-tray/src/main.rs | 10 +- crates/cfc-tray/src/model.rs | 7 +- crates/cfc-ui/src/main.rs | 8 + 8 files changed, 291 insertions(+), 8 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index ff2da68..d4933b5 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -723,6 +723,7 @@ dependencies = [ "cfc-core", "cfc-proto", "hyper-util", + "libc", "thiserror 2.0.20", "tokio", "tokio-stream", diff --git a/crates/cfc-cli/src/error.rs b/crates/cfc-cli/src/error.rs index 6a31ee3..682230f 100644 --- a/crates/cfc-cli/src/error.rs +++ b/crates/cfc-cli/src/error.rs @@ -123,6 +123,16 @@ mod tests { assert_eq!(err.exit_code(), EXIT_UNREACHABLE); } + #[test] + fn a_denial_prints_the_daemons_reason_verbatim_with_exit_1() { + let reason = "read-only access: the caller is being traced. Firewall changes are \ + accepted only from the installed Colony Firewall app and tray, or \ + from root (sudo cfc ...)."; + let err: CliError = ClientError::from(tonic::Status::permission_denied(reason)).into(); + assert_eq!(err.exit_code(), EXIT_RUNTIME); + assert_eq!(err.to_string(), reason); + } + #[test] fn anyhow_context_chain_is_preserved_on_one_line() { use anyhow::Context; diff --git a/crates/cfc-client/Cargo.toml b/crates/cfc-client/Cargo.toml index 714e07a..8636260 100644 --- a/crates/cfc-client/Cargo.toml +++ b/crates/cfc-client/Cargo.toml @@ -19,3 +19,5 @@ tower = { workspace = true } hyper-util = { workspace = true } anyhow = { workspace = true } thiserror = { workspace = true } +# close_range(2) and prctl(2) for `seal_official_process`. +libc = { workspace = true } diff --git a/crates/cfc-client/src/lib.rs b/crates/cfc-client/src/lib.rs index 69acba8..2387af5 100644 --- a/crates/cfc-client/src/lib.rs +++ b/crates/cfc-client/src/lib.rs @@ -38,10 +38,18 @@ pub enum ClientError { /// problem far more often than anything else. #[error( "permission denied on {path} - add your user to the colony-firewall group \ - (sudo usermod -aG colony-firewall $USER) then log out and back in, or run as root" + (sudo usermod -aG colony-firewall $USER) then log out and back in, or run as root. \ + The group gives read access and lets the Colony Firewall app and tray connect; \ + firewall changes come from the app, the tray or sudo cfc" )] PermissionDenied { path: PathBuf }, + /// The daemon refused the request (PERMISSION_DENIED). The text is the + /// daemon's own reason, made display-safe, and is shown as it is: it + /// already says what to do (use the app or tray, or sudo). + #[error("{0}")] + Denied(String), + /// The socket inode is there but nothing is listening: a crashed or /// SIGKILLed daemon leaves exactly this behind. #[error( @@ -58,7 +66,7 @@ pub enum ClientError { }, #[error("rpc: {0}")] - Rpc(#[from] tonic::Status), + Rpc(tonic::Status), #[error("transport: {0}")] Transport(#[from] tonic::transport::Error), @@ -69,6 +77,16 @@ pub enum ClientError { StreamClosed, } +impl From for ClientError { + fn from(status: tonic::Status) -> Self { + if status.code() == tonic::Code::PermissionDenied { + ClientError::Denied(convert::display_safe(status.message())) + } else { + ClientError::Rpc(status) + } + } +} + impl ClientError { /// True when the daemon could not be reached at all, as opposed to the /// daemon answering with an error. Callers use this to distinguish @@ -81,7 +99,7 @@ impl ClientError { | ClientError::Connect { .. } | ClientError::Transport(_) => true, ClientError::Rpc(status) => status.code() == tonic::Code::Unavailable, - ClientError::StreamClosed => false, + ClientError::StreamClosed | ClientError::Denied(_) => false, } } } @@ -158,8 +176,29 @@ pub struct VerdictOutcome { pub persist_note: Option, } +/// Per-request deadline of [`Client::connect`]. +pub const REQUEST_TIMEOUT: Duration = Duration::from_secs(5); +/// Per-request deadline of [`Client::connect_interactive`]: longer than the +/// daemon's 120 s wait for a polkit password dialog, so the daemon's own +/// answer (authorized, dismissed, timed out) always arrives first. +pub const INTERACTIVE_TIMEOUT: Duration = Duration::from_secs(150); + impl Client { + /// Connects with the 5 s per-request deadline every ordinary call uses. pub async fn connect(socket_path: impl AsRef) -> Result { + Self::connect_with_timeout(socket_path, REQUEST_TIMEOUT).await + } + + /// Connects for a request that may wait on an administrator password + /// (pause, resume): the daemon asks polkit and polkit asks the user. + pub async fn connect_interactive(socket_path: impl AsRef) -> Result { + Self::connect_with_timeout(socket_path, INTERACTIVE_TIMEOUT).await + } + + async fn connect_with_timeout( + socket_path: impl AsRef, + timeout: Duration, + ) -> Result { let path = socket_path.as_ref().to_path_buf(); let connect_path = path.clone(); @@ -177,7 +216,7 @@ impl Client { path: path.clone(), source: e.into(), })? - .timeout(Duration::from_secs(5)); + .timeout(timeout); let channel = endpoint .connect_with_connector(service_fn(move |_: Uri| { @@ -324,6 +363,61 @@ impl Client { } } +// --------------------------------------------------------------------------- +// Official client prologue +// --------------------------------------------------------------------------- + +/// Makes this process recognisable to the daemon as the installed Colony +/// Firewall app or tray. The first statement of their `main`, before any +/// thread, runtime, D-Bus or display connection exists; never called by the +/// CLI. +/// +/// 1. Closes every inherited descriptor above stderr, so no control-socket +/// connection opened by whoever exec'd this binary survives into it. The +/// daemon accepts a connection only when the official process itself holds +/// it, which closes the "connect, write a request, then exec the app" +/// route. +/// 2. Marks the process non-dumpable. The kernel then gives `/proc/` to +/// root, which is the marker the daemon checks, and same-user `ptrace`, +/// `/proc//mem` and `pidfd_getfd` are refused for the rest of the +/// process's life. +/// +/// A failure is returned for the caller to log once logging is up. The app +/// keeps working, read-only: the daemon refuses its changes with a reason. +pub fn seal_official_process() -> std::io::Result<()> { + close_inherited_fds()?; + // SAFETY: prctl(PR_SET_DUMPABLE) takes plain integers and touches no + // memory of ours. + if unsafe { libc::prctl(libc::PR_SET_DUMPABLE, 0, 0, 0, 0) } != 0 { + return Err(std::io::Error::last_os_error()); + } + Ok(()) +} + +fn close_inherited_fds() -> std::io::Result<()> { + // SAFETY: close_range(2) only closes descriptors; nothing in this + // single-threaded prologue owns one above stderr yet. + if unsafe { libc::syscall(libc::SYS_close_range, 3u32, u32::MAX, 0u32) } == 0 { + return Ok(()); + } + let error = std::io::Error::last_os_error(); + if error.raw_os_error() != Some(libc::ENOSYS) { + return Err(error); + } + // Kernels before 5.9: list first, then close, so the directory handle is + // not closed under the iteration. + let open: Vec = std::fs::read_dir("/proc/self/fd")? + .filter_map(|entry| entry.ok()?.file_name().to_str()?.parse().ok()) + .filter(|fd| *fd >= 3) + .collect(); + for fd in open { + // SAFETY: as above; EBADF for the listing's own, now closed, handle + // is harmless. + unsafe { libc::close(fd) }; + } + Ok(()) +} + // --------------------------------------------------------------------------- // Resilient (self-reconnecting) subscriptions // --------------------------------------------------------------------------- @@ -436,7 +530,7 @@ async fn pump_once( } Err(status) => { return PumpOutcome::Lost { - err: ClientError::Rpc(status), + err: status.into(), connected: true, } } @@ -574,6 +668,162 @@ mod tests { assert!(!ClientError::StreamClosed.is_unreachable()); } + #[test] + fn permission_denied_status_becomes_denied_with_the_daemon_message() { + let err: ClientError = + tonic::Status::permission_denied("read-only access: use sudo\u{1b}[2J").into(); + match &err { + ClientError::Denied(message) => { + assert!(message.starts_with("read-only access: use sudo")); + assert!(!message.contains('\u{1b}'), "display-safe: {message}"); + } + other => panic!("expected Denied, got {other:?}"), + } + assert_eq!(err.to_string(), "read-only access: use sudo\\u{1b}[2J"); + assert!(!err.is_unreachable()); + let err: ClientError = tonic::Status::internal("boom").into(); + assert!(matches!(err, ClientError::Rpc(_))); + } + + #[test] + fn seal_official_process_closes_inherited_fds_and_sets_non_dumpable() { + // In a forked child: the prologue closes every descriptor of the + // process it runs in, which would break the test harness. + // SAFETY: the child only makes raw syscalls (close_range succeeds on + // every kernel CI runs, so the allocating fallback is not reached) + // and leaves with _exit. + let child = unsafe { libc::fork() }; + assert!(child >= 0, "fork failed"); + if child == 0 { + let code = unsafe { + if libc::dup2(1, 57) != 57 { + libc::_exit(10); + } + if seal_official_process().is_err() { + libc::_exit(11); + } + if libc::fcntl(57, libc::F_GETFD) != -1 { + libc::_exit(12); + } + if libc::fcntl(2, libc::F_GETFD) == -1 { + libc::_exit(13); + } + if libc::prctl(libc::PR_GET_DUMPABLE, 0, 0, 0, 0) != 0 { + libc::_exit(14); + } + 0 + }; + unsafe { libc::_exit(code) }; + } + let mut status = 0; + // SAFETY: waiting on our own child with a valid out-pointer. + assert_eq!(unsafe { libc::waitpid(child, &mut status, 0) }, child); + assert!(libc::WIFEXITED(status), "child died: {status}"); + assert_eq!(libc::WEXITSTATUS(status), 0, "child check failed"); + } + + #[tokio::test] + async fn connect_interactive_outlives_five_seconds() { + use cfc_proto::v1::firewall_server::{Firewall, FirewallServer}; + use tonic::{Request, Response, Status}; + + // A daemon that takes 6 s to answer SetPaused, as one waiting on a + // polkit dialog does. Only SetPaused is reached. + struct Slow; + #[tonic::async_trait] + impl Firewall for Slow { + type StreamPromptsStream = tokio_stream::Empty>; + type StreamConnectionsStream = + tokio_stream::Empty>; + async fn set_paused( + &self, + req: Request, + ) -> Result, Status> { + tokio::time::sleep(Duration::from_secs(6)).await; + Ok(Response::new(proto::SetPausedResponse { + paused: req.into_inner().paused, + resume_at_unix_ms: 0, + })) + } + async fn stream_prompts( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn submit_verdict( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn list_rules( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn upsert_rule( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn apply_rules( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn delete_rule( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn stream_connections( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn get_status( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + async fn list_events( + &self, + _: Request, + ) -> Result, Status> { + unreachable!() + } + } + + let dir = std::env::temp_dir().join(format!("cfc-client-{}", std::process::id())); + std::fs::create_dir_all(&dir).unwrap(); + let socket = dir.join("slow.sock"); + let _ = std::fs::remove_file(&socket); + let listener = tokio::net::UnixListener::bind(&socket).unwrap(); + tokio::spawn( + tonic::transport::Server::builder() + .add_service(FirewallServer::new(Slow)) + .serve_with_incoming(tokio_stream::wrappers::UnixListenerStream::new(listener)), + ); + + let mut quick = Client::connect(&socket).await.unwrap(); + let err = quick.set_paused(true, 0).await.unwrap_err(); + assert!( + matches!(&err, ClientError::Rpc(status) if status.code() == tonic::Code::Cancelled + || status.code() == tonic::Code::DeadlineExceeded), + "the 5 s client gives up first: {err:?}" + ); + let mut patient = Client::connect_interactive(&socket).await.unwrap(); + assert!(patient.set_paused(true, 0).await.unwrap().paused); + let _ = std::fs::remove_dir_all(&dir); + } + #[test] fn backoff_doubles_and_saturates() { let mut d = RECONNECT_INITIAL; diff --git a/crates/cfc-daemon/tests/ipc_integration.rs b/crates/cfc-daemon/tests/ipc_integration.rs index ca171a8..8f25f17 100644 --- a/crates/cfc-daemon/tests/ipc_integration.rs +++ b/crates/cfc-daemon/tests/ipc_integration.rs @@ -362,6 +362,7 @@ async fn next_message(stream: &mut tonic::Streaming) -> T { fn status_of(err: ClientError) -> tonic::Status { match err { ClientError::Rpc(status) => status, + ClientError::Denied(message) => tonic::Status::permission_denied(message), other => panic!("expected an RPC status, got: {other}"), } } diff --git a/crates/cfc-tray/src/main.rs b/crates/cfc-tray/src/main.rs index 575df04..67895bc 100644 --- a/crates/cfc-tray/src/main.rs +++ b/crates/cfc-tray/src/main.rs @@ -938,21 +938,27 @@ fn warm_notification_spec_version() { } fn main() -> anyhow::Result<()> { + // First, before D-Bus or the runtime exist: the daemon accepts answers + // and pause requests only from a sealed, installed copy of this tray. + let sealed = cfc_client::seal_official_process(); warm_notification_spec_version(); tokio::runtime::Builder::new_multi_thread() .enable_all() .build() .context("building the tokio runtime")? - .block_on(run()) + .block_on(run(sealed)) } -async fn run() -> anyhow::Result<()> { +async fn run(sealed: std::io::Result<()>) -> anyhow::Result<()> { tracing_subscriber::fmt() .with_env_filter( tracing_subscriber::EnvFilter::try_from_default_env() .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info")), ) .init(); + if let Err(error) = sealed { + warn!("could not seal the process ({error}); the daemon will treat this tray as read-only"); + } let socket = socket_path_from_env(); info!(socket = %socket.display(), "starting colony-firewall-tray"); diff --git a/crates/cfc-tray/src/model.rs b/crates/cfc-tray/src/model.rs index 965d2d0..6ccaed7 100644 --- a/crates/cfc-tray/src/model.rs +++ b/crates/cfc-tray/src/model.rs @@ -97,6 +97,7 @@ pub fn unreachable_hint(err: &ClientError) -> String { ClientError::Connect { .. } | ClientError::Transport(_) => { "connection failed — is colony-firewalld healthy?".into() } + ClientError::Denied(_) => "read-only - restart the tray, or use sudo cfc".into(), ClientError::Rpc(_) | ClientError::StreamClosed => "daemon answered with an error".into(), } } @@ -636,7 +637,11 @@ mod tests { #[test] fn hints_are_short_and_actionable() { let p = PathBuf::from("/run/colony-firewall/cfc.sock"); - let cases: [(ClientError, &str); 4] = [ + let cases: [(ClientError, &str); 5] = [ + ( + ClientError::Denied("read-only access: ...".into()), + "sudo cfc", + ), ( ClientError::SocketMissing { path: p.clone() }, "systemctl status colony-firewalld", diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 0f113f9..fe2b4a7 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -48,12 +48,20 @@ const DEADLINE_TICK_MS: u64 = 400; const PROMPT_ARM_MS: i64 = 1_000; fn main() -> iced::Result { + // First, before any thread, display or daemon connection exists: the + // daemon accepts changes only from a sealed, installed copy of this app. + let sealed = cfc_client::seal_official_process(); tracing_subscriber::fmt() .with_env_filter( tracing_subscriber::EnvFilter::try_from_default_env() .unwrap_or_else(|_| tracing_subscriber::EnvFilter::new("info,cfc_ui=info")), ) .init(); + if let Err(error) = sealed { + tracing::warn!( + "could not seal the process ({error}); the daemon will treat this app as read-only" + ); + } iced::application(App::new, App::update, App::view) .title(App::title) From 86c0469327c8c7be44252938c9b65fb8da7993d6 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:43:21 +0200 Subject: [PATCH 109/125] feat(daemon)!: accept changes only from root and the installed app and tray Group membership no longer grants control. A non-root peer may answer prompts, write or delete rules, pause or import only when its process runs one of the root-sealed [ipc] official_clients binaries (by device and inode), ran the sealing prologue, is not traced, holds this connection itself (found through UNIX_DIAG) and mapped no executable file from outside sealed directories, all between two matching start-time reads. Every other non-root peer is read-only: its writes are refused with the reason, and its prompt subscription neither counts as a UI nor enters a prompt's audience. A peer with the daemon's own uid keeps full control. BREAKING CHANGE: a non-root cfc can no longer change the firewall; use sudo or the app. The 0.7 app and tray are read-only until restarted. --- crates/cfc-client/src/lib.rs | 5 +- crates/cfc-daemon/examples/prompt_demo.rs | 10 +- crates/cfc-daemon/src/config.rs | 70 ++- crates/cfc-daemon/src/ipc.rs | 521 ++++++++++++++++++--- crates/cfc-daemon/src/lib.rs | 1 + crates/cfc-daemon/src/official.rs | 517 ++++++++++++++++++++ crates/cfc-daemon/src/prompts.rs | 71 ++- crates/cfc-daemon/src/sock_diag.rs | 171 ++++++- crates/cfc-daemon/tests/ipc_integration.rs | 52 +- systemd/daemon.toml.sample | 45 +- 10 files changed, 1300 insertions(+), 163 deletions(-) create mode 100644 crates/cfc-daemon/src/official.rs diff --git a/crates/cfc-client/src/lib.rs b/crates/cfc-client/src/lib.rs index 2387af5..1f13c5c 100644 --- a/crates/cfc-client/src/lib.rs +++ b/crates/cfc-client/src/lib.rs @@ -377,8 +377,9 @@ impl Client { /// daemon accepts a connection only when the official process itself holds /// it, which closes the "connect, write a request, then exec the app" /// route. -/// 2. Marks the process non-dumpable. The kernel then gives `/proc/` to -/// root, which is the marker the daemon checks, and same-user `ptrace`, +/// 2. Marks the process non-dumpable. The kernel then gives the files under +/// `/proc/` to root, which is the marker the daemon checks, and +/// same-user `ptrace`, /// `/proc//mem` and `pidfd_getfd` are refused for the rest of the /// process's life. /// diff --git a/crates/cfc-daemon/examples/prompt_demo.rs b/crates/cfc-daemon/examples/prompt_demo.rs index fd6f033..e005199 100644 --- a/crates/cfc-daemon/examples/prompt_demo.rs +++ b/crates/cfc-daemon/examples/prompt_demo.rs @@ -72,12 +72,10 @@ async fn main() -> anyhow::Result<()> { let (_ipc, prompt_tx) = ipc::spawn( IpcOptions { socket_path: socket.clone(), - ipc: IpcConfig { - group: "colony-firewall".into(), - // Demo socket in the user's own temporary directory: let the - // invoking user talk to it. - require_group: false, - }, + // The demo daemon runs as the invoking user, and a peer with the + // daemon's own uid has full control: the user's app, tray or CLI + // can drive it without any group or official-client setup. + ipc: IpcConfig::default(), pause_default_secs: 120, dry_run: true, }, diff --git a/crates/cfc-daemon/src/config.rs b/crates/cfc-daemon/src/config.rs index a5df462..045154c 100644 --- a/crates/cfc-daemon/src/config.rs +++ b/crates/cfc-daemon/src/config.rs @@ -272,12 +272,16 @@ pub struct IpcConfig { /// daemon chowns the socket to `root:` and chmods it 0660, so /// group membership *is* the access check. pub group: String, - /// Require the socket to be group-gated before a non-root peer may - /// call a mutating RPC. When the group cannot be resolved the socket - /// stays root-only and non-root mutations are refused. Setting this to - /// false lets any peer that manages to connect mutate rules — only do - /// that if you gate the socket some other way (e.g. filesystem ACLs). + /// Require proved membership of `group` before an official client (the + /// installed app or tray) may change anything. Setting this to false + /// waives the group check for official clients only; every other + /// non-root peer stays read-only either way. pub require_group: bool, + /// The installed Colony Firewall app and tray: the only non-root + /// programs that may answer prompts, edit rules or ask to pause. Each + /// must be an absolute path to a root-owned file nobody else can write, + /// in root-owned directories nobody else can write. Bound at startup. + pub official_clients: Vec, } impl Default for IpcConfig { @@ -285,10 +289,39 @@ impl Default for IpcConfig { Self { group: "colony-firewall".to_string(), require_group: true, + official_clients: crate::official::DEFAULT_CLIENTS + .iter() + .map(std::path::PathBuf::from) + .collect(), } } } +impl IpcConfig { + /// One line per `official_clients` entry that can never match, for the + /// startup log: the app or tray it names will be read-only. + pub fn official_client_warnings(&self) -> Vec { + self.official_clients + .iter() + .filter_map(|path| { + let why = if !path.is_absolute() { + "is not an absolute path" + } else if !path.exists() { + "does not exist" + } else if crate::official::sealed_identity(path).is_none() { + "is not a root-owned file in root-owned directories that only root can write" + } else { + return None; + }; + Some(format!( + "[ipc] official_clients entry {} {why}; that program will be read-only", + path.display() + )) + }) + .collect() + } +} + /// Binary package provenance (see `crate::provenance`). #[derive(Debug, Clone, Copy, Serialize, Deserialize)] #[serde(default)] @@ -836,6 +869,12 @@ enabled = " Auto ""# assert_eq!(cfg.ipc.group, "wheel"); assert!(!cfg.ipc.require_group); + assert_eq!( + cfg.ipc.official_clients, + IpcConfig::default().official_clients, + "official_clients keeps its packaged default" + ); + // A partial [ipc] section keeps per-field defaults. let cfg = Config::from_toml_str("[ipc]\ngroup = \"wheel\"\n").unwrap(); assert_eq!(cfg.ipc.group, "wheel"); @@ -848,6 +887,27 @@ enabled = " Auto ""# assert!(!cfg.nfqueue.fail_open); } + #[test] + fn official_clients_default_to_the_installed_app_and_tray() { + assert_eq!( + IpcConfig::default().official_clients, + [ + std::path::PathBuf::from("/usr/bin/colony-firewall"), + std::path::PathBuf::from("/usr/bin/colony-firewall-tray") + ] + ); + let cfg = Config::from_toml_str( + "[ipc]\nofficial_clients = [\"bin/gui\", \"/nonexistent/cfc-gui\", \"/tmp\"]\n", + ) + .unwrap(); + let warnings = cfg.ipc.official_client_warnings(); + assert_eq!(warnings.len(), 3, "{warnings:?}"); + assert!(warnings[0].contains("bin/gui is not an absolute path")); + assert!(warnings[1].contains("does not exist")); + assert!(warnings[2].contains("root-owned")); + assert!(warnings.iter().all(|w| w.ends_with("will be read-only"))); + } + #[test] fn sample_config_file_parses() { // Permanently parse-checks the shipped sample. diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 3c7baba..97dd4bc 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -8,23 +8,36 @@ //! //! 1. **The socket file.** After bind, the daemon chowns the socket to //! `root:<[ipc] group>` and chmods it `0660`. The kernel therefore -//! refuses `connect(2)` to anyone outside that group. Membership *is* -//! the credential; there is no in-band authentication. If the group +//! refuses `connect(2)` to anyone outside that group. If the group //! cannot be resolved (package installed without the sysusers fragment) //! the daemon logs a prominent warning, leaves the socket `0600` //! (root-only) and keeps running, so a root CLI still works. //! -//! 2. **Per-RPC peer credentials.** Mutations require uid 0 or actual -//! membership of the configured group. The kernel peer gid proves primary -//! membership. Supplementary membership requires `/proc//status` -//! with the same effective uid and process starttime captured at accept. -//! Missing evidence is refused. `require_group = false` explicitly opts -//! out for deployments authorizing their control socket another way. +//! 2. **Per-RPC peer identity.** Reading (status, rules, the live feed, +//! the event log, receiving prompts) is open to every peer that could +//! connect. Changing the firewall (answering a prompt, writing or deleting +//! rules, pause and resume, importing rules) is open to: +//! - **root**, and the daemon's own uid (root in production; a process +//! with the daemon's uid could ptrace it anyway); +//! - **the official app and tray**: a proved member of the group (the +//! kernel peer gid, or `/proc//status` read between two matching +//! start-time reads) whose process passes [`crate::official::check`]: +//! it runs one of the installed, root-sealed `[ipc] official_clients` +//! binaries, ran their sealing prologue, holds this very connection, is +//! not traced and mapped no executable file from outside sealed +//! directories. `require_group = false` waives the group proof for +//! official clients only. //! -//! Consequence worth stating plainly: **every member of the configured -//! group is fully trusted.** Group membership grants the ability to allow -//! or deny any traffic on the host. It is not a multi-user privilege -//! boundary; put only administrators of this machine in it. +//! Every other peer is read-only: its change RPCs get PERMISSION_DENIED +//! with the reason, its prompt subscription does not count as a UI (so +//! `no_ui_action` still applies), and it never enters a prompt's audience. +//! +//! Consequence worth stating plainly: group membership alone no longer +//! changes anything. It lets the desktop session read the daemon's state and +//! lets the installed app and tray connect; changes come from those two +//! programs or from `sudo cfc`. Code running *inside* the official app (a +//! preloaded payload that moved itself to anonymous memory, synthetic X11 +//! input) is still trusted; see docs/HARDENING.md. //! //! # Prompt ownership //! @@ -37,8 +50,9 @@ //! everyone. So another user's UI never even learns the prompt id. //! 2. **Answers are checked against who was told.** A stream records //! `prompt_id -> peer uid` as it hands an event to its client -//! ([`PromptAudience`]); `SubmitVerdict` requires the caller's uid to -//! appear in that prompt's audience. Root may always answer. +//! ([`PromptAudience`]) - only for a stream that may answer; `SubmitVerdict` +//! requires the caller's uid to appear in that prompt's audience. Root may +//! always answer. //! //! Step 2 alone was bookkeeping without teeth - every subscriber received //! every prompt, so every subscriber was in every audience. Step 1 is what @@ -223,17 +237,47 @@ pub struct PeerId { pub gid: u32, pub pid: Option, pub starttime: Option, + /// Inode of the daemon's end of this connection, which names the + /// client's end through sock_diag. `None` means "never official". + pub sock_ino: Option, } /// Privilege an RPC requires. #[derive(Debug, Clone, Copy, PartialEq, Eq)] enum Access { - /// Observing daemon state. + /// Observing daemon state: any peer that could connect. ReadOnly, - /// Changing the firewall's behaviour. - Mutate, + /// Changing the firewall's behaviour: root, or the official app or tray. + Control, +} + +/// Where a peer stands for [`Access::Control`], before its image is checked. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Gate { + /// Root, or the daemon's own uid (root in production; a process of the + /// daemon's uid could ptrace it anyway). + Privileged, + /// A group member (or anyone, with `require_group = false`): allowed + /// only when it is the installed app or tray. + NeedOfficial, + /// Not a proved member of the configured group. + DenyGroup, +} + +fn gate(peer_uid: u32, own_uid: u32, group_ok: bool) -> Gate { + if peer_uid == 0 || peer_uid == own_uid { + Gate::Privileged + } else if group_ok { + Gate::NeedOfficial + } else { + Gate::DenyGroup + } } +/// The official-client check: `official::check` in production, a stub in +/// unit tests. Not reachable from config or from any client. +type OfficialCheck = fn(&PeerId, &[PathBuf]) -> Result; + /// Outcome of securing the socket file, and the policy knobs that decide /// what it implies for callers. #[derive(Debug, Clone)] @@ -246,15 +290,6 @@ struct SocketAuth { require_group: bool, } -/// Pure policy over membership proved from the individual peer credentials. -fn authorize_uid(uid: u32, level: Access, group_member: bool, require_group: bool) -> bool { - match level { - // Layer 1 (socket mode) already decided who may connect at all. - Access::ReadOnly => true, - Access::Mutate => uid == 0 || !require_group || group_member, - } -} - /// Extracts kernel-reported peer credentials from a request. fn peer_of(req: &Request) -> Result { if let Some(peer) = req.extensions().get::() { @@ -275,6 +310,7 @@ fn peer_of(req: &Request) -> Result { .pid() .and_then(|pid| u32::try_from(pid).ok()) .and_then(crate::process_resolve::read_starttime), + sock_ino: None, }) } @@ -291,16 +327,25 @@ impl PeerStream { .and_then(|pid| u32::try_from(pid).ok()) .and_then(crate::process_resolve::read_starttime); Ok(Self { - stream, peer: PeerId { uid: credentials.uid(), gid: credentials.gid(), pid, starttime, + sock_ino: socket_inode(&stream), }, + stream, }) } } +fn socket_inode(stream: &tokio::net::UnixStream) -> Option { + use std::os::fd::AsRawFd; + // SAFETY: zeroed stat is valid out-param storage; the fd is open for + // the duration of the call. + let mut st: libc::stat = unsafe { std::mem::zeroed() }; + (unsafe { libc::fstat(stream.as_raw_fd(), &mut st) } == 0).then_some(st.st_ino) +} + impl Connected for PeerStream { type ConnectInfo = PeerId; fn connect_info(&self) -> PeerId { @@ -441,6 +486,11 @@ struct FirewallService { /// Live default policy; SIGHUP swaps it, so status reflects reloads. policy: SharedPolicy, auth: SocketAuth, + /// The daemon's effective uid; see [`Gate::Privileged`]. + own_uid: u32, + /// `[ipc] official_clients`, bound at startup. + official_clients: Arc<[PathBuf]>, + official: OfficialCheck, audience: Arc, /// Wall-clock deadline of the current pause, 0 when not paused. Held /// here rather than in `Stats` so the pause timer and `GetStatus` agree. @@ -450,27 +500,69 @@ struct FirewallService { } impl FirewallService { - /// Resolves the caller and checks it may perform `level`. - fn authorize(&self, req: &Request, level: Access) -> Result { + /// Resolves the caller of `rpc` and checks it may perform `level`. + async fn authorize( + &self, + req: &Request, + rpc: &'static str, + level: Access, + ) -> Result { let peer = peer_of(req)?; - if authorize_uid( - peer.uid, - level, - peer_is_group_member(peer, self.auth.group_gid), - self.auth.require_group, - ) { + if level == Access::ReadOnly { return Ok(peer); } - warn!( - peer_uid = peer.uid, - peer_pid = ?peer.pid, - group = %self.auth.group, - "refusing mutating RPC: caller is not a member of the configured group" - ); - Err(Status::permission_denied(format!( - "mutating RPCs require uid 0 or membership of group '{}'", - self.auth.group - ))) + match self.standing(peer).await { + Ok(official) => { + info!( + rpc, + peer_uid = peer.uid, + peer_pid = ?peer.pid, + auth = if official.is_some() { "official" } else { "privileged" }, + official_exe = ?official, + "authorized" + ); + Ok(peer) + } + Err(status) => { + warn!( + rpc, + peer_uid = peer.uid, + peer_pid = ?peer.pid, + reason = status.message(), + outcome = "permission_denied", + "refusing a firewall change" + ); + Err(status) + } + } + } + + /// May `peer` change the firewall? `Ok(None)` for a privileged peer, + /// `Ok(Some(exe))` for the official app or tray, `Err` (with the reason + /// the client shows) for everyone else, who is read-only. + async fn standing(&self, peer: PeerId) -> Result, Status> { + let group_ok = !self.auth.require_group || peer_is_group_member(peer, self.auth.group_gid); + match gate(peer.uid, self.own_uid, group_ok) { + Gate::Privileged => Ok(None), + Gate::DenyGroup => Err(Status::permission_denied(format!( + "mutating RPCs require uid 0 or membership of group '{}'", + self.auth.group + ))), + Gate::NeedOfficial => { + let (check, list) = (self.official, self.official_clients.clone()); + tokio::task::spawn_blocking(move || check(&peer, &list)) + .await + .unwrap_or_else(|error| Err(format!("the identity check failed ({error})"))) + .map(Some) + .map_err(|reason| { + Status::permission_denied(format!( + "read-only access: {reason}. Firewall changes are accepted only \ + from the installed Colony Firewall app and tray, or from root \ + (sudo cfc ...)." + )) + }) + } + } } async fn upsert_rule_checked( @@ -608,9 +700,25 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, Access::ReadOnly)?; + let peer = self + .authorize(&req, "StreamPrompts", Access::ReadOnly) + .await?; + // Every peer is shown the prompts addressed to it; only one that may + // answer them is counted as a UI and enters their audience. + let answering = match self.standing(peer).await { + Ok(_) => true, + Err(status) => { + tracing::debug!( + peer_uid = peer.uid, + peer_pid = ?peer.pid, + reason = status.message(), + "read-only prompt subscriber" + ); + false + } + }; let (tx, rx) = mpsc::channel(64); - let mut sub = self.router.subscribe(peer.uid); + let mut sub = self.router.subscribe(peer.uid, answering); let audience = self.audience.clone(); let uid = peer.uid; tokio::spawn(async move { @@ -635,7 +743,9 @@ impl Firewall for FirewallService { // subscriber is about to learn the prompt id, so it // must be entitled to answer it by the time it can. if let Ok(id) = event.prompt_id.parse::() { - audience.record(id, uid); + if answering { + audience.record(id, uid); + } } if tx.send(Ok(event)).await.is_err() { break; @@ -657,7 +767,9 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, Access::Mutate)?; + let peer = self + .authorize(&req, "SubmitVerdict", Access::Control) + .await?; let req = req.into_inner(); // Ownership: only a peer this prompt was actually delivered to may @@ -826,7 +938,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - self.authorize(&req, Access::ReadOnly)?; + self.authorize(&req, "ListRules", Access::ReadOnly).await?; let snapshot = self.engine.snapshot(); let rules = snapshot.rules.iter().map(convert::rule_to_pb).collect(); Ok(Response::new(ListRulesResponse { rules })) @@ -836,7 +948,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, Access::Mutate)?; + let peer = self.authorize(&req, "UpsertRule", Access::Control).await?; self.upsert_rule_checked(peer, req.into_inner()) .await .map(Response::new) @@ -847,7 +959,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, Access::Mutate)?; + let peer = self.authorize(&req, "ApplyRules", Access::Control).await?; self.apply_rules_checked(peer, req.into_inner()) .await .map(Response::new) @@ -858,7 +970,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, Access::Mutate)?; + let peer = self.authorize(&req, "DeleteRule", Access::Control).await?; let id_str = req.into_inner().id; let id = uuid::Uuid::parse_str(&id_str) .map_err(|e| Status::invalid_argument(format!("bad uuid: {e}")))?; @@ -888,7 +1000,8 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - self.authorize(&req, Access::ReadOnly)?; + self.authorize(&req, "StreamConnections", Access::ReadOnly) + .await?; let (tx, rx) = mpsc::channel(256); let mut sub = self.observed_tx.subscribe(); tokio::spawn(async move { @@ -924,7 +1037,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - self.authorize(&req, Access::ReadOnly)?; + self.authorize(&req, "GetStatus", Access::ReadOnly).await?; let rules_count = self.engine.rule_count() as u64; let policy = self.policy(); let paused = self.stats.is_paused(); @@ -964,7 +1077,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, Access::Mutate)?; + let peer = self.authorize(&req, "SetPaused", Access::Control).await?; let msg = req.into_inner(); if !msg.paused { @@ -1036,7 +1149,7 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - self.authorize(&req, Access::ReadOnly)?; + self.authorize(&req, "ListEvents", Access::ReadOnly).await?; let (limit, offset, filter) = event_query_from_pb(&req.into_inner()).map_err(Status::invalid_argument)?; let rows = self @@ -1428,6 +1541,9 @@ pub async fn spawn( .with_context(|| format!("binding {}", socket_path.display()))?; // Tighten ownership/mode before the first client can connect. let auth = secure_socket(&socket_path, &opts.ipc); + for warning in opts.ipc.official_client_warnings() { + warn!("{warning}"); + } let incoming = tokio_stream::wrappers::UnixListenerStream::new(uds) .map(|stream| stream.and_then(PeerStream::new)); @@ -1439,6 +1555,9 @@ pub async fn spawn( stats, policy, auth, + own_uid: nix::unistd::geteuid().as_raw(), + official_clients: opts.ipc.official_clients.clone().into(), + official: crate::official::check, audience: Arc::new(PromptAudience::default()), resume_at_ms: Arc::new(AtomicI64::new(0)), pause_default_secs: opts.pause_default_secs, @@ -1506,30 +1625,284 @@ mod tests { // -- authorization ------------------------------------------------------ #[test] - fn read_only_rpcs_are_open_to_any_connected_peer() { - for gated in [true, false] { - for require in [true, false] { - assert!(authorize_uid(1000, Access::ReadOnly, gated, require)); - assert!(authorize_uid(0, Access::ReadOnly, gated, require)); + fn gate_table() { + // (peer uid, daemon uid, group proved) -> standing + let cases = [ + (0, 0, false, Gate::Privileged), + (0, 0, true, Gate::Privileged), + (1000, 1000, false, Gate::Privileged), + (0, 1000, false, Gate::Privileged), + (1000, 0, true, Gate::NeedOfficial), + (1000, 0, false, Gate::DenyGroup), + ]; + for (peer, own, group_ok, want) in cases { + assert_eq!(gate(peer, own, group_ok), want, "{peer} {own} {group_ok}"); + } + } + + // -- the service's authorization table --------------------------------- + + const OWN_UID: u32 = 999; + const GROUP_GID: u32 = 4242; + /// The pid the stub official check accepts. + const OFFICIAL_PID: i32 = 77; + + fn stub_official(peer: &PeerId, _: &[PathBuf]) -> Result { + if peer.pid == Some(OFFICIAL_PID) { + Ok(PathBuf::from("/usr/bin/colony-firewall")) + } else { + Err("not the installed app".into()) + } + } + + fn service(require_group: bool) -> FirewallService { + let store = RuleStore::open_in_memory().unwrap(); + let policy: SharedPolicy = Arc::new(std::sync::RwLock::new(crate::config::DefaultPolicy { + no_ui_action: cfc_core::Action::Deny, + timeout_action: cfc_core::Action::Deny, + inbound_action: cfc_core::Action::Deny, + prompt_timeout_secs: 3600, + })); + let engine = Engine::new(store.snapshot().unwrap(), policy.clone()); + let stats = Stats::new(); + let (verdict_tx, verdicts) = std::sync::mpsc::channel(); + // Verdicts are not asserted here; keep the receiver alive. + std::mem::forget(verdicts); + FirewallService { + router: PromptRouter::new(policy.clone(), stats.clone(), verdict_tx), + engine, + store, + observed_tx: broadcast::channel(16).0, + stats, + policy, + auth: SocketAuth { + group: "cfc-test".into(), + group_gid: Some(GROUP_GID), + group_gated: true, + require_group, + }, + own_uid: OWN_UID, + official_clients: Arc::from(Vec::new()), + official: stub_official, + audience: Arc::new(PromptAudience::default()), + resume_at_ms: Arc::new(AtomicI64::new(0)), + pause_default_secs: 600, + dry_run: true, + } + } + + fn peer(uid: u32, gid: u32, pid: i32) -> PeerId { + PeerId { + uid, + gid, + pid: Some(pid), + starttime: None, + sock_ino: None, + } + } + + fn request(message: T, peer: PeerId) -> Request { + let mut req = Request::new(message); + req.extensions_mut().insert(peer); + req + } + + fn rule_pb() -> RuleInfo { + let mut rule = cfc_core::Rule::new( + "smtp", + cfc_core::Action::Deny, + cfc_core::RuleScope { + dst_port: Some(25), + ..cfc_core::RuleScope::any() + }, + ); + rule.name = "smtp".into(); + convert::rule_to_pb(&rule) + } + + /// Which change RPCs `peer` got through, in the order UpsertRule, + /// DeleteRule, SetPaused, ApplyRules. A refusal must be + /// PERMISSION_DENIED; any other error fails the test. + async fn changes(svc: &FirewallService, peer: PeerId) -> [bool; 4] { + fn passed(result: Result) -> bool { + match result { + Ok(_) => true, + Err(status) => { + assert_eq!(status.code(), tonic::Code::PermissionDenied, "{status:?}"); + false + } } } + let upsert = svc + .upsert_rule(request( + UpsertRuleRequest { + rule: Some(rule_pb()), + }, + peer, + )) + .await; + let delete = svc + .delete_rule(request( + DeleteRuleRequest { + id: uuid::Uuid::new_v4().to_string(), + }, + peer, + )) + .await; + let pause = svc + .set_paused(request( + SetPausedRequest { + paused: false, + duration_secs: 0, + }, + peer, + )) + .await; + let apply = svc + .apply_rules(request( + ApplyRulesRequest { + rules: vec![rule_pb()], + replace: false, + }, + peer, + )) + .await; + [passed(upsert), passed(delete), passed(pause), passed(apply)] } - #[test] - fn root_may_always_mutate() { - for gated in [true, false] { - assert!(authorize_uid(0, Access::Mutate, gated, true)); + async fn reads(svc: &FirewallService, peer: PeerId) { + svc.list_rules(request(ListRulesRequest {}, peer)) + .await + .unwrap(); + svc.get_status(request(StatusRequest {}, peer)) + .await + .unwrap(); + svc.list_events(request(ListEventsRequest::default(), peer)) + .await + .unwrap(); + svc.stream_connections(request(SubscribeRequest::default(), peer)) + .await + .unwrap(); + svc.stream_prompts(request(SubscribeRequest::default(), peer)) + .await + .unwrap(); + } + + #[tokio::test] + async fn authorization_table() { + let root = peer(0, 0, 1); + let own = peer(OWN_UID, OWN_UID, 2); + let official_member = peer(1000, GROUP_GID, OFFICIAL_PID); + let official_outsider = peer(1001, 1001, OFFICIAL_PID); + let read_only_member = peer(1000, GROUP_GID, 5); + let all = [true; 4]; + let none = [false; 4]; + + let svc = service(true); + for (who, want) in [ + (root, all), + (own, all), + (official_member, all), + (official_outsider, none), + (read_only_member, none), + ] { + reads(&svc, who).await; + assert_eq!(changes(&svc, who).await, want, "{who:?}"); } + + // require_group = false waives the group check, never the image check. + let svc = service(false); + assert_eq!(changes(&svc, official_outsider).await, all); + assert_eq!(changes(&svc, read_only_member).await, none); + assert_eq!(changes(&svc, peer(1002, 1002, 6)).await, none); } - #[test] - fn non_root_mutation_requires_proved_peer_group_membership() { - // Only proved peer membership authorizes a non-root mutation. - assert!(authorize_uid(1000, Access::Mutate, true, true)); - // A socket mode is not membership evidence for an individual peer. - assert!(!authorize_uid(1000, Access::Mutate, false, true)); - // Explicit opt-out: the admin gates the socket some other way. - assert!(authorize_uid(1000, Access::Mutate, false, false)); + #[tokio::test] + async fn a_read_only_refusal_says_why_and_what_to_use_instead() { + let svc = service(true); + let status = svc + .set_paused(request( + SetPausedRequest::default(), + peer(1000, GROUP_GID, 5), + )) + .await + .unwrap_err(); + assert!(status + .message() + .starts_with("read-only access: not the installed app.")); + assert!( + status.message().contains("sudo cfc"), + "{}", + status.message() + ); + assert!(!svc.stats.is_paused()); + } + + /// A pending prompt about uid 1000's process, with an answering UI so it + /// waits, and uid 1000 in its audience. + async fn pending_prompt(svc: &FirewallService) -> crate::prompts::PromptSubscription { + use std::net::{IpAddr, Ipv4Addr}; + let ui = svc.router.subscribe(1000, true); + let (tx, rx) = mpsc::channel(4); + tokio::spawn(crate::prompts::run_router_task(rx, svc.router.clone())); + let mut process = cfc_core::Process::unknown(4321); + process.uid = Some(1000); + tx.send(PromptRequest { + prompt_id: 5, + connection: cfc_core::Connection::new( + cfc_core::Protocol::Tcp, + cfc_core::Direction::Outbound, + IpAddr::V4(Ipv4Addr::LOCALHOST), + 40000, + IpAddr::V4(Ipv4Addr::new(1, 1, 1, 1)), + 443, + ), + process, + }) + .await + .unwrap(); + while svc.stats.prompts_pending() == 0 { + tokio::task::yield_now().await; + } + svc.audience.record(5, 1000); + ui + } + + fn answer(peer: PeerId) -> Request { + request( + VerdictRequest { + prompt_id: "5".into(), + action: cfc_proto::v1::Action::Allow as i32, + duration: cfc_proto::v1::Duration::Once as i32, + persist_scope: None, + }, + peer, + ) + } + + #[tokio::test] + async fn read_only_peer_cannot_answer_even_its_own_prompt() { + let svc = service(true); + let _ui = pending_prompt(&svc).await; + let status = svc + .submit_verdict(answer(peer(1000, GROUP_GID, 5))) + .await + .unwrap_err(); + assert_eq!(status.code(), tonic::Code::PermissionDenied); + assert_eq!(svc.stats.prompts_pending(), 1, "the prompt still waits"); + } + + #[tokio::test] + async fn official_peer_answers_without_password() { + let svc = service(true); + let _ui = pending_prompt(&svc).await; + let reply = svc + .submit_verdict(answer(peer(1000, GROUP_GID, OFFICIAL_PID))) + .await + .unwrap() + .into_inner(); + assert!(reply.accepted); + assert_eq!(svc.stats.prompts_pending(), 0); } #[test] diff --git a/crates/cfc-daemon/src/lib.rs b/crates/cfc-daemon/src/lib.rs index 86bd07a..eb7a002 100644 --- a/crates/cfc-daemon/src/lib.rs +++ b/crates/cfc-daemon/src/lib.rs @@ -18,6 +18,7 @@ pub mod dns; pub mod ebpf; pub mod ipc; pub mod nfqueue; +pub mod official; pub mod packet; pub mod process_resolve; pub mod prompts; diff --git a/crates/cfc-daemon/src/official.rs b/crates/cfc-daemon/src/official.rs new file mode 100644 index 0000000..9a67030 --- /dev/null +++ b/crates/cfc-daemon/src/official.rs @@ -0,0 +1,517 @@ +//! Is the peer on a control connection the installed Colony Firewall app or +//! tray? +//! +//! Only those two programs (and root) may change the firewall. The daemon +//! cannot ask a process who it is, so it checks what the kernel says about +//! the process holding the connection, in this order: +//! +//! 1. its start time matches the one captured at accept (the "before" read); +//! 2. it runs in the host mount and user namespaces, so no private mount or +//! user namespace can show it a different `/usr/bin`; +//! 3. it is not traced and its effective uid is the connection's uid; +//! 4. it is non-dumpable, which is the mark `seal_official_process` leaves: +//! the kernel hands the files under `/proc/` to root exactly then; +//! 5. its image is, by device and inode, one of the allowlisted binaries, +//! and that binary is root-owned, not group/other-writable, in root-owned +//! directories nobody else can write (re-checked on every call, so an +//! upgraded binary counts and the old, deleted image does not); +//! 6. it holds the client end of *this* connection itself, on a descriptor +//! above stderr (the prologue closed every inherited one); +//! 7. every executable file mapping comes from a sealed path, which refuses +//! an `LD_PRELOAD` or `LD_AUDIT` library loaded from the user's files; +//! 8. its start time still matches (the "after" read), so none of the above +//! was read from a process that replaced it under the same pid. +//! +//! Steps 4 and 6 close the exec-after-connect route: connect, write a whole +//! request, then exec the official binary with the socket inherited. That +//! process has not run the prologue yet, or has closed the inherited +//! connection by the time it is sealed. +//! +//! What it cannot see: code already running inside the official image that +//! moved itself into anonymous memory (anonymous executable mappings are +//! not judged, GPU drivers JIT into them), and synthetic input to the GUI +//! under X11. See docs/HARDENING.md. + +use crate::ipc::PeerId; +use std::collections::HashMap; +use std::os::unix::fs::MetadataExt as _; +use std::path::{Path, PathBuf}; + +/// Packaged default for `[ipc] official_clients`: the installed GUI and tray. +pub const DEFAULT_CLIENTS: [&str; 2] = + ["/usr/bin/colony-firewall", "/usr/bin/colony-firewall-tray"]; + +/// Checks `peer` against `allowlist`. Blocking (reads `/proc` and asks +/// sock_diag): call it from `spawn_blocking`. `Ok` carries the matched +/// allowlist entry; `Err` a reason fit to show the user. +pub fn check(peer: &PeerId, allowlist: &[PathBuf]) -> Result { + let pid = peer + .pid + .filter(|pid| *pid > 0) + .ok_or("the caller's process id is unknown")? as u32; + let (Some(start), Some(sock_ino)) = (peer.starttime, peer.sock_ino) else { + return Err("the caller's process could not be identified".into()); + }; + bracketed(pid, start, crate::process_resolve::read_starttime, || { + inspect(pid, peer.uid, sock_ino, allowlist) + }) +} + +/// Runs `body` between two start-time reads that must both match `expected`. +fn bracketed( + pid: u32, + expected: u64, + read: impl Fn(u32) -> Option, + body: impl FnOnce() -> Result, +) -> Result { + const GONE: &str = "the caller exited, or its process id now names another process"; + if read(pid) != Some(expected) { + return Err(GONE.into()); + } + let outcome = body()?; + if read(pid) != Some(expected) { + return Err(GONE.into()); + } + Ok(outcome) +} + +fn inspect(pid: u32, uid: u32, sock_ino: u64, allowlist: &[PathBuf]) -> Result { + let proc = PathBuf::from(format!("/proc/{pid}")); + for ns in ["mnt", "user"] { + let theirs = std::fs::read_link(proc.join("ns").join(ns)); + let host = std::fs::read_link(Path::new("/proc/1/ns").join(ns)); + match (theirs, host) { + (Ok(theirs), Ok(host)) if theirs == host => {} + _ => return Err(format!("the caller runs in a private {ns} namespace")), + } + } + let status = std::fs::read_to_string(proc.join("status")) + .map_err(|e| format!("the caller's status is unreadable ({e})"))?; + status_is_clean(&status, uid)?; + // The kernel gives a non-dumpable process's /proc files to root (the + // directory itself keeps the owner's uid, so a file inside is asked). + let owner = std::fs::metadata(proc.join("status")) + .map_err(|e| format!("the caller's /proc entry is unreadable ({e})"))? + .uid(); + if owner != 0 { + return Err( + "the caller did not seal itself at startup (an old or modified build; \ + restart the app after an upgrade)" + .into(), + ); + } + let matched = image_matches(&proc, allowlist)?; + holds_connection(&proc, sock_ino)?; + maps_are_sealed( + &std::fs::read_to_string(proc.join("maps")) + .map_err(|e| format!("the caller's memory map is unreadable ({e})"))?, + )?; + Ok(matched) +} + +/// `TracerPid` is 0 and the effective uid is `uid`. +fn status_is_clean(status: &str, uid: u32) -> Result<(), String> { + let field = |name: &str| { + status + .lines() + .find_map(|line| line.strip_prefix(name)) + .map(str::split_whitespace) + }; + let tracer = field("TracerPid:").and_then(|mut v| v.next()?.parse::().ok()); + match tracer { + Some(0) => {} + Some(_) => return Err("the caller is being traced".into()), + None => return Err("the caller's status has no TracerPid".into()), + } + let effective = field("Uid:").and_then(|mut v| v.nth(1)?.parse::().ok()); + if effective != Some(uid) { + return Err("the caller's effective uid is not the connection's uid".into()); + } + Ok(()) +} + +/// The running image is one of the sealed allowlist entries, by dev/ino. +fn image_matches(proc: &Path, allowlist: &[PathBuf]) -> Result { + // stat follows the magic link to the mapped inode, even a deleted one. + let image = std::fs::metadata(proc.join("exe")) + .map_err(|e| format!("the caller's executable is unreadable ({e})"))?; + let key = (image.dev(), image.ino()); + if let Some(entry) = allowlist + .iter() + .find(|entry| sealed_identity(entry) == Some(key)) + { + return Ok(entry.clone()); + } + let shown = std::fs::read_link(proc.join("exe")).unwrap_or_default(); + let name = shown + .to_string_lossy() + .trim_end_matches(crate::process_resolve::DELETED_SUFFIX) + .rsplit('/') + .next() + .unwrap_or_default() + .to_string(); + Err( + match allowlist.iter().find(|entry| { + entry + .file_name() + .is_some_and(|f| f.to_string_lossy() == name) + }) { + Some(entry) => format!( + "this {name} is not the installed {} (restart it after an upgrade)", + entry.display() + ), + None => format!( + "{} is not an installed Colony Firewall app or tray", + shown.display() + ), + }, + ) +} + +/// `(dev, ino)` of `path` when it and every ancestor pass the sealed test: +/// a regular file (directories for the ancestors) owned by root and not +/// writable by group or other. No sticky-directory exception: the files the +/// daemon trusts here live in root-owned directories nobody else can write. +pub fn sealed_identity(path: &Path) -> Option<(u64, u64)> { + use cfc_core::exe_path::file_is_sealed; + if !path.is_absolute() { + return None; + } + let meta = std::fs::symlink_metadata(path).ok()?; + if !meta.is_file() || !file_is_sealed(meta.uid(), meta.mode()) { + return None; + } + for dir in path.ancestors().skip(1) { + let m = std::fs::symlink_metadata(dir).ok()?; + if !m.is_dir() || !file_is_sealed(m.uid(), m.mode()) { + return None; + } + } + Some((meta.dev(), meta.ino())) +} + +fn sealed_dir(path: &Path) -> bool { + use cfc_core::exe_path::file_is_sealed; + path.ancestors().all(|dir| { + std::fs::symlink_metadata(dir) + .is_ok_and(|m| m.is_dir() && file_is_sealed(m.uid(), m.mode())) + }) +} + +/// The process holds the client end of the daemon's connection `sock_ino` +/// on a descriptor of its own above stderr. +fn holds_connection(proc: &Path, sock_ino: u64) -> Result<(), String> { + let peer = crate::sock_diag::unix_peer_inode(sock_ino) + .map_err(|e| format!("the connection's client end is unknown (unix_diag: {e})"))?; + let wanted = format!("socket:[{peer}]"); + let fds = std::fs::read_dir(proc.join("fd")) + .map_err(|e| format!("the caller's descriptors are unreadable ({e})"))?; + let held = fds.filter_map(Result::ok).any(|entry| { + entry + .file_name() + .to_str() + .and_then(|n| n.parse::().ok()) + .is_some_and(|n| n >= 3) + && std::fs::read_link(entry.path()) + .is_ok_and(|target| target.as_os_str() == wanted.as_str()) + }); + if held { + Ok(()) + } else { + Err("the connection is not held by the caller itself".into()) + } +} + +/// Every executable file mapping comes from a sealed file. +/// +/// A live path must be sealed and be the mapped inode. The device is not +/// compared: on btrfs `stat` reports the subvolume's device while the maps +/// line carries the superblock's, so they differ for every file. A deleted +/// path (a library an upgrade replaced while the app ran) must have a sealed +/// parent directory other than `/`: only root can have created a file there. +/// `/` is excluded because memfd and SysV shared memory show up as +/// `/memfd:…` and `/SYSV…` "(deleted)". Anonymous mappings are not judged: +/// GPU drivers JIT into them. +fn maps_are_sealed(maps: &str) -> Result<(), String> { + let mut seen: HashMap<&str, bool> = HashMap::new(); + for line in maps.lines() { + let mut fields = line.splitn(6, ' '); + let (Some(_), Some(perms), Some(_), Some(_), Some(inode)) = ( + fields.next(), + fields.next(), + fields.next(), + fields.next(), + fields.next(), + ) else { + continue; + }; + let path = fields.next().unwrap_or_default().trim_start(); + if !perms.contains('x') || !path.starts_with('/') { + continue; + } + let ok = *seen.entry(path).or_insert_with(|| { + match path.strip_suffix(crate::process_resolve::DELETED_SUFFIX) { + Some(gone) => Path::new(gone) + .parent() + .is_some_and(|dir| dir != Path::new("/") && sealed_dir(dir)), + None => inode.parse::().is_ok_and(|inode| { + sealed_identity(Path::new(path)).is_some_and(|(_, ino)| ino == inode) + }), + } + }); + if !ok { + return Err(format!("the caller loaded {path}")); + } + } + Ok(()) +} + +#[cfg(test)] +mod tests { + use super::*; + use std::cell::Cell; + use std::process::{Child, Command}; + + fn sealed(path: &str) -> bool { + sealed_identity(Path::new(path)).is_some() + } + + struct Reaped(Child); + impl Drop for Reaped { + fn drop(&mut self) { + let _ = self.0.kill(); + let _ = self.0.wait(); + } + } + + fn spawn(program: &Path) -> Reaped { + // A freshly copied binary can be briefly "busy": another test thread + // forked while its write descriptor was open. + let mut attempt = 0; + let child = loop { + match Command::new(program).arg("30").spawn() { + Ok(child) => break Reaped(child), + Err(e) if e.kind() == std::io::ErrorKind::ExecutableFileBusy && attempt < 100 => { + attempt += 1; + std::thread::sleep(std::time::Duration::from_millis(10)); + } + Err(e) => panic!("spawning {}: {e}", program.display()), + } + }; + // Wait for the exec to land, so /proc//exe is the new image. + let exe = PathBuf::from(format!("/proc/{}/exe", child.0.id())); + for _ in 0..200 { + if std::fs::read_link(&exe).is_ok_and(|p| p.ends_with(program.file_name().unwrap())) { + break; + } + std::thread::sleep(std::time::Duration::from_millis(5)); + } + child + } + + #[test] + fn strict_sealed_rejects_sticky_and_group_writable_ancestors() { + let dir = tempfile::tempdir().unwrap(); + let file = dir.path().join("tool"); + std::fs::write(&file, b"x").unwrap(); + assert!(!sealed(file.to_str().unwrap()), "user-owned"); + assert!(!sealed("relative/path")); + assert!(!sealed("/tmp"), "a directory is not an image"); + // /tmp is sticky and world-writable: root-owned, still not sealed. + assert!(!sealed_dir(Path::new("/tmp"))); + if sealed("/usr/bin/env") { + assert!(sealed_dir(Path::new("/usr/bin"))); + } else { + eprintln!("skipped: /usr/bin/env is not root-sealed here"); + } + } + + #[test] + fn an_installed_image_matches_by_dev_and_ino() { + let sleep = Path::new("/usr/bin/sleep"); + if !sealed("/usr/bin/sleep") { + eprintln!("skipped: /usr/bin/sleep is not root-sealed here"); + return; + } + let child = spawn(sleep); + let proc = PathBuf::from(format!("/proc/{}", child.0.id())); + assert_eq!(image_matches(&proc, &[sleep.into()]).unwrap(), sleep); + + let dir = tempfile::tempdir().unwrap(); + let copy = dir.path().join("sleep"); + std::fs::copy(sleep, ©).unwrap(); + let copied = spawn(©); + let proc = PathBuf::from(format!("/proc/{}", copied.0.id())); + let error = image_matches(&proc, &[sleep.into()]).unwrap_err(); + assert!( + error.contains("not the installed /usr/bin/sleep"), + "{error}" + ); + } + + #[test] + fn a_deleted_image_is_refused() { + let dir = tempfile::tempdir().unwrap(); + let copy = dir.path().join("sleep"); + if std::fs::copy("/usr/bin/sleep", ©).is_err() { + eprintln!("skipped: no /usr/bin/sleep"); + return; + } + let child = spawn(©); + std::fs::remove_file(©).unwrap(); + let proc = PathBuf::from(format!("/proc/{}", child.0.id())); + let error = image_matches(&proc, &["/usr/bin/sleep".into()]).unwrap_err(); + assert!(error.contains("restart"), "{error}"); + } + + #[test] + fn tracer_pid_parsing() { + let clean = "Name:\tx\nTracerPid:\t0\nUid:\t1000\t1000\t1000\t1000\n"; + assert!(status_is_clean(clean, 1000).is_ok()); + let traced = clean.replace("TracerPid:\t0", "TracerPid:\t4242"); + assert_eq!( + status_is_clean(&traced, 1000).unwrap_err(), + "the caller is being traced" + ); + assert!(status_is_clean("Uid:\t1000\t1000\t1000\t1000\n", 1000).is_err()); + } + + #[test] + fn uid_mismatch_in_status_is_refused() { + // Real uid 1000, effective uid 1001: the effective one decides. + let status = "TracerPid:\t0\nUid:\t1000\t1001\t1000\t1000\n"; + assert!(status_is_clean(status, 1000).is_err()); + assert!(status_is_clean(status, 1001).is_ok()); + } + + #[test] + fn non_dumpable_is_detected_from_proc_ownership() { + if nix::unistd::geteuid().is_root() { + eprintln!("skipped: every /proc entry is root's when the tests run as root"); + return; + } + let owner = |pid: i32| { + std::fs::metadata(format!("/proc/{pid}/status")) + .unwrap() + .uid() + }; + // SAFETY: the child only makes raw syscalls and never returns. + let child = unsafe { libc::fork() }; + assert!(child >= 0); + if child == 0 { + unsafe { + libc::prctl(libc::PR_SET_DUMPABLE, 0, 0, 0, 0); + loop { + libc::pause(); + } + } + } + let plain = spawn(Path::new("/usr/bin/sleep")); + let mut sealed_owner = owner(child); + for _ in 0..200 { + if sealed_owner == 0 { + break; + } + std::thread::sleep(std::time::Duration::from_millis(5)); + sealed_owner = owner(child); + } + let plain_owner = owner(plain.0.id() as i32); + // SAFETY: killing and reaping our own child. + unsafe { + libc::kill(child, libc::SIGKILL); + libc::waitpid(child, std::ptr::null_mut(), 0); + } + assert_eq!( + sealed_owner, 0, + "a non-dumpable process's /proc files are root's" + ); + assert_ne!(plain_owner, 0); + } + + #[test] + fn a_socket_held_only_on_stdio_does_not_count() { + let (ours, theirs) = std::os::unix::net::UnixStream::pair().unwrap(); + let inode = |fd: i32| { + // SAFETY: zeroed stat is valid out-param storage; fd is open. + let mut st: libc::stat = unsafe { std::mem::zeroed() }; + assert_eq!(unsafe { libc::fstat(fd, &mut st) }, 0); + st.st_ino + }; + use std::os::fd::AsRawFd; + if crate::sock_diag::unix_peer_inode(inode(ours.as_raw_fd())).is_err() { + eprintln!("skipped: unix_diag unavailable here"); + return; + } + // This process holds `theirs` above stderr: it counts. + assert!(holds_connection(Path::new("/proc/self"), inode(ours.as_raw_fd())).is_ok()); + // A child holding it only on stdout does not. + let child = Reaped( + Command::new("/usr/bin/sleep") + .arg("30") + .stdout(std::process::Stdio::from(std::os::fd::OwnedFd::from( + theirs, + ))) + .spawn() + .unwrap(), + ); + let proc = PathBuf::from(format!("/proc/{}", child.0.id())); + // Its fds are inspected by the test as the same user; the socket is + // still ours, so ask about our end. + let error = holds_connection(&proc, inode(ours.as_raw_fd())).unwrap_err(); + assert!(error.contains("not held"), "{error}"); + } + + #[test] + fn maps_check_accepts_sealed_libraries_and_refuses_user_files() { + if !sealed("/usr/bin/sleep") { + eprintln!("skipped: /usr/bin/sleep is not root-sealed here"); + return; + } + let child = spawn(Path::new("/usr/bin/sleep")); + let maps = std::fs::read_to_string(format!("/proc/{}/maps", child.0.id())).unwrap(); + assert_eq!(maps_are_sealed(&maps), Ok(())); + + let own = std::env::current_exe().unwrap(); + if sealed(own.to_str().unwrap()) { + eprintln!("skipped: the test binary itself is root-sealed"); + } else { + let maps = std::fs::read_to_string("/proc/self/maps").unwrap(); + let error = maps_are_sealed(&maps).unwrap_err(); + assert!(error.starts_with("the caller loaded /"), "{error}"); + } + + // Synthetic lines: memfd and SysV mappings are refused, an anonymous + // one is not judged, a library an upgrade replaced is accepted. + let line = |path: &str| format!("7f00-7f01 r-xp 00000000 00:1c 99 {path}\n"); + assert!(maps_are_sealed(&line("/memfd:payload (deleted)")).is_err()); + assert!(maps_are_sealed(&line("/SYSV00000000 (deleted)")).is_err()); + assert!(maps_are_sealed("7f00-7f01 r-xp 00000000 00:00 0 \n").is_ok()); + if sealed_dir(Path::new("/usr/lib")) { + assert!(maps_are_sealed(&line("/usr/lib/libgone.so.1 (deleted)")).is_ok()); + } + assert!(maps_are_sealed(&line("/home/u/libgone.so.1 (deleted)")).is_err()); + assert!( + maps_are_sealed("7f00-7f01 r--p 00000000 00:1c 99 /home/u/x.so\n").is_ok(), + "not executable" + ); + } + + #[test] + fn starttime_changed_between_reads_is_refused() { + let reads = Cell::new(0); + let changing = |_| { + reads.set(reads.get() + 1); + Some(if reads.get() == 1 { 100 } else { 101 }) + }; + let error = bracketed(1, 100, changing, || Ok(())).unwrap_err(); + assert!(error.contains("another process"), "{error}"); + assert!(bracketed( + 1, + 100, + |_| Some(99), + || -> Result<(), String> { panic!("the body must not run after a failed first read") } + ) + .is_err()); + assert_eq!(bracketed(1, 100, |_| Some(100), || Ok(7)), Ok(7)); + } +} diff --git a/crates/cfc-daemon/src/prompts.rs b/crates/cfc-daemon/src/prompts.rs index b217886..c3a6987 100644 --- a/crates/cfc-daemon/src/prompts.rs +++ b/crates/cfc-daemon/src/prompts.rs @@ -104,8 +104,9 @@ struct RouterInner { /// resolution path to remove an id sends the verdict. pending: Mutex>, broadcast_tx: broadcast::Sender, - /// Census of live `StreamPrompts` subscribers: peer uid -> how many - /// streams that uid has open. Maintained by [`PromptSubscription`], + /// Census of live *answering* `StreamPrompts` subscribers (root, or the + /// official app and tray): peer uid -> how many streams that uid has + /// open. Read-only watchers are not counted. Maintained by [`PromptSubscription`], /// which registers on creation and deregisters on drop, so it tracks /// the broadcast receivers exactly as closely as /// `broadcast_tx.receiver_count()` did - a stream whose client has gone @@ -180,14 +181,22 @@ impl PromptRouter { /// /// The uid is the kernel-reported `SO_PEERCRED` uid of the client, not /// anything the client said about itself; it decides which prompts the - /// subscription may see (see [`should_deliver`]) and is counted in the - /// router's census until the returned value is dropped. - pub fn subscribe(&self, peer_uid: u32) -> PromptSubscription { - self.inner.register(peer_uid); + /// subscription may see (see [`should_deliver`]). + /// + /// `answering` says whether this peer may answer what it is shown (root, + /// or the official app and tray). Only such a subscription is counted in + /// the router's census, until the returned value is dropped: a read-only + /// watcher sees prompts but is not a UI, so it must not hold a prompt for + /// `prompt_timeout_secs` that nobody can answer instead of letting + /// `no_ui_action` apply now. + pub fn subscribe(&self, peer_uid: u32, answering: bool) -> PromptSubscription { + if answering { + self.inner.register(peer_uid); + } PromptSubscription { rx: self.inner.broadcast_tx.subscribe(), inner: self.inner.clone(), - uid: peer_uid, + uid: answering.then_some(peer_uid), } } @@ -292,7 +301,8 @@ impl PromptRouter { pub struct PromptSubscription { rx: broadcast::Receiver, inner: Arc, - uid: u32, + /// The census entry this subscription holds; `None` for a read-only one. + uid: Option, } impl PromptSubscription { @@ -307,7 +317,9 @@ impl PromptSubscription { impl Drop for PromptSubscription { fn drop(&mut self) { - self.inner.unregister(self.uid); + if let Some(uid) = self.uid { + self.inner.unregister(uid); + } } } @@ -498,7 +510,7 @@ mod tests { let (tx, rx) = std::sync::mpsc::channel(); let stats = Stats::new(); let router = PromptRouter::new(shared(dp(3600)), stats.clone(), tx); - let _sub = router.subscribe(1000); + let _sub = router.subscribe(1000, true); router.enqueue(req_owned_by(5, 1001), PromptBinding::default()); @@ -511,12 +523,35 @@ mod tests { assert!(router.submit("5", user_allow()).is_none()); } + #[tokio::test] + async fn a_read_only_subscriber_does_not_count_as_a_ui() { + let (tx, rx) = std::sync::mpsc::channel(); + let stats = Stats::new(); + let router = PromptRouter::new(shared(dp(3600)), stats.clone(), tx); + let mut watcher = router.subscribe(1000, false); + + // Only a watcher: no_ui_action now, nothing held for the timeout. + router.enqueue(req_owned_by(1, 1000), PromptBinding::default()); + let pv = rx.try_recv().expect("no_ui_action applies immediately"); + assert_eq!(pv.verdict.source, VerdictSource::DefaultPolicy); + assert_eq!(stats.prompts_pending(), 0); + + // With an answering UI it waits, and the watcher sees it too. + let _ui = router.subscribe(1000, true); + router.enqueue(req_owned_by(2, 1000), PromptBinding::default()); + assert!( + rx.try_recv().is_err(), + "an answering UI makes the prompt wait" + ); + assert_eq!(watcher.recv().await.unwrap().prompt_id, "2"); + } + #[tokio::test] async fn a_prompt_a_subscriber_may_see_is_broadcast() { let (tx, rx) = std::sync::mpsc::channel(); let stats = Stats::new(); let router = PromptRouter::new(shared(dp(3600)), stats.clone(), tx); - let mut sub = router.subscribe(1000); + let mut sub = router.subscribe(1000, true); router.enqueue(req_owned_by(6, 1000), PromptBinding::default()); @@ -529,7 +564,7 @@ mod tests { async fn a_root_subscriber_is_an_audience_for_every_prompt() { let (tx, rx) = std::sync::mpsc::channel(); let router = PromptRouter::new(shared(dp(3600)), Stats::new(), tx); - let mut sub = router.subscribe(0); + let mut sub = router.subscribe(0, true); router.enqueue(req_owned_by(8, 1001), PromptBinding::default()); @@ -544,8 +579,8 @@ mod tests { // Two windows for the same uid: the first drop must not deregister // the session. - let sub_a = router.subscribe(1000); - let sub_b = router.subscribe(1000); + let sub_a = router.subscribe(1000, true); + let sub_b = router.subscribe(1000, true); drop(sub_a); router.enqueue(req_owned_by(1, 1000), PromptBinding::default()); assert!(rx.try_recv().is_err(), "uid 1000 still has a UI open"); @@ -597,7 +632,7 @@ mod tests { let (tx, rx) = std::sync::mpsc::channel(); let stats = Stats::new(); let router = PromptRouter::new(shared(dp(3600)), stats.clone(), tx); - let mut sub = router.subscribe(1000); + let mut sub = router.subscribe(1000, true); router.enqueue(req(1), PromptBinding::default()); let event = sub.recv().await.unwrap(); @@ -645,7 +680,7 @@ mod tests { let (tx, rx) = std::sync::mpsc::channel(); let stats = Stats::new(); let router = PromptRouter::new(shared(dp(1)), stats.clone(), tx); - let _sub = router.subscribe(1000); // keep a UI "connected" + let _sub = router.subscribe(1000, true); // keep a UI "connected" router.enqueue(req(9), PromptBinding::default()); assert_eq!(stats.prompts_pending(), 1); @@ -748,7 +783,7 @@ mod tests { // returns - the three legs the persist path stands on. let (tx, _rx) = std::sync::mpsc::channel(); let router = PromptRouter::new(shared(dp(3600)), Stats::new(), tx); - let mut sub = router.subscribe(1000); + let mut sub = router.subscribe(1000, true); let binding = PromptBinding { exe: Some(std::path::PathBuf::from("/home/u/.local/bin/tool")), @@ -772,7 +807,7 @@ mod tests { async fn a_sealed_prompt_does_not_announce_a_binding() { let (tx, _rx) = std::sync::mpsc::channel(); let router = PromptRouter::new(shared(dp(3600)), Stats::new(), tx); - let mut sub = router.subscribe(1000); + let mut sub = router.subscribe(1000, true); router.enqueue(req_owned_by(10, 1000), PromptBinding::default()); let event = sub.recv().await.unwrap(); assert!(!event.binds_to_hash); diff --git a/crates/cfc-daemon/src/sock_diag.rs b/crates/cfc-daemon/src/sock_diag.rs index 4032c55..70f7141 100644 --- a/crates/cfc-daemon/src/sock_diag.rs +++ b/crates/cfc-daemon/src/sock_diag.rs @@ -134,7 +134,12 @@ fn ask(req: &[u8; REQ_LEN]) -> Option { } } } - match slot.as_ref().map(|s| s.round_trip(req, seq)) { + let mut buf = [0u8; 8192]; + let reply = slot.as_ref().map(|s| match s.exchange(req, seq, &mut buf) { + Some(n) => parse_response(&buf[..n], req[17]).map_or(Reply::NotFound, Reply::Found), + None => Reply::Desync, + }); + match reply { Some(Reply::Found(info)) => Some(info), // A correctly-sequenced "no such socket". The socket is clean, so // it is kept; the caller falls back to /proc as before. @@ -283,7 +288,9 @@ impl DiagSocket { Ok(sock) } - fn round_trip(&self, req: &[u8], seq: u32) -> Reply { + /// Sends `req` and receives its answer into `buf`. `Some(len)` only for + /// an answer carrying `seq`; `None` means the socket's state is unknown. + fn exchange(&self, req: &[u8], seq: u32, buf: &mut [u8]) -> Option { let fd = self.0.as_raw_fd(); // SAFETY: zeroed sockaddr_nl is a valid "to the kernel" address. @@ -307,10 +314,9 @@ impl DiagSocket { "sock_diag send failed ({}); falling back to /proc", std::io::Error::last_os_error() ); - return Reply::Desync; + return None; } - let mut buf = [0u8; 8192]; // SAFETY: buf is a valid writable buffer of the stated length. let n = unsafe { libc::recv(fd, buf.as_mut_ptr().cast(), buf.len(), 0) }; if n <= 0 { @@ -318,21 +324,107 @@ impl DiagSocket { "sock_diag recv failed ({}); falling back to /proc", std::io::Error::last_os_error() ); - return Reply::Desync; + return None; } - let buf = &buf[..n as usize]; + let n = n as usize; // The answer must be to *this* request. Anything else means a previous // request's answer arrived after its timeout, and this socket cannot // be trusted to be at a message boundary any more. - if reply_seq(buf) != Some(seq) { + if reply_seq(&buf[..n]) != Some(seq) { trace!("sock_diag answered a different request; discarding the socket"); - return Reply::Desync; + return None; + } + Some(n) + } +} + +// --------------------------------------------------------------------------- +// AF_UNIX: the other end of a connection +// --------------------------------------------------------------------------- + +const UNIX_DIAG_REQ_LEN: usize = 24; +const UNIX_REQ_LEN: usize = NLMSG_HDR_LEN + UNIX_DIAG_REQ_LEN; +/// Fixed part of struct unix_diag_msg (before the attributes). +const UNIX_DIAG_MSG_LEN: usize = 16; +const UDIAG_SHOW_PEER: u32 = 0x4; +const UNIX_DIAG_PEER: u16 = 2; +const NLMSG_ERROR: u16 = 2; + +/// Inode of the socket at the other end of the AF_UNIX socket `inode`. +/// +/// One exact UNIX_DIAG query with `UDIAG_SHOW_PEER`. The daemon passes the +/// inode of its own end of a control connection and gets the client's end, +/// which it then looks for among the client's descriptors. Errors are the +/// kernel's (`ENOENT` for a socket that is gone, `EOPNOTSUPP` or +/// `EPROTONOSUPPORT` without the `unix_diag` module) or `InvalidData` for an +/// answer without a peer. +pub fn unix_peer_inode(inode: u64) -> std::io::Result { + use std::io::{Error, ErrorKind}; + let inode = u32::try_from(inode) + .map_err(|_| Error::new(ErrorKind::InvalidInput, "socket inode out of range"))?; + let seq = 1; + let req = build_unix_request(inode, seq); + let socket = DiagSocket::open()?; + let mut buf = [0u8; 8192]; + let n = socket + .exchange(&req, seq, &mut buf) + .ok_or_else(|| Error::new(ErrorKind::TimedOut, "no answer from sock_diag"))?; + parse_unix_peer(&buf[..n], inode) +} + +fn build_unix_request(inode: u32, seq: u32) -> [u8; UNIX_REQ_LEN] { + let mut buf = [0u8; UNIX_REQ_LEN]; + buf[0..4].copy_from_slice(&(UNIX_REQ_LEN as u32).to_ne_bytes()); + buf[4..6].copy_from_slice(&SOCK_DIAG_BY_FAMILY.to_ne_bytes()); + buf[6..8].copy_from_slice(&(libc::NLM_F_REQUEST as u16).to_ne_bytes()); + buf[8..12].copy_from_slice(&seq.to_ne_bytes()); + // struct unix_diag_req: family, protocol, pad, states, ino, show, cookie. + buf[16] = libc::AF_UNIX as u8; + buf[20..24].copy_from_slice(&u32::MAX.to_ne_bytes()); + buf[24..28].copy_from_slice(&inode.to_ne_bytes()); + buf[28..32].copy_from_slice(&UDIAG_SHOW_PEER.to_ne_bytes()); + buf[32..40].copy_from_slice(&[0xFF; 8]); // NOCOOKIE + buf +} + +fn parse_unix_peer(buf: &[u8], inode: u32) -> std::io::Result { + use std::io::{Error, ErrorKind}; + let bad = || Error::new(ErrorKind::InvalidData, "malformed unix_diag answer"); + let header = buf.get(..NLMSG_HDR_LEN).ok_or_else(bad)?; + let msg_len = u32::from_ne_bytes(header[0..4].try_into().map_err(|_| bad())?) as usize; + let msg_type = u16::from_ne_bytes(header[4..6].try_into().map_err(|_| bad())?); + let msg = buf.get(NLMSG_HDR_LEN..msg_len).ok_or_else(bad)?; + if msg_type == NLMSG_ERROR { + let errno = i32::from_ne_bytes( + msg.get(..4) + .ok_or_else(bad)? + .try_into() + .map_err(|_| bad())?, + ); + return Err(Error::from_raw_os_error(-errno)); + } + if msg_type != SOCK_DIAG_BY_FAMILY || msg.len() < UNIX_DIAG_MSG_LEN { + return Err(bad()); + } + if u32::from_ne_bytes(msg[4..8].try_into().map_err(|_| bad())?) != inode { + return Err(bad()); + } + let mut attrs = &msg[UNIX_DIAG_MSG_LEN..]; + while attrs.len() >= 4 { + let len = u16::from_ne_bytes([attrs[0], attrs[1]]) as usize; + let kind = u16::from_ne_bytes([attrs[2], attrs[3]]); + if len < 4 || len > attrs.len() { + break; } - match parse_response(buf, req[17]) { - Some(info) => Reply::Found(info), - None => Reply::NotFound, + if kind == UNIX_DIAG_PEER && len >= 8 { + let peer = u32::from_ne_bytes(attrs[4..8].try_into().map_err(|_| bad())?); + if peer != 0 { + return Ok(u64::from(peer)); + } } + attrs = &attrs[(len + 3) & !3..]; } + Err(Error::new(ErrorKind::InvalidData, "the socket has no peer")) } #[cfg(test)] @@ -514,6 +606,63 @@ mod tests { assert_eq!(got, None); } + #[test] + fn unix_peer_reply_parsing() { + let mut buf = vec![0u8; NLMSG_HDR_LEN + UNIX_DIAG_MSG_LEN + 8 + 8]; + let len = buf.len() as u32; + buf[0..4].copy_from_slice(&len.to_ne_bytes()); + buf[4..6].copy_from_slice(&SOCK_DIAG_BY_FAMILY.to_ne_bytes()); + buf[NLMSG_HDR_LEN + 4..NLMSG_HDR_LEN + 8].copy_from_slice(&41u32.to_ne_bytes()); + let attrs = NLMSG_HDR_LEN + UNIX_DIAG_MSG_LEN; + // An unrelated attribute first (RQLEN), then the peer. + buf[attrs..attrs + 2].copy_from_slice(&8u16.to_ne_bytes()); + buf[attrs + 2..attrs + 4].copy_from_slice(&4u16.to_ne_bytes()); + buf[attrs + 8..attrs + 10].copy_from_slice(&8u16.to_ne_bytes()); + buf[attrs + 10..attrs + 12].copy_from_slice(&UNIX_DIAG_PEER.to_ne_bytes()); + buf[attrs + 12..attrs + 16].copy_from_slice(&42u32.to_ne_bytes()); + assert_eq!(parse_unix_peer(&buf, 41).unwrap(), 42); + assert!( + parse_unix_peer(&buf, 40).is_err(), + "an answer about another socket" + ); + + let mut error = vec![0u8; NLMSG_HDR_LEN + 4]; + let error_len = error.len() as u32; + error[0..4].copy_from_slice(&error_len.to_ne_bytes()); + error[4..6].copy_from_slice(&NLMSG_ERROR.to_ne_bytes()); + error[NLMSG_HDR_LEN..].copy_from_slice(&(-libc::ENOENT).to_ne_bytes()); + assert_eq!( + parse_unix_peer(&error, 41).unwrap_err().raw_os_error(), + Some(libc::ENOENT) + ); + let req = build_unix_request(41, 9); + assert_eq!(reply_seq(&req), Some(9)); + assert_eq!(&req[24..28], &41u32.to_ne_bytes()); + } + + #[test] + fn unix_peer_inode_names_the_other_end() { + let (a, b) = std::os::unix::net::UnixStream::pair().unwrap(); + let inode = |fd: i32| { + // SAFETY: zeroed stat is valid out-param storage; fd is open. + let mut st: libc::stat = unsafe { std::mem::zeroed() }; + assert_eq!(unsafe { libc::fstat(fd, &mut st) }, 0); + st.st_ino + }; + match unix_peer_inode(inode(a.as_raw_fd())) { + Ok(peer) => assert_eq!(peer, inode(b.as_raw_fd())), + Err(error) + if matches!( + error.raw_os_error(), + Some(libc::EPERM | libc::ENOENT | libc::EOPNOTSUPP | libc::EPROTONOSUPPORT) + ) => + { + eprintln!("skipped: unix_diag unavailable here ({error})"); + } + Err(error) => panic!("unix_peer_inode: {error}"), + } + } + fn socket_inode(sock: &UdpSocket) -> u64 { // SAFETY: zeroed stat is valid out-param storage; fd is open. let mut st: libc::stat = unsafe { std::mem::zeroed() }; diff --git a/crates/cfc-daemon/tests/ipc_integration.rs b/crates/cfc-daemon/tests/ipc_integration.rs index 8f25f17..0e1bc3f 100644 --- a/crates/cfc-daemon/tests/ipc_integration.rs +++ b/crates/cfc-daemon/tests/ipc_integration.rs @@ -122,6 +122,7 @@ impl TestDaemonBuilder { // authorization assertions deterministic. group: format!("cfc-absent-{}", uuid::Uuid::new_v4()), require_group: self.require_group, + ..IpcConfig::default() }, pause_default_secs: self.pause_default_secs, dry_run: false, @@ -820,43 +821,32 @@ async fn unattributed_prompt_is_delivered_to_any_session() { // Peer-credential authorization (wave 3) // --------------------------------------------------------------------------- -/// The production shape: `require_group = true` with a socket the daemon -/// could not gate (no such group / not root). Mutating RPCs must be refused -/// for non-root peers; read-only RPCs must still work. +/// A client with the daemon's own uid has full control, with +/// `require_group = true` and a socket the daemon could not gate. In +/// production the daemon is root, so this is root; here it is what lets every +/// other round trip in this file run unprivileged. Who else may change what +/// (the official app and tray, read-only peers, polkit) is pinned by the +/// `authorization_table` unit test in `src/ipc.rs`, which can fake peers this +/// single-uid process cannot be. #[tokio::test] -async fn require_group_refuses_non_root_mutation() { +async fn the_daemons_own_uid_has_full_control() { let d = TestDaemon::builder().require_group(true).build().await; let mut client = d.client().await; - // Read-only stays open: layer 1 (the socket mode) already decided who - // may connect at all. client.status().await.expect("status is read-only"); - assert!(client - .list_rules() - .await - .expect("list_rules is read-only") - .is_empty()); - - let result = client + let id = client .upsert_rule(rule_pb("blocked", pb::Action::Deny, scope_port(25))) - .await; - - if running_as_root() { - assert!( - result.is_ok(), - "root may always mutate, gated socket or not" - ); - } else { - let status = status_of(result.expect_err("a non-root mutation must be refused")); - assert_eq!(status.code(), tonic::Code::PermissionDenied); - assert!( - status.message().contains("uid 0"), - "unexpected message: {}", - status.message() - ); - // And nothing was written on the way to the refusal. - assert!(client.list_rules().await.expect("listing rules").is_empty()); - } + .await + .expect("the daemon's own uid may change rules"); + assert!(client.delete_rule(&id).await.expect("and delete them")); + assert!(client.set_paused(true, 60).await.expect("and pause").paused); + assert!( + !client + .set_paused(false, 0) + .await + .expect("and resume") + .paused + ); } // --------------------------------------------------------------------------- diff --git a/systemd/daemon.toml.sample b/systemd/daemon.toml.sample index 1a789b2..ca61462 100644 --- a/systemd/daemon.toml.sample +++ b/systemd/daemon.toml.sample @@ -251,20 +251,25 @@ enabled = true # # 1. The socket file. After bind the daemon chowns it to root: and # chmods it 0660, so the kernel refuses connect(2) to anyone outside -# the group. Group membership IS the credential - there is no in-band -# authentication, no per-user identity, no password. -# 2. Peer credentials. Every connection carries SO_PEERCRED, and the -# daemon logs the calling uid/pid for every mutating RPC. Mutating -# RPCs (UpsertRule, ApplyRules, DeleteRule, SetPaused, SubmitVerdict) need uid 0 -# or a socket that is genuinely group-gated. Read-only RPCs +# the group. Membership is what lets a process connect at all. +# 2. Peer identity. Every connection carries SO_PEERCRED. Read-only RPCs # (ListRules, GetStatus, ListEvents, StreamConnections, StreamPrompts) -# are open to any peer that got through layer 1. +# are open to any peer that got through layer 1. Changes (SubmitVerdict, +# UpsertRule, DeleteRule, SetPaused, ApplyRules) are accepted only from: +# - root (sudo cfc ...); +# - the installed Colony Firewall app and tray (official_clients +# below), run by a member of the group: the daemon checks that the +# calling process runs that exact installed, root-owned binary, is +# not traced, holds the connection itself and loaded no executable +# code from outside root-owned directories. +# Every other program of the desktop user is READ-ONLY, including a +# non-root `cfc`. Its prompt subscription is not counted as a UI, so +# no_ui_action still applies when only it is listening. # -# Say it plainly: EVERY MEMBER OF THIS GROUP IS FULLY TRUSTED. Membership -# grants the ability to allow or deny any traffic on this host. This is not -# a multi-user privilege boundary - add only administrators of this machine. +# Group membership therefore no longer grants control by itself: it gives +# the desktop session read access and lets the app and tray connect. # -# One exception to "any group member can do anything": prompt answers. +# Prompt answers have one more check. # Prompts are addressed by the uid that owns the process they are about. # A StreamPrompts subscription is only handed a prompt when the subscriber's # uid matches that owner, and only a peer a prompt was actually handed to @@ -287,9 +292,17 @@ enabled = true # `usermod -aG colony-firewall ` and log back in. group = "colony-firewall" -# Whether a non-root peer must reach the daemon through a genuinely -# group-gated socket before it may change anything. Leave this true. Set it -# to false only if you gate the socket some other way (e.g. filesystem -# ACLs); with it false, any process that manages to connect can rewrite -# your firewall rules. +# Whether the official app and tray must be run by a proved member of the +# group before they may change anything. Leave this true. Setting it to false +# waives only the group check, never the official-client check: every other +# non-root program stays read-only either way. require_group = true + +# The only non-root programs that may answer prompts, edit rules and ask to +# pause: absolute paths of the installed GUI and tray. Each must be a +# root-owned file that only root can write, in root-owned directories that +# only root can write; an entry that is not is reported at startup and that +# program stays read-only. After an upgrade, restart the app and tray: a +# process still running the replaced binary is read-only until then. +# Bound at startup (restart the daemon to change it). +#official_clients = ["/usr/bin/colony-firewall", "/usr/bin/colony-firewall-tray"] From 59d76b230de4ee121aa2f2b53eedd88ecc2e0557 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:47:46 +0200 Subject: [PATCH 110/125] feat(daemon)!: require polkit for pause, resume and rule import from the app SetPaused (both directions) and ApplyRules from the official app or tray now ask polkit's CheckAuthorization for the calling process with user interaction allowed, under org.projectcolony.firewall.pause and org.projectcolony.firewall.import-rules, and give up after 120 s with the dialog cancelled. Root is never asked, nor are prompt answers or single-rule edits. The GUI waits on a 150 s client and disables Pause meanwhile; the tray runs the request on its own task so prompts keep flowing, and shows a denial as a notification. The CLI help says changes need sudo. BREAKING CHANGE: pausing or resuming from the app or tray asks for an administrator password and needs a polkit agent in the session. --- Cargo.lock | 1 + crates/cfc-cli/src/main.rs | 38 +++--- crates/cfc-daemon/Cargo.toml | 4 + crates/cfc-daemon/src/ipc.rs | 188 ++++++++++++++++++++++++++- crates/cfc-daemon/src/lib.rs | 1 + crates/cfc-daemon/src/polkit.rs | 209 +++++++++++++++++++++++++++++++ crates/cfc-proto/proto/cfc.proto | 17 ++- crates/cfc-tray/src/main.rs | 81 +++++++----- crates/cfc-tray/src/model.rs | 32 +++++ crates/cfc-ui/src/main.rs | 41 +++++- 10 files changed, 555 insertions(+), 57 deletions(-) create mode 100644 crates/cfc-daemon/src/polkit.rs diff --git a/Cargo.lock b/Cargo.lock index d4933b5..2d70800 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -778,6 +778,7 @@ dependencies = [ "tracing", "tracing-subscriber", "uuid", + "zbus", ] [[package]] diff --git a/crates/cfc-cli/src/main.rs b/crates/cfc-cli/src/main.rs index 030fe9a..b7a6191 100644 --- a/crates/cfc-cli/src/main.rs +++ b/crates/cfc-cli/src/main.rs @@ -29,7 +29,12 @@ Exit codes: 4 daemon unreachable (not running, stale socket, or no socket permission) Anywhere a rule id is accepted you may also pass a unique id prefix or the -rule's name."; +rule's name. + +Changing the firewall (adding, editing, removing or importing rules, pause and +resume, answering prompts) needs root: run cfc with sudo, or use the Colony +Firewall app or tray. As a regular user cfc is read-only and the daemon +answers a change with the reason it refused it (exit 1)."; #[derive(Debug, Parser)] #[command( @@ -74,15 +79,17 @@ enum Command { }, /// Show daemon status. Status, - /// Rules CRUD. + /// Rules CRUD (changes need sudo). Rules { #[command(subcommand)] cmd: RulesCmd, }, - /// Answer connection prompts from this terminal. + /// Answer connection prompts from this terminal (needs sudo). /// /// Without a subscriber the daemon applies its no-UI action to every - /// prompt, so this is how a headless machine gets a say. + /// prompt, so this is how a headless machine gets a say: run it as + /// `sudo cfc prompts`. As a regular user it only watches; the daemon + /// neither counts it as a UI nor accepts its answers. /// /// Keys: a=allow, d=deny, r=reject, s=skip (let it time out), q=quit. /// Then a duration (1=once, 2=until restart, 3=always) and, for the @@ -107,7 +114,8 @@ enum Command { }, /// Query the persisted verdict log ("what did this app contact?"). Log(events::LogArgs), - /// Temporarily allow all flows. + /// Temporarily allow all flows (needs sudo; the app and tray ask for an + /// administrator password instead). Pause { /// How long to stay paused, e.g. 30m, 2h. Omitted means the /// daemon's configured default; the daemon clamps the maximum. @@ -115,7 +123,7 @@ enum Command { value_parser = humantime::parse_duration)] duration: Option, }, - /// Resume normal filtering immediately. + /// Resume normal filtering immediately (needs sudo). Resume, /// Print a shell completion script. #[command(hide = true)] @@ -140,15 +148,15 @@ enum RulesCmd { List, /// Show every field of one rule. Show { id: String }, - /// Delete a rule. + /// Delete a rule (needs sudo). Remove { id: String }, - /// Flip a rule's enabled state. + /// Flip a rule's enabled state (needs sudo). Toggle { id: String }, - /// Enable a rule (idempotent). + /// Enable a rule (idempotent; needs sudo). Enable { id: String }, - /// Disable a rule (idempotent). + /// Disable a rule (idempotent; needs sudo). Disable { id: String }, - /// Add a new rule. + /// Add a new rule (needs sudo). Add(rules::AddArgs), /// Export all rules as JSON to stdout. Export { @@ -156,7 +164,7 @@ enum RulesCmd { #[arg(long)] out: Option, }, - /// Import rules from a JSON file (or stdin if omitted). + /// Import rules from a JSON file, or stdin if omitted (needs sudo). Import { /// File to read; reads stdin if omitted. file: Option, @@ -164,7 +172,7 @@ enum RulesCmd { #[arg(long)] replace: bool, }, - /// Import rules from an opensnitch rules directory or single JSON file. + /// Import rules from an opensnitch rules directory or single JSON file (needs sudo). /// /// Rules with no faithful equivalent here (hostnames, regexps, unknown /// operands) cannot be converted. By default one of them stops the import @@ -180,7 +188,7 @@ enum RulesCmd { #[arg(long, conflicts_with = "replace")] skip_unconvertible: bool, }, - /// Install a small set of sensible starter rules: system DNS, NTP + /// Install a small set of sensible starter rules (needs sudo): system DNS, NTP /// (timesyncd/chrony), DHCP clients (dhcpcd/NetworkManager/networkd), /// pacman/paru HTTPS, and the SSH client. BootstrapDefaults { @@ -188,7 +196,7 @@ enum RulesCmd { #[arg(long)] dry_run: bool, }, - /// Install or remove a named set of allow rules. + /// Install or remove a named set of allow rules (needs sudo). /// /// Every outbound rule in a bundle names an executable - there is no way /// to write "allow tcp/443" here, because a payload phoning home uses 443 diff --git a/crates/cfc-daemon/Cargo.toml b/crates/cfc-daemon/Cargo.toml index e0874de..8200df1 100644 --- a/crates/cfc-daemon/Cargo.toml +++ b/crates/cfc-daemon/Cargo.toml @@ -80,6 +80,10 @@ dns-lookup = { workspace = true } # workspace-style) on purpose -- only the daemon reads a package database. flate2 = "1.1" +# polkit CheckAuthorization on the system bus (pause, resume, rule import). +# Already in Cargo.lock through cfc-tray, with the same features. +zbus = { version = "5", default-features = false, features = ["tokio"] } + parking_lot = { workspace = true } chrono = { workspace = true } uuid = { workspace = true } diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 97dd4bc..95dd03d 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -28,6 +28,11 @@ //! directories. `require_group = false` waives the group proof for //! official clients only. //! +//! Pause, resume and `ApplyRules` change the whole firewall at once, so +//! an official client also needs a polkit authorization for them +//! ([`crate::polkit`]); root does not. Answering a prompt and editing one +//! rule never ask for a password. +//! //! Every other peer is read-only: its change RPCs get PERMISSION_DENIED //! with the reason, its prompt subscription does not count as a UI (so //! `no_ui_action` still applies), and it never enters a prompt's audience. @@ -249,6 +254,9 @@ enum Access { ReadOnly, /// Changing the firewall's behaviour: root, or the official app or tray. Control, + /// A change to the whole firewall at once: as `Control`, and an official + /// client must also get this polkit action authorized (root need not). + Elevated(&'static str), } /// Where a peer stands for [`Access::Control`], before its image is checked. @@ -278,6 +286,10 @@ fn gate(peer_uid: u32, own_uid: u32, group_ok: bool) -> Gate { /// unit tests. Not reachable from config or from any client. type OfficialCheck = fn(&PeerId, &[PathBuf]) -> Result; +/// The polkit check: `polkit::check` in production, a stub in unit tests. +type PolkitCheck = + fn(PeerId, &'static str) -> futures::future::BoxFuture<'static, Result<(), String>>; + /// Outcome of securing the socket file, and the policy knobs that decide /// what it implies for callers. #[derive(Debug, Clone)] @@ -491,6 +503,7 @@ struct FirewallService { /// `[ipc] official_clients`, bound at startup. official_clients: Arc<[PathBuf]>, official: OfficialCheck, + polkit: PolkitCheck, audience: Arc, /// Wall-clock deadline of the current pause, 0 when not paused. Held /// here rather than in `Stats` so the pause timer and `GetStatus` agree. @@ -511,14 +524,29 @@ impl FirewallService { if level == Access::ReadOnly { return Ok(peer); } - match self.standing(peer).await { - Ok(official) => { + let outcome = match self.standing(peer).await { + // Only an official client is asked; root never is. + Ok(Some(exe)) => match level { + Access::Elevated(action) => (self.polkit)(peer, action) + .await + .map(|()| (Some(exe), Some(action))) + .map_err(|reason| { + Status::permission_denied(format!("{reason} (polkit action {action})")) + }), + _ => Ok((Some(exe), None)), + }, + Ok(None) => Ok((None, None)), + Err(status) => Err(status), + }; + match outcome { + Ok((official, polkit)) => { info!( rpc, peer_uid = peer.uid, peer_pid = ?peer.pid, auth = if official.is_some() { "official" } else { "privileged" }, official_exe = ?official, + polkit = ?polkit, "authorized" ); Ok(peer) @@ -959,7 +987,13 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, "ApplyRules", Access::Control).await?; + let peer = self + .authorize( + &req, + "ApplyRules", + Access::Elevated(crate::polkit::IMPORT_RULES), + ) + .await?; self.apply_rules_checked(peer, req.into_inner()) .await .map(Response::new) @@ -1077,7 +1111,11 @@ impl Firewall for FirewallService { &self, req: Request, ) -> Result, Status> { - let peer = self.authorize(&req, "SetPaused", Access::Control).await?; + // Both directions: resuming is as much "the firewall now behaves + // differently" as pausing. + let peer = self + .authorize(&req, "SetPaused", Access::Elevated(crate::polkit::PAUSE)) + .await?; let msg = req.into_inner(); if !msg.paused { @@ -1558,6 +1596,7 @@ pub async fn spawn( own_uid: nix::unistd::geteuid().as_raw(), official_clients: opts.ipc.official_clients.clone().into(), official: crate::official::check, + polkit: |peer, action| Box::pin(crate::polkit::check(peer, action)), audience: Arc::new(PromptAudience::default()), resume_at_ms: Arc::new(AtomicI64::new(0)), pause_default_secs: opts.pause_default_secs, @@ -1655,7 +1694,32 @@ mod tests { } } + fn polkit_allows( + _: PeerId, + _: &'static str, + ) -> futures::future::BoxFuture<'static, Result<(), String>> { + Box::pin(async { Ok(()) }) + } + + fn polkit_denies( + _: PeerId, + _: &'static str, + ) -> futures::future::BoxFuture<'static, Result<(), String>> { + Box::pin(async { Err("authorization dialog dismissed".to_string()) }) + } + + fn polkit_must_not_be_asked( + _: PeerId, + action: &'static str, + ) -> futures::future::BoxFuture<'static, Result<(), String>> { + panic!("polkit was asked for {action}") + } + fn service(require_group: bool) -> FirewallService { + service_with(require_group, polkit_allows) + } + + fn service_with(require_group: bool, polkit: PolkitCheck) -> FirewallService { let store = RuleStore::open_in_memory().unwrap(); let policy: SharedPolicy = Arc::new(std::sync::RwLock::new(crate::config::DefaultPolicy { no_ui_action: cfc_core::Action::Deny, @@ -1684,6 +1748,7 @@ mod tests { own_uid: OWN_UID, official_clients: Arc::from(Vec::new()), official: stub_official, + polkit, audience: Arc::new(PromptAudience::default()), resume_at_ms: Arc::new(AtomicI64::new(0)), pause_default_secs: 600, @@ -1815,6 +1880,121 @@ mod tests { assert_eq!(changes(&svc, official_outsider).await, all); assert_eq!(changes(&svc, read_only_member).await, none); assert_eq!(changes(&svc, peer(1002, 1002, 6)).await, none); + + // Pause and import need polkit from the official client, and only + // from it: a refusal there leaves the single-rule edits alone. + let svc = service_with(true, polkit_denies); + assert_eq!( + changes(&svc, official_member).await, + [true, true, false, false] + ); + assert_eq!(changes(&svc, root).await, all); + assert_eq!(changes(&svc, own).await, all); + } + + #[tokio::test] + async fn polkit_is_never_asked_for_root_or_for_prompt_answers() { + let svc = service_with(true, polkit_must_not_be_asked); + for who in [peer(0, 0, 1), peer(OWN_UID, OWN_UID, 2)] { + assert_eq!(changes(&svc, who).await, [true; 4]); + } + let official = peer(1000, GROUP_GID, OFFICIAL_PID); + let [upsert, delete, ..] = changes_without_elevation(&svc, official).await; + assert!(upsert && delete); + let _ui = pending_prompt(&svc).await; + assert!( + svc.submit_verdict(answer(official)) + .await + .unwrap() + .into_inner() + .accepted + ); + } + + /// UpsertRule and DeleteRule only. + async fn changes_without_elevation(svc: &FirewallService, peer: PeerId) -> [bool; 2] { + let upsert = svc + .upsert_rule(request( + UpsertRuleRequest { + rule: Some(rule_pb()), + }, + peer, + )) + .await + .is_ok(); + let delete = svc + .delete_rule(request( + DeleteRuleRequest { + id: uuid::Uuid::new_v4().to_string(), + }, + peer, + )) + .await + .is_ok(); + [upsert, delete] + } + + #[tokio::test] + async fn a_denied_polkit_leaves_the_firewall_unpaused() { + let svc = service_with(true, polkit_denies); + let official = peer(1000, GROUP_GID, OFFICIAL_PID); + let status = svc + .set_paused(request( + SetPausedRequest { + paused: true, + duration_secs: 60, + }, + official, + )) + .await + .unwrap_err(); + assert_eq!(status.code(), tonic::Code::PermissionDenied); + assert!( + status.message().contains("dismissed"), + "{}", + status.message() + ); + assert!(status.message().contains(crate::polkit::PAUSE)); + assert!(!svc.stats.is_paused()); + + let status = svc + .apply_rules(request( + ApplyRulesRequest { + rules: vec![rule_pb()], + replace: true, + }, + official, + )) + .await + .unwrap_err(); + assert!(status.message().contains(crate::polkit::IMPORT_RULES)); + assert_eq!(svc.engine.rule_count(), 0, "the store is unchanged"); + } + + #[tokio::test] + async fn resume_also_requires_polkit() { + let svc = service_with(true, polkit_denies); + svc.set_paused(request( + SetPausedRequest { + paused: true, + duration_secs: 60, + }, + peer(0, 0, 1), + )) + .await + .unwrap(); + let status = svc + .set_paused(request( + SetPausedRequest { + paused: false, + duration_secs: 0, + }, + peer(1000, GROUP_GID, OFFICIAL_PID), + )) + .await + .unwrap_err(); + assert_eq!(status.code(), tonic::Code::PermissionDenied); + assert!(svc.stats.is_paused(), "still paused"); } #[tokio::test] diff --git a/crates/cfc-daemon/src/lib.rs b/crates/cfc-daemon/src/lib.rs index eb7a002..8aa9f2f 100644 --- a/crates/cfc-daemon/src/lib.rs +++ b/crates/cfc-daemon/src/lib.rs @@ -20,6 +20,7 @@ pub mod ipc; pub mod nfqueue; pub mod official; pub mod packet; +pub mod polkit; pub mod process_resolve; pub mod prompts; pub mod provenance; diff --git a/crates/cfc-daemon/src/polkit.rs b/crates/cfc-daemon/src/polkit.rs new file mode 100644 index 0000000..08e5b63 --- /dev/null +++ b/crates/cfc-daemon/src/polkit.rs @@ -0,0 +1,209 @@ +//! Administrator authorization through polkit. +//! +//! Pause, resume and rule import change the whole firewall at once, so even +//! the official app and tray must have them confirmed by an administrator: +//! the daemon asks polkit's `CheckAuthorization` for the calling process, +//! with user interaction allowed, and the user's polkit agent shows its +//! password dialog. The shipped policy +//! (`pkg/org.projectcolony.firewall.policy`) asks for `auth_admin_keep`, so +//! one password covers a few minutes. Root never gets here, and neither do +//! prompt answers or single-rule edits. +//! +//! One fresh system-bus connection per call: these calls are rare, and a +//! connection kept open would be one more thing to babysit across D-Bus +//! restarts. + +use crate::ipc::PeerId; +use std::collections::HashMap; +use std::sync::atomic::{AtomicU64, Ordering}; +use std::time::Duration; +use zbus::zvariant::Value; + +/// Pause or resume filtering (`SetPaused`, both directions). +pub const PAUSE: &str = "org.projectcolony.firewall.pause"; +/// Import, replace or bundle-install rules (`ApplyRules`). +pub const IMPORT_RULES: &str = "org.projectcolony.firewall.import-rules"; +/// How long the daemon waits for the user to answer the dialog. Clients wait +/// longer (`cfc_client::INTERACTIVE_TIMEOUT`), so this answer reaches them. +pub const TIMEOUT: Duration = Duration::from_secs(120); + +const SERVICE_UNKNOWN: &str = "org.freedesktop.DBus.Error.ServiceUnknown"; +/// `CheckAuthorizationFlags.AllowUserInteraction`. +const ALLOW_USER_INTERACTION: u32 = 1; + +#[zbus::proxy( + interface = "org.freedesktop.PolicyKit1.Authority", + default_service = "org.freedesktop.PolicyKit1", + default_path = "/org/freedesktop/PolicyKit1/Authority" +)] +trait Authority { + fn check_authorization( + &self, + subject: &(&str, HashMap<&str, Value<'_>>), + action_id: &str, + details: HashMap<&str, &str>, + flags: u32, + cancellation_id: &str, + ) -> zbus::Result<(bool, bool, HashMap)>; + + fn cancel_check_authorization(&self, cancellation_id: &str) -> zbus::Result<()>; +} + +/// Asks polkit whether `peer` may perform `action`, letting its agent ask +/// the user. `Err` carries the reason the client shows. +pub async fn check(peer: PeerId, action: &'static str) -> Result<(), String> { + static NEXT: AtomicU64 = AtomicU64::new(0); + let (Some(pid), Some(start_time)) = ( + peer.pid.and_then(|pid| u32::try_from(pid).ok()), + peer.starttime, + ) else { + return Err("the caller's process could not be identified for polkit".into()); + }; + let unreachable = |e: zbus::Error| { + format!( + "this needs administrator authorization, but the system D-Bus is unreachable \ + ({e}); use sudo cfc pause, resume or rules import" + ) + }; + let bus = zbus::connection::Builder::system() + .map_err(unreachable)? + .method_timeout(TIMEOUT + Duration::from_secs(10)) + .build() + .await + .map_err(unreachable)?; + let authority = AuthorityProxy::builder(&bus) + .cache_properties(zbus::proxy::CacheProperties::No) + .build() + .await + .map_err(|e| call_error(&e))?; + let subject = ( + "unix-process", + HashMap::from([ + ("pid", Value::from(pid)), + // Clock ticks since boot, the same field 22 polkit reads. + ("start-time", Value::from(start_time)), + ("uid", Value::from(peer.uid as i32)), + ]), + ); + let cancellation = format!("cfc-{pid}-{}", NEXT.fetch_add(1, Ordering::Relaxed)); + let call = authority.check_authorization( + &subject, + action, + HashMap::new(), + ALLOW_USER_INTERACTION, + &cancellation, + ); + match tokio::time::timeout(TIMEOUT, call).await { + Ok(Ok((authorized, challenge, details))) => outcome(authorized, challenge, &details), + Ok(Err(e)) => Err(call_error(&e)), + Err(_) => { + // Close the dialog the user did not answer. + let _ = authority.cancel_check_authorization(&cancellation).await; + Err(format!( + "authorization timed out after {} s", + TIMEOUT.as_secs() + )) + } + } +} + +/// What a `CheckAuthorization` answer means for the caller. +fn outcome( + authorized: bool, + challenge: bool, + details: &HashMap, +) -> Result<(), String> { + if authorized { + Ok(()) + } else if details.get("polkit.dismissed").map(String::as_str) == Some("true") { + Err("authorization dialog dismissed".into()) + } else if challenge { + Err( + "no polkit authentication agent answered in your session (start one, e.g. \ + hyprpolkitagent or polkit-gnome) or use sudo cfc" + .into(), + ) + } else { + Err("not authorized by polkit policy".into()) + } +} + +fn call_error(e: &zbus::Error) -> String { + let unknown = match e { + zbus::Error::MethodError(name, ..) => name.as_str() == SERVICE_UNKNOWN, + zbus::Error::FDO(fdo) => matches!(**fdo, zbus::fdo::Error::ServiceUnknown(_)), + _ => false, + }; + if unknown { + "polkit is not installed or not running; use sudo cfc pause, resume or rules import".into() + } else { + format!("polkit check failed: {e}") + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn outcome_mapping() { + let none = HashMap::new(); + assert_eq!(outcome(true, false, &none), Ok(())); + assert_eq!( + outcome(true, true, &none), + Ok(()), + "authorized wins over a stale challenge flag" + ); + let dismissed = HashMap::from([("polkit.dismissed".to_string(), "true".to_string())]); + assert_eq!( + outcome(false, true, &dismissed).unwrap_err(), + "authorization dialog dismissed" + ); + let no_agent = outcome(false, true, &none).unwrap_err(); + assert!(no_agent.starts_with("no polkit authentication agent answered")); + assert!(no_agent.contains("sudo cfc")); + assert_eq!( + outcome(false, false, &none).unwrap_err(), + "not authorized by polkit policy" + ); + } + + #[test] + fn service_unknown_maps_to_not_installed() { + let message = zbus::message::Message::method_call("/", "CheckAuthorization") + .unwrap() + .build(&()) + .unwrap(); + let unknown = zbus::Error::MethodError( + zbus::names::OwnedErrorName::try_from(SERVICE_UNKNOWN).unwrap(), + None, + message.clone(), + ); + assert!(call_error(&unknown).starts_with("polkit is not installed")); + let fdo = zbus::Error::FDO(Box::new(zbus::fdo::Error::ServiceUnknown("x".into()))); + assert!(call_error(&fdo).starts_with("polkit is not installed")); + let other = zbus::Error::MethodError( + zbus::names::OwnedErrorName::try_from("org.freedesktop.DBus.Error.AccessDenied") + .unwrap(), + None, + message, + ); + assert!(call_error(&other).starts_with("polkit check failed")); + } + + /// By hand, as a user with a polkit agent: expects a dialog. Without an + /// agent: expects the "no polkit authentication agent" message. + #[tokio::test] + #[ignore = "asks the real polkit on the system bus"] + async fn live_check_against_the_system_bus() { + let pid = std::process::id(); + let peer = PeerId { + uid: nix::unistd::getuid().as_raw(), + gid: nix::unistd::getgid().as_raw(), + pid: Some(pid as i32), + starttime: crate::process_resolve::read_starttime(pid), + sock_ino: None, + }; + println!("{:?}", check(peer, PAUSE).await); + } +} diff --git a/crates/cfc-proto/proto/cfc.proto b/crates/cfc-proto/proto/cfc.proto index 3d5cf92..d373fde 100644 --- a/crates/cfc-proto/proto/cfc.proto +++ b/crates/cfc-proto/proto/cfc.proto @@ -5,10 +5,19 @@ package cfc.v1; // Colony Firewall Control IPC schema. // Spoken between the root daemon (server) and UI/CLI clients over a Unix // domain socket. +// +// Authorization: reading (ListRules, GetStatus, ListEvents, the two streams) +// is open to any peer that can connect. Changes (SubmitVerdict, UpsertRule, +// DeleteRule, ApplyRules, SetPaused) are accepted from root and from the +// installed Colony Firewall app and tray only; every other peer gets +// PERMISSION_DENIED with the reason. ApplyRules and SetPaused from the app or +// tray also need a polkit authorization, which may keep the call waiting on +// a password dialog for up to 120 s. service Firewall { // The UI subscribes once and receives a streaming feed of new connections - // that need a verdict from the user. + // that need a verdict from the user. Any peer may subscribe; only one that + // may answer (root, the official app or tray) counts as a connected UI. rpc StreamPrompts(SubscribeRequest) returns (stream PromptEvent); // The UI returns a verdict for a specific prompt id. @@ -18,7 +27,8 @@ service Firewall { rpc ListRules(ListRulesRequest) returns (ListRulesResponse); rpc UpsertRule(UpsertRuleRequest) returns (UpsertRuleResponse); rpc DeleteRule(DeleteRuleRequest) returns (DeleteRuleResponse); - // Validate and apply a complete import atomically. + // Validate and apply a complete import atomically. From the app or tray + // it needs polkit action org.projectcolony.firewall.import-rules. rpc ApplyRules(ApplyRulesRequest) returns (ApplyRulesResponse); // Live observed connections (for the live view, no verdict requested). @@ -32,7 +42,8 @@ service Firewall { // flow no rule matched is allowed straight through instead of raising a // prompt. A pause always ends by itself: duration_secs (or the daemon's // [pause] default_secs when 0) is clamped to at most 24h, and the - // response reports the resume deadline. + // response reports the resume deadline. Pausing and resuming from the app + // or tray need polkit action org.projectcolony.firewall.pause. rpc SetPaused(SetPausedRequest) returns (SetPausedResponse); // Persisted verdict/audit log, newest first. Read-only. diff --git a/crates/cfc-tray/src/main.rs b/crates/cfc-tray/src/main.rs index 67895bc..81313f4 100644 --- a/crates/cfc-tray/src/main.rs +++ b/crates/cfc-tray/src/main.rs @@ -121,6 +121,8 @@ enum Cmd { OverflowResult { key: String, }, + /// A pause or resume request finished; `Err` is the reason to show. + PauseDone(Result<(), String>), } struct TrayApp { @@ -357,31 +359,38 @@ async fn refresh( } /// Ask the daemon to pause (`duration_secs`, 0 = daemon default) or -/// resume. Failures are logged, never fatal - the next poll will show the -/// truth either way. -async fn set_paused(client: &mut Option, socket: &Path, paused: bool, duration_secs: u32) { - let verb = if paused { "pause" } else { "resume" }; - if client.is_none() { - match Client::connect(socket).await { - Ok(c) => *client = Some(c), +/// resume, on a task of its own: the daemon asks polkit first, and the +/// password dialog may stay open for up to two minutes, during which the +/// main loop must keep showing prompts. The outcome comes back as +/// [`Cmd::PauseDone`]. +fn spawn_set_paused( + tx: mpsc::UnboundedSender, + socket: PathBuf, + paused: bool, + duration_secs: u32, +) { + tokio::spawn(async move { + let verb = if paused { "pause" } else { "resume" }; + let result = async { + let mut client = Client::connect_interactive(&socket).await?; + client.set_paused(paused, duration_secs).await + } + .await; + let _ = tx.send(Cmd::PauseDone(match result { + Ok(resp) => { + debug!( + paused = resp.paused, + resume_at_unix_ms = resp.resume_at_unix_ms, + "{verb} acknowledged" + ); + Ok(()) + } Err(e) => { - warn!("cannot {verb}: {e}"); - return; + warn!("{verb} failed: {e}"); + Err(model::pause_failed_body(verb, &e)) } - } - } - let c = client.as_mut().expect("connected above"); - match c.set_paused(paused, duration_secs).await { - Ok(resp) => debug!( - paused = resp.paused, - resume_at_unix_ms = resp.resume_at_unix_ms, - "{verb} acknowledged" - ), - Err(e) => { - warn!("{verb} failed: {e}"); - *client = None; - } - } + })); + }); } /// Launches the GUI, detached: resolved via PATH, environment (including @@ -967,6 +976,10 @@ async fn run(sealed: std::io::Result<()>) -> anyhow::Result<()> { // Notification wait tasks route their results over the same channel // as menu clicks; the main loop is the only place the client lives. let handle_tx = tx.clone(); + let pause_tx = tx.clone(); + // A pause or resume is waiting on the daemon (and maybe on a password + // dialog); further clicks are ignored until it answers. + let mut pause_in_flight = false; let tray = TrayApp { view: DaemonView::Connecting, tx, @@ -1050,16 +1063,24 @@ async fn run(sealed: std::io::Result<()>) -> anyhow::Result<()> { break; } Some(Cmd::OpenGui) => open_gui(), + Some(Cmd::Pause(_) | Cmd::Resume) if pause_in_flight => { + debug!("pause or resume already waiting; click ignored"); + } Some(Cmd::Pause(secs)) => { - set_paused(&mut client, &socket, true, secs).await; - // Refresh immediately so the menu flips to "Resume - // now" without waiting out the poll interval. - if !refresh(&handle, &mut client, &socket, &mut gate, &mut was_reachable, generic).await { - break; - } + pause_in_flight = true; + spawn_set_paused(pause_tx.clone(), socket.clone(), true, secs); } Some(Cmd::Resume) => { - set_paused(&mut client, &socket, false, 0).await; + pause_in_flight = true; + spawn_set_paused(pause_tx.clone(), socket.clone(), false, 0); + } + Some(Cmd::PauseDone(result)) => { + pause_in_flight = false; + if let Err(body) = result { + notify_brief(body); + } + // Refresh immediately so the menu flips to "Resume + // now" without waiting out the poll interval. if !refresh(&handle, &mut client, &socket, &mut gate, &mut was_reachable, generic).await { break; } diff --git a/crates/cfc-tray/src/model.rs b/crates/cfc-tray/src/model.rs index 6ccaed7..99b0e88 100644 --- a/crates/cfc-tray/src/model.rs +++ b/crates/cfc-tray/src/model.rs @@ -102,6 +102,17 @@ pub fn unreachable_hint(err: &ClientError) -> String { } } +/// Notification body for a pause or resume the daemon did not carry out. +/// A denial is the daemon's own reason (a dismissed password dialog, no +/// polkit agent, a tray that must be restarted), which already says what to +/// do; anything else is a plain failure. +pub fn pause_failed_body(verb: &str, err: &ClientError) -> String { + match err { + ClientError::Denied(reason) => format!("Could not {verb} the firewall: {reason}"), + other => format!("Could not {verb} the firewall ({other})"), + } +} + /// "2h 05m" / "5m 00s" / "42s". Negative input clamps to "0s". pub fn format_countdown(secs: i64) -> String { let s = secs.max(0); @@ -670,6 +681,27 @@ mod tests { } } + #[test] + fn a_denied_pause_says_why() { + let body = pause_failed_body( + "pause", + &ClientError::Denied( + "authorization dialog dismissed (polkit action org.projectcolony.firewall.pause)" + .into(), + ), + ); + assert_eq!( + body, + "Could not pause the firewall: authorization dialog dismissed \ + (polkit action org.projectcolony.firewall.pause)" + ); + let body = pause_failed_body("resume", &ClientError::StreamClosed); + assert!( + body.starts_with("Could not resume the firewall ("), + "{body}" + ); + } + // --- menu model --------------------------------------------------------- #[test] diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index fe2b4a7..26bee6b 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -103,6 +103,9 @@ pub struct App { /// Reconnect) or enforcement resuming (it replaces Resume). Pause stays /// disabled for [`PROMPT_ARM_MS`] after it (see [`App::pause_armed`]). pub pause_shown_at_ms: i64, + /// A pause or resume request is in flight, possibly waiting on the + /// administrator password dialog. Pause and Resume stay disabled. + pub pause_pending: bool, pub retry_at_ms: Option, /// Set when a gRPC stream drops; the badge shows "reconnecting" instead /// of the footer being rewritten every two seconds. @@ -587,6 +590,7 @@ impl App { retry_attempts: 0, retry_at_ms: None, pause_shown_at_ms: 0, + pause_pending: false, stream_trouble: false, now_ms: now_ms(), }; @@ -1084,13 +1088,17 @@ impl App { } Message::TogglePaused => { let current = self.status.as_ref().map(|s| s.paused).unwrap_or(false); - if !current && !self.pause_armed() { + if self.pause_pending || (!current && !self.pause_armed()) { return Task::none(); } + self.pause_pending = true; + self.log + .info("Waiting for administrator authorization...", self.now_ms); let socket = self.socket_path.clone(); Task::perform(set_paused(socket, !current), Message::PausedSet) } Message::PausedSet(Ok((paused, resume_at_unix_ms))) => { + self.pause_pending = false; if !paused { self.now_ms = now_ms(); self.pause_shown_at_ms = self.now_ms; @@ -1113,6 +1121,7 @@ impl App { Task::none() } Message::PausedSet(Err(e)) => { + self.pause_pending = false; self.log.error(format!("pause failed: {e}"), self.now_ms); Task::none() } @@ -1537,13 +1546,16 @@ impl App { if paused { button(text("Resume").size(12)) .padding([4, 14]) - .on_press(Message::TogglePaused) + .on_press_maybe((!self.pause_pending).then_some(Message::TogglePaused)) .style(iced::widget::button::primary) .into() } else { button(text("Pause").size(12)) .padding([4, 14]) - .on_press_maybe(self.pause_armed().then_some(Message::TogglePaused)) + .on_press_maybe( + (self.pause_armed() && !self.pause_pending) + .then_some(Message::TogglePaused), + ) .style(iced::widget::button::secondary) .into() } @@ -1705,9 +1717,12 @@ async fn delete_rule(path: PathBuf, id: String) -> Result<(String, bool), String } /// Returns `(paused, resume_at_unix_ms)`. `duration_secs = 0` lets the -/// daemon apply its configured default and report the real deadline. +/// daemon apply its configured default and report the real deadline. The +/// daemon asks polkit first, so this may wait on a password dialog. async fn set_paused(path: PathBuf, paused: bool) -> Result<(bool, i64), String> { - let mut client = Client::connect(&path).await.map_err(|e| e.to_string())?; + let mut client = Client::connect_interactive(&path) + .await + .map_err(|e| e.to_string())?; let resp = client .set_paused(paused, 0) .await @@ -2564,6 +2579,22 @@ mod tests { assert_eq!(app.update(Message::TogglePaused).units(), 1); } + #[test] + fn pause_waits_for_authorization_and_ignores_clicks_meanwhile() { + let (mut app, _) = App::new(); + app.status = Some(proto::StatusResponse::default()); + app.now_ms += PROMPT_ARM_MS; + assert_eq!(app.update(Message::TogglePaused).units(), 1); + assert!(app.pause_pending); + assert_eq!(app.update(Message::TogglePaused).units(), 0, "in flight"); + let _ = app.update(Message::PausedSet(Err( + "authorization dialog dismissed (polkit action org.projectcolony.firewall.pause)" + .into(), + ))); + assert!(!app.pause_pending); + assert_eq!(app.update(Message::TogglePaused).units(), 1); + } + #[test] fn pause_ignores_a_click_right_after_reconnecting() { let (mut app, _) = App::new(); From 55743f68570e4904ba17cc8f04331aec74e31a99 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:50:05 +0200 Subject: [PATCH 111/125] build(packaging): ship the polkit policy for pause and rule import pkg/org.projectcolony.firewall.policy declares org.projectcolony.firewall.pause and org.projectcolony.firewall.import-rules (auth_admin_keep). Every channel installs it to /usr/share/polkit-1/actions: both PKGBUILDs (polkit as an optdepend), the RPM spec (Recommends: polkit, co-owned directories) and the tarball installer through colony.json, which release.yml now stages. The SELinux module (0.3.0) lets the daemon talk to polkit on the system bus and stat the binaries and libraries the official-client check judges; RHEL CI validates the policy XML. The pacman upgrade note tells 0.7 users to restart the app and tray. --- .github/workflows/release.yml | 1 + .github/workflows/rhel.yml | 12 ++++++ packaging/rpm/colony-firewall-control.spec | 8 ++++ packaging/selinux/TESTING.md | 2 + packaging/selinux/colony_firewall.te | 43 +++++++++++++++++++++- pkg/PKGBUILD | 6 +++ pkg/PKGBUILD-git | 6 +++ pkg/README.md | 14 +++++-- pkg/colony-firewall-control.install | 16 +++++++- pkg/colony.json | 5 ++- pkg/org.projectcolony.firewall.policy | 35 ++++++++++++++++++ systemd/colony-firewalld.service | 7 ++++ 12 files changed, 146 insertions(+), 9 deletions(-) create mode 100644 pkg/org.projectcolony.firewall.policy diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 64abd5a..6407f9e 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -192,6 +192,7 @@ jobs: pkg/colony-firewall-autostart.desktop \ pkg/colony-firewall-tray-autostart.desktop \ pkg/colony-firewall.svg \ + pkg/org.projectcolony.firewall.policy \ "${STAGE}/" # The kernel-side object. Fatal if absent: install.sh cannot express diff --git a/.github/workflows/rhel.yml b/.github/workflows/rhel.yml index cbaa4a2..81c72e1 100644 --- a/.github/workflows/rhel.yml +++ b/.github/workflows/rhel.yml @@ -83,6 +83,18 @@ jobs: make -f /usr/share/selinux/devel/Makefile colony_firewall.pp ls -l colony_firewall.pp + - name: Validate the polkit policy + run: | + set -eux + dnf -y install polkit libxml2 + dtd="$(find / -xdev -name policyconfig-1.dtd 2>/dev/null | head -n1)" + if [ -n "${dtd}" ]; then + xmllint --noout --nonet --dtdvalid "${dtd}" pkg/org.projectcolony.firewall.policy + else + echo "::warning::no policyconfig-1.dtd in this image; checking well-formedness only" + xmllint --noout --nonet pkg/org.projectcolony.firewall.policy + fi + - name: Show what the source declares # Nothing in this job LOADS the module: a container has no policy # store for `semodule` to install into, and `sedismod`/`sedispol` are diff --git a/packaging/rpm/colony-firewall-control.spec b/packaging/rpm/colony-firewall-control.spec index 6ddcd5c..cde4bd3 100644 --- a/packaging/rpm/colony-firewall-control.spec +++ b/packaging/rpm/colony-firewall-control.spec @@ -44,6 +44,9 @@ Requires(postun): systemd Recommends: libxkbcommon Recommends: wayland Suggests: libnotify +# The app and tray ask for an administrator password through polkit before +# they pause, resume or import rules; without it only sudo cfc can. +Recommends: polkit %description Colony Firewall Control asks before a program is allowed to reach the network, @@ -135,6 +138,8 @@ install -Dpm 0644 pkg/colony-firewall-tray-autostart.desktop \ %{buildroot}%{_sysconfdir}/xdg/autostart/colony-firewall-tray.desktop install -Dpm 0644 pkg/colony-firewall.svg \ %{buildroot}%{_datadir}/icons/hicolor/scalable/apps/colony-firewall.svg +install -Dpm 0644 pkg/org.projectcolony.firewall.policy \ + %{buildroot}%{_datadir}/polkit-1/actions/org.projectcolony.firewall.policy # Where the eBPF object goes if one is installed later. Shipping the directory # means the loader's ownership check (root-owned, unwritable by anyone else) @@ -248,6 +253,9 @@ fi %{_sysconfdir}/xdg/autostart/colony-firewall.desktop %{_sysconfdir}/xdg/autostart/colony-firewall-tray.desktop %{_datadir}/icons/hicolor/scalable/apps/colony-firewall.svg +%dir %{_datadir}/polkit-1 +%dir %{_datadir}/polkit-1/actions +%{_datadir}/polkit-1/actions/org.projectcolony.firewall.policy %{_datadir}/bash-completion/completions/cfc %{_datadir}/zsh/site-functions/_cfc %{_datadir}/fish/vendor_completions.d/cfc.fish diff --git a/packaging/selinux/TESTING.md b/packaging/selinux/TESTING.md index 9cc3816..3bd2905 100644 --- a/packaging/selinux/TESTING.md +++ b/packaging/selinux/TESTING.md @@ -112,6 +112,8 @@ chance to be needed. | /proc attribution walk | `curl` from a second user account; the prompt must name curl's real path and pid | every prompt says `exe= pid=0`; AVCs from `domain_read_all_domains_state` targets. Enforcing, this is the outage mode: no exe rule can ever match | | rpm provenance | automatic: one `rpm -qa` at startup and after any `dnf install`. Install any small package, wait ~2 minutes, then check a prompt or `cfc log` shows package names | everything reports `Unpackaged` plus one provenance warning in the journal; AVC on `rpm_exec_t` or `rpm_var_lib_t` | | control socket, unconfined client | `cfc status` and `cfc rules list` as a normal logged-in user in the `colony-firewall` group (not root, not sudo) | connection refused/denied; AVC with the client's domain (`unconfined_t`) and `colony_firewall_runtime_t` | +| official clients | run the installed `colony-firewall` GUI as a group member, then delete a rule from it; then run `cfc rules remove ` without sudo, which must be refused with "read-only access" | the GUI's change is refused with a reason naming a stat or "unix_diag"; AVCs with `getattr` on `bin_t`/`lib_t`/`usr_t` or on `netlink_tcpdiag_socket` | +| polkit over D-Bus | press Pause in the GUI: your polkit agent must show "Authentication is required to pause or resume the firewall"; cancel it, then press Pause again and authenticate | "system D-Bus is unreachable" or "polkit check failed" instead of the dialog; AVCs on `system_dbusd_t` (`unix_stream_socket connectto`, `dbus send_msg`) or `policykit_t` | | sqlite WAL in /var/lib | answer any prompt with a persistent choice (**a**, then `3`=always), then `ls /var/lib/colony-firewall/` - `rules.db-wal` and `rules.db-shm` must exist while the daemon runs | the `map` denial is the quiet one: no error anywhere, just journal-mode SQLite and a 2.5x write regression. An AVC with class `file` permission `map` on `colony_firewall_var_lib_t` is the tell | Let the daemon run for at least a few minutes of normal use - browse diff --git a/packaging/selinux/colony_firewall.te b/packaging/selinux/colony_firewall.te index bdf0f94..5725824 100644 --- a/packaging/selinux/colony_firewall.te +++ b/packaging/selinux/colony_firewall.te @@ -1,4 +1,4 @@ -policy_module(colony_firewall, 0.2.0) +policy_module(colony_firewall, 0.3.0) ######################################## # @@ -23,7 +23,12 @@ policy_module(colony_firewall, 0.2.0) # * runs rpm(8) once per package-database generation, for provenance; # * runs nft(8) once at start, to flush a legacy Fast Allow set, and once a # minute, to ask whether the filtering table is loaded; -# * serves a unix socket under /run/colony-firewall to the CLI, tray and GUI. +# * serves a unix socket under /run/colony-firewall to the CLI, tray and GUI, +# and checks that a non-root client is the installed app or tray: it stats +# the client's image and every executable file it mapped, and asks +# sock_diag (UNIX_DIAG) for the client end of the connection; +# * asks polkit over the system bus before the app or tray may pause, +# resume or import rules. # # Every one of those is a separate way to be denied, and the failure modes # differ: a denied bpf() degrades to sock_diag + /proc and the firewall keeps @@ -248,6 +253,40 @@ optional_policy(` allow colony_firewalld_t rpm_var_lib_t:file read_file_perms; ') +######################################## +# +# Official clients and polkit +# +# Before a non-root client may change anything, the daemon checks it is the +# installed app or tray: it stats that binary and every directory above it, +# and every executable file the client mapped (binaries and libraries), to +# prove each is root-owned and unwritable by anyone else. The /proc reads +# (maps, fd, ns, status, and /proc/1/ns) are covered by the attribution +# group above; the UNIX_DIAG query by the netlink_tcpdiag_socket rule (it is +# the same NETLINK_SOCK_DIAG socket). +# +# A denial here makes the app and tray read-only, with the reason in their +# error message; root (sudo cfc) keeps working. +# +######################################## + +corecmd_getattr_all_executables(colony_firewalld_t) +corecmd_search_bin(colony_firewalld_t) +files_list_root(colony_firewalld_t) +files_search_usr(colony_firewalld_t) +libs_search_lib(colony_firewalld_t) + +# CheckAuthorization on org.freedesktop.PolicyKit1. Denied, pause, resume and +# rule import from the app or tray fail with "system D-Bus is unreachable" or +# "polkit check failed"; sudo cfc still works. +optional_policy(` + dbus_system_bus_client(colony_firewalld_t) + + optional_policy(` + policykit_dbus_chat(colony_firewalld_t) + ') +') + ######################################## # # Own files diff --git a/pkg/PKGBUILD b/pkg/PKGBUILD index 1f2449e..a90fd49 100644 --- a/pkg/PKGBUILD +++ b/pkg/PKGBUILD @@ -35,6 +35,7 @@ optdepends=( 'libbpf: mandatory interface filters for application confinement' 'libnotify: desktop notifications on new prompts' 'libx11: run the GUI in X11/XWayland sessions' + 'polkit: administrator prompt when the app or tray pauses filtering or imports rules' ) backup=('etc/colony-firewall/daemon.toml') install=colony-firewall-control.install @@ -127,6 +128,11 @@ package() { install -Dm644 pkg/colony-firewall.svg \ "$pkgdir/usr/share/icons/hicolor/scalable/apps/colony-firewall.svg" + # polkit actions the daemon asks about when the app or tray pauses, + # resumes or imports rules. + install -Dm644 pkg/org.projectcolony.firewall.policy \ + "$pkgdir/usr/share/polkit-1/actions/org.projectcolony.firewall.policy" + # Docs & license install -Dm644 README.md "$pkgdir/usr/share/doc/$pkgname/README.md" install -Dm644 docs/ARCHITECTURE.md "$pkgdir/usr/share/doc/$pkgname/ARCHITECTURE.md" diff --git a/pkg/PKGBUILD-git b/pkg/PKGBUILD-git index b6e4add..5f6ec58 100644 --- a/pkg/PKGBUILD-git +++ b/pkg/PKGBUILD-git @@ -29,6 +29,7 @@ optdepends=( 'libbpf: mandatory interface filters for application confinement' 'libnotify: desktop notifications on new prompts' 'libx11: run the GUI in X11/XWayland sessions' + 'polkit: administrator prompt when the app or tray pauses filtering or imports rules' ) provides=('colony-firewall-control') conflicts=('colony-firewall-control') @@ -126,6 +127,11 @@ package() { install -Dm644 pkg/colony-firewall.svg \ "$pkgdir/usr/share/icons/hicolor/scalable/apps/colony-firewall.svg" + # polkit actions the daemon asks about when the app or tray pauses, + # resumes or imports rules. + install -Dm644 pkg/org.projectcolony.firewall.policy \ + "$pkgdir/usr/share/polkit-1/actions/org.projectcolony.firewall.policy" + # Docs & license install -Dm644 README.md "$pkgdir/usr/share/doc/$pkgname/README.md" install -Dm644 docs/ARCHITECTURE.md "$pkgdir/usr/share/doc/$pkgname/ARCHITECTURE.md" diff --git a/pkg/README.md b/pkg/README.md index 19cf85f..ae039d4 100644 --- a/pkg/README.md +++ b/pkg/README.md @@ -40,7 +40,13 @@ Key design points: managers and later external ruleset flushes. - **`colony-firewall.sysusers`** creates the `colony-firewall` group used to gate access to the daemon's gRPC UNIX socket. Users join with - `usermod -aG colony-firewall `. + `usermod -aG colony-firewall `. Membership gives read access and + lets the installed app and tray connect; changes come from those two + programs or from `sudo cfc`. +- **`org.projectcolony.firewall.policy`** declares the polkit actions the + daemon asks about when the app or tray pauses, resumes or imports rules + (`auth_admin_keep`). Installed to `/usr/share/polkit-1/actions/`; polkit + is an optional dependency, and without it only `sudo cfc` can do those. - **XDG autostart** launches the GUI in every desktop session so prompts actually reach the user. Per-user opt-out: copy the file to `~/.config/autostart/` and set `Hidden=true`. @@ -224,9 +230,9 @@ The version in the manifest's `asset` filename is checked by Follow the Manual section of the top-level `README.md`, then its First run section: enable enforcement, then seed the starter rules with -`sudo cfc rules bootstrap-defaults` right away. `sudo` matters there, because -group membership from `usermod -aG colony-firewall` only applies after a new -login, and until the rules exist, unmatched DHCP, DNS and NTP flows are denied. +`sudo cfc rules bootstrap-defaults` right away. `sudo` is required there: a +non-root `cfc` is read-only, whatever its groups, and until the rules exist, +unmatched DHCP, DNS and NTP flows are denied. ## Uninstall behavior (all channels) diff --git a/pkg/colony-firewall-control.install b/pkg/colony-firewall-control.install index ea0fde7..eb0421d 100644 --- a/pkg/colony-firewall-control.install +++ b/pkg/colony-firewall-control.install @@ -16,11 +16,15 @@ post_install() { DHCP, DNS and NTP flows are denied): sudo cfc rules bootstrap-defaults - 3. Let your desktop user talk to the daemon socket: + 3. Let your desktop session read the daemon and the Colony Firewall + app and tray connect to it: usermod -aG colony-firewall then log out/in. The 'colony-firewall' group is created by systemd-sysusers from /usr/lib/sysusers.d/colony-firewall.conf - (applied automatically by the pacman sysusers hook). + (applied automatically by the pacman sysusers hook). Changes come + from the app and tray (pause, resume and rule import ask for an + administrator password through polkit) or from sudo cfc; any other + program of yours, including cfc without sudo, is read-only. ==> The GUI autostarts for all desktop users via /etc/xdg/autostart/colony-firewall.desktop; per-user opt-out: copy it @@ -45,6 +49,14 @@ post_upgrade() { echo "Firewall rules could not be refreshed; reload colony-firewall-nft and inspect the journal before relying on filtering." >&2 return 1 } + if [ "$(vercmp "$2" 0.8.0)" -lt 0 ]; then + cat <<'EOF' +==> 0.8.0: restart Colony Firewall and its tray now; until then they are + read-only (the daemon trusts only the installed binaries). Pause/resume + and rule import ask for an administrator password through polkit. + cfc without sudo is read-only: use sudo cfc for changes. +EOF + fi cat <<'EOF' ==> colony-firewall-control upgraded. Changes: /usr/share/doc/colony-firewall-control/README.md and diff --git a/pkg/colony.json b/pkg/colony.json index b89f219..ed505f4 100644 --- a/pkg/colony.json +++ b/pkg/colony.json @@ -28,6 +28,7 @@ "install -D -m 0644 colony-firewall-autostart.desktop /etc/xdg/autostart/colony-firewall.desktop", "install -D -m 0644 colony-firewall-tray-autostart.desktop /etc/xdg/autostart/colony-firewall-tray.desktop", "install -D -m 0644 colony-firewall.svg /usr/share/icons/hicolor/scalable/apps/colony-firewall.svg", + "install -D -m 0644 org.projectcolony.firewall.policy /usr/share/polkit-1/actions/org.projectcolony.firewall.policy", "install -D -m 0644 cfc-ebpf.o /usr/lib/colony-firewall/cfc-ebpf.o", "systemd-sysusers", "systemctl daemon-reload", @@ -46,6 +47,7 @@ "rm -f /usr/lib/colony-firewall/inbound-lockout-guard.sh", "rm -f /usr/share/applications/colony-firewall.desktop /etc/xdg/autostart/colony-firewall.desktop /etc/xdg/autostart/colony-firewall-tray.desktop", "rm -f /usr/share/icons/hicolor/scalable/apps/colony-firewall.svg", + "rm -f /usr/share/polkit-1/actions/org.projectcolony.firewall.policy", "rm -f /usr/lib/colony-firewall/cfc-ebpf.o", "rmdir --ignore-fail-on-non-empty /usr/lib/colony-firewall || true", "systemctl daemon-reload" @@ -60,6 +62,7 @@ ], "optionalDepends": { "libnotify": "desktop notifications", - "libx11": "run the GUI in X11/XWayland sessions" + "libx11": "run the GUI in X11/XWayland sessions", + "polkit": "administrator prompt when the app or tray pauses filtering or imports rules" } } diff --git a/pkg/org.projectcolony.firewall.policy b/pkg/org.projectcolony.firewall.policy new file mode 100644 index 0000000..8d351bc --- /dev/null +++ b/pkg/org.projectcolony.firewall.policy @@ -0,0 +1,35 @@ + + + + + + + Project Colony + https://github.com/Project-Colony/Colony-Firewall-Control + colony-firewall + + + Pause or resume Colony Firewall filtering + Authentication is required to pause or resume the firewall + + auth_admin_keep + auth_admin_keep + auth_admin_keep + + + + + Import or replace Colony Firewall rules + Authentication is required to import or replace the firewall rules + + auth_admin_keep + auth_admin_keep + auth_admin_keep + + + diff --git a/systemd/colony-firewalld.service b/systemd/colony-firewalld.service index 9a193fe..d68bad5 100644 --- a/systemd/colony-firewalld.service +++ b/systemd/colony-firewalld.service @@ -242,6 +242,13 @@ UMask=0077 # AF_NETLINK carries NFQUEUE, sock_diag and nftables; AF_INET/AF_INET6 the # Reject raw sockets; AF_UNIX the control socket. Nothing opens a packet # socket, and AF_PACKET with CAP_NET_RAW would sniff or inject below netfilter. +# +# AF_UNIX also carries the polkit check (pause, resume and rule import from +# the app or tray) to the system bus, and no other directive is needed for +# it: connect(2) to /run/dbus/system_bus_socket is not blocked by +# ProtectSystem=strict (a read-only mount does not stop connecting to a +# socket inode), sendmsg/recvmsg are in @system-service, and the connection +# is opened lazily on the first such request, so no After=dbus ordering. RestrictAddressFamilies=AF_UNIX AF_INET AF_INET6 AF_NETLINK [Install] From 3ea6b7eef10aff0dda283998d4711a38813088c9 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:53:32 +0200 Subject: [PATCH 112/125] docs: describe the read-only, official-client and polkit control model README gains a who-can-change-what table and sudo in every write example. HARDENING rewrites the control-socket section: read-only peers, how the installed app and tray are recognised, polkit for pause, resume and import, what is still trusted (code inside the official app, synthetic X11 input) and the setgid follow-up. TROUBLESHOOTING explains each refusal reason and each polkit outcome. ARCHITECTURE, SECURITY, TODO, the smoke test and stale code comments follow, and the changelog lists the breaking changes with the 0.7.0 upgrade steps. --- CHANGELOG.md | 37 +++++++++ README.md | 51 ++++++++---- SECURITY.md | 9 +- TODO.md | 1 + crates/cfc-daemon/src/config.rs | 2 +- crates/cfc-daemon/src/convert.rs | 5 +- crates/cfc-daemon/src/ipc.rs | 4 +- docs/ARCHITECTURE.md | 47 +++++++---- docs/HARDENING.md | 138 +++++++++++++++++++++++-------- docs/TROUBLESHOOTING.md | 79 ++++++++++++++++-- scripts/smoke-test.sh | 8 +- 11 files changed, 295 insertions(+), 86 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index a25d605..8728f35 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -66,6 +66,43 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Security +- **Breaking: only root and the installed app and tray may change the + firewall.** Group membership no longer grants control. A non-root peer may + answer prompts, add, edit or delete rules, pause, resume or import only when + its process runs the installed, root-sealed `/usr/bin/colony-firewall` or + `/usr/bin/colony-firewall-tray` (by device and inode), sealed itself at + startup (inherited descriptors closed, non-dumpable), holds the connection + itself, is not traced, runs in the host namespaces and mapped no executable + file from outside root-owned directories. Every other program of the + desktop user, a non-root `cfc` included, is read-only: its changes are + refused with the reason (exit 1 in `cfc`), its prompt subscription does not + count as a connected UI and it cannot answer prompts. A peer with the + daemon's own uid keeps full control (root in production). New + `[ipc] official_clients` key; `require_group` now waives the group check + for the app and tray only. See docs/HARDENING.md for what this still + trusts. +- **Breaking: pause, resume and rule import from the app or tray ask for an + administrator password** through polkit + (`org.projectcolony.firewall.pause`, `org.projectcolony.firewall.import-rules`, + `auth_admin_keep`). The policy file is installed by every package; polkit + is an optional dependency. Root is never asked, and answering a prompt or + editing a rule never asks. The tray no longer blocks its prompt + notifications while the dialog is open. + + Upgrading from 0.7.0: + 1. Scripts that ran `cfc` as a regular user to change rules, pause or + answer prompts must use `sudo cfc`. Reading (`status`, `rules list`, + `log`, `live`) is unchanged. + 2. Restart Colony Firewall and its tray after the upgrade. The 0.7 + processes, and any process still running a replaced binary, are + read-only until relaunched. + 3. Pausing from the app or tray needs a polkit agent in the session + (most desktops run one; on Hyprland, `hyprpolkitagent`). Without one + the request is refused with that reason and `sudo cfc pause` works. + 4. A non-root `cfc prompts` no longer counts as a UI: on a headless machine + run `sudo cfc prompts`, or `no_ui_action` applies. + 5. A user-wide `LD_PRELOAD` or a Vulkan layer loaded from your home + directory makes the app read-only; the refusal names the library. - `cfc prompts`: keys typed while no prompt was shown, such as an answer typed just as a prompt expired, answered the next prompt as soon as it was printed, and an arrow key skipped one prompt and left `A` or `D` to answer diff --git a/README.md b/README.md index 616d32c..cbcea35 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,7 @@ NFQUEUE in the kernel, per-app pop-ups in iced, gRPC IPC over a Unix socket. - iced GUI with parchment / burgundy theme, four tabs (Prompts / Rules / Live / Stats), a countdown on every prompt, and desktop notifications when the window is hidden -- **Answer prompts from a terminal** (`cfc prompts`) - headless servers +- **Answer prompts from a terminal** (`sudo cfc prompts`) - headless servers and SSH sessions are not second-class citizens - **Persistent verdict log** (`cfc log`): what did this app contact, and what did we do about it @@ -167,7 +167,9 @@ sudo install -Dm644 pkg/colony-firewall-tray-autostart.desktop \ sudo systemctl daemon-reload # The control socket is root:colony-firewall 0660. Join the group, then -# log out and back in, or the GUI and cfc get "permission denied". +# log out and back in, or the GUI, the tray and cfc get "permission denied". +# Membership gives read access and lets the installed app and tray connect; +# changes come from those two or from sudo cfc (see Who can change what). sudo usermod -aG colony-firewall "$USER" ``` @@ -200,9 +202,9 @@ without prompting: sudo cfc rules bootstrap-defaults # same as: cfc rules bundle add system ``` -(`sudo` because group membership from `usermod -aG colony-firewall` only -takes effect in a new login session. After logging out and back in, plain -`cfc` works.) +(`sudo` because `cfc` run as a regular user is read-only: it can show status, +rules and the logs, but only root, or the installed Colony Firewall app and +tray, can change the firewall.) This installs twelve allow rules - systemd-resolved DNS (:53), systemd-timesyncd and chronyd NTP (:123/udp), the DHCP clients (dhcpcd, @@ -244,7 +246,7 @@ colony-firewall On a headless machine, answer them from the terminal instead: ```sh -cfc prompts +sudo cfc prompts ``` With no subscriber at all the daemon applies `no_ui_action` to unmatched @@ -252,7 +254,9 @@ remote flows without asking anyone. **That is a denial under every profile.** "Nobody is connected" is a permanent condition on a headless box, not a passing one, and answering it with an allow would mean those hosts had no outbound firewall whatsoever. Stored rules are what such a -machine runs on; `cfc prompts` is how you add more without a GUI. +machine runs on; `sudo cfc prompts` is how you add more without a GUI. A +`cfc prompts` without sudo only watches: the daemon neither counts it as a UI +nor accepts its answers. This does not refuse inbound SSH: the ruleset hooks `output` on `ct state new` only, so an inbound SSH session's replies are @@ -414,14 +418,14 @@ cfc status # Answer prompts from this terminal - no GUI needed. # a=allow d=deny r=reject s=skip q=quit, then duration and scope. -cfc prompts +sudo cfc prompts -# Add a rule from the command line -cfc rules add --action allow --exe /usr/bin/curl --dst-port 443 +# Add a rule from the command line (changes need sudo; reading does not) +sudo cfc rules add --action allow --exe /usr/bin/curl --dst-port 443 # Rules take an id, a unique id prefix, or the rule's name cfc rules show curl-https -cfc rules disable 3f2a +sudo cfc rules disable 3f2a # Watch traffic decisions in real time (colorized), with filters cfc live --denied @@ -431,18 +435,33 @@ cfc live --exe firefox --follow cfc log --since 24h cfc log --exe firefox --action deny -# Pause enforcement for a bounded window (the daemon auto-resumes) -cfc pause --for 30m -cfc resume +# Pause enforcement for a bounded window (the daemon auto-resumes). +# From the app or tray, Pause asks for an administrator password instead. +sudo cfc pause --for 30m +sudo cfc resume # Back up rules cfc rules export --out rules.json # Migrate from an existing opensnitch install. A rule with no equivalent # here (hostname, regexp) stops it; --skip-unconvertible imports the rest. -cfc rules import-opensnitch /etc/opensnitchd/rules +sudo cfc rules import-opensnitch /etc/opensnitchd/rules ``` +### Who can change what + +| who | read status, rules, logs, live view, prompts | answer prompts, add/edit/delete rules | pause, resume, import rules | +|---|---|---|---| +| root (`sudo cfc`) | yes | yes | yes | +| the installed Colony Firewall app and tray, run by a `colony-firewall` group member | yes | yes | after an administrator password (polkit, kept a few minutes) | +| any other program of a group member, including `cfc` without sudo | yes | no | no | + +The daemon recognises the app and tray by the running image: it must be the +installed, root-owned `/usr/bin/colony-firewall` or `colony-firewall-tray`, +started normally (not traced, no library preloaded from your files). After +an upgrade, restart both; until then they are read-only. Details and limits +are in [docs/HARDENING.md](docs/HARDENING.md). + Executable rules require the canonical mapped target explicitly. An alias such as `/bin/tool` on a system where `/bin` links to `/usr/bin` is refused; review and name `/usr/bin/tool` instead. Rules remain attached to that fixed @@ -499,7 +518,7 @@ Every profile denies on timeout: a prompt you were shown and did not answer must not become an allow. The profiles differ in how long they wait, and in what happens when there is nobody subscribed to ask. -Use `strict` only when you always have the UI running (or `cfc prompts`), +Use `strict` only when you always have the UI running (or `sudo cfc prompts`), otherwise you lose network when the daemon starts before a subscriber does (fail-closed posture). diff --git a/SECURITY.md b/SECURITY.md index 560efde..a619588 100644 --- a/SECURITY.md +++ b/SECURITY.md @@ -26,13 +26,18 @@ form for this repository: Please do **not** open a public issue for anything you believe is exploitable (privilege escalation via the daemon, rule-bypass of the NFQUEUE filter, -crafted-packet parsing crashes, socket permission problems, etc.). +crafted-packet parsing crashes, socket permission problems, a non-root program +other than the installed app and tray getting a firewall change accepted, +etc.). Out of scope: the limits documented in [What this firewall does not protect against](docs/HARDENING.md#what-this-firewall-does-not-protect-against), such as root processes, established or inherited flows, replies on inbound-initiated connections while inbound filtering is off, local relays, -AF_PACKET and raw sockets, and code running as an allowed program's user. +AF_PACKET and raw sockets, and code running as an allowed program's user, +and the residual risks listed under +[The control socket and who can talk to it](docs/HARDENING.md#the-control-socket-and-who-can-talk-to-it) +(code already running inside the official app or tray). A way around the firewall that this list does not describe, or a description that turns out to be wrong, is in scope. diff --git a/TODO.md b/TODO.md index 073adc8..62611cd 100644 --- a/TODO.md +++ b/TODO.md @@ -253,6 +253,7 @@ What defeats it completely: | **DNS tunnelling** | the resolver must be allowed for anything to work. CFC *observes* answers; it does not inspect or block queries. | | **Inherited or passed socket descriptors** | Existing connection authorization is not rechecked for each sending executable; socket attribution is ambiguous when ownership is shared. | | **CAP_NET_RAW packet sockets** | Packet-layer egress can bypass the IP OUTPUT hook. Layer-2 confinement is outside the shipped rules. | +| **Code inside the official app or tray** | the daemon accepts changes only from root and the installed, sealed app and tray, checking the running process (image, prologue, connection, tracer, file-backed executable mappings). Code already running inside them is not seen: a self-unmapping `LD_PRELOAD` payload living in anonymous memory, or synthetic X11/XWayland input clicking the GUI. Setgid binaries trusted by connect-time gid would close the first; not done. | | **Prompt fatigue** | demonstrated on this machine: ten Firefox prompts in a row, all denied, browser lost. A malicious installer generating thirty prompts trains the user to click Allow. | And one tradeoff worth stating plainly: the ruleset is **fail-closed for diff --git a/crates/cfc-daemon/src/config.rs b/crates/cfc-daemon/src/config.rs index 045154c..bb73ad1 100644 --- a/crates/cfc-daemon/src/config.rs +++ b/crates/cfc-daemon/src/config.rs @@ -270,7 +270,7 @@ impl Default for EventsConfig { pub struct IpcConfig { /// Unix group granted access to the control socket. After bind the /// daemon chowns the socket to `root:` and chmods it 0660, so - /// group membership *is* the access check. + /// group membership is what lets a process connect at all. pub group: String, /// Require proved membership of `group` before an official client (the /// installed app or tray) may change anything. Setting this to false diff --git a/crates/cfc-daemon/src/convert.rs b/crates/cfc-daemon/src/convert.rs index a5db97b..ba52eef 100644 --- a/crates/cfc-daemon/src/convert.rs +++ b/crates/cfc-daemon/src/convert.rs @@ -240,7 +240,8 @@ pub fn scope_to_pb(s: &RuleScope) -> pb::RuleScope { /// `.and_then(|n| IpNet::from_str(&n).ok())`, so a typo'd CIDR became `None` - /// turning an Allow scoped `exe + 10.0.0.0/8` into an Allow scoped `exe`, i.e. /// "this program may reach anywhere". A client's own validation is not a -/// substitute: `UpsertRule` accepts whatever any group member sends. +/// substitute: `UpsertRule` accepts whatever the calling app, tray or root +/// CLI sends. /// /// Three fields can fail: `dst_net`, `protocol` and `dst_port`. The last is the /// least obvious and was missed on the first pass - the wire type is `uint32` @@ -755,7 +756,7 @@ mod tests { fn a_malformed_dst_net_is_refused_not_dropped() { // The whole point. Dropping it to None turned "this program may reach // 10.0.0.0/8" into "this program may reach anywhere" - a silent - // widening of policy on the path any group member can reach. + // widening of policy on the path every rule write goes through. let mut pb = scope_to_pb(&RuleScope::any()); pb.exe_path = "/usr/bin/curl".into(); pb.dst_net = "10.0.0.0/33".into(); diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 95dd03d..f3ccd04 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -89,8 +89,8 @@ use tracing::{info, warn}; /// Validates a rule's explicit executable target without blocking IPC. /// -/// `canonicalize` is a synchronous syscall on a path any `colony-firewall` -/// group member supplies, and this runs inside a `#[tonic::async_trait]` +/// `canonicalize` is a synchronous syscall on a path a client supplies, and +/// this runs inside a `#[tonic::async_trait]` /// handler on the shared runtime. A path under a hung NFS mount or an /// unreachable autofs trigger would otherwise park a worker thread with no /// timeout; enough concurrent calls and the prompt-delivery tasks stall, which diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index c0ea326..06aaf2d 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -13,8 +13,9 @@ Two long-running processes: [iced](https://iced.rs/). The CLI tool `cfc` shares the same gRPC client path as the UI, and covers -the same surface: it can answer prompts (`cfc prompts`), which is how a -headless machine gets a say. +the same surface: it can answer prompts (`sudo cfc prompts`), which is how a +headless machine gets a say. Run without sudo it is read-only (see +[IPC and the trust model](#ipc-and-the-trust-model)). ``` +----------------------------+ +----------------+ @@ -24,7 +25,7 @@ headless machine gets a say. +------------+---------------+ +-------+--------+ | | | tonic gRPC over UDS, 0660 root:colony-firewall - | (SO_PEERCRED checked per RPC) + | (peer process checked per change RPC) v v +--------------------------------------------------+ | colony-firewalld (systemd, root) | @@ -326,24 +327,38 @@ the entire attack surface. Two layers: is never briefly group-readable by the wrong group. If the group does not exist the daemon does not refuse to start: it warns with the exact fix and leaves the socket 0600, root-only. -2. **Peer credentials.** Every connection carries `SO_PEERCRED`. Mutating - RPCs (`UpsertRule`, `ApplyRules`, `DeleteRule`, `SetPaused`, `SubmitVerdict`) require - uid 0 or a socket that is genuinely group-gated. Read-only RPCs - (`ListRules`, `GetStatus`, `ListEvents`, `StreamConnections`, - `StreamPrompts`) are open to any peer that got past layer 1. - -Group membership *is* the credential - there is no in-band authentication. -Everyone in the group is fully trusted. The one exception is prompt -ownership: the daemon records which subscriber uids actually received each -prompt and refuses a verdict from anyone else, so one desktop session cannot -answer another's. Root is exempt. +2. **Who the peer process is.** Read-only RPCs (`ListRules`, `GetStatus`, + `ListEvents`, `StreamConnections`, `StreamPrompts`) are open to any peer + that got past layer 1. Change RPCs (`SubmitVerdict`, `UpsertRule`, + `DeleteRule`, `ApplyRules`, `SetPaused`) are accepted from root (or the + daemon's own uid) and from the installed app and tray only. From + `SO_PEERCRED`'s pid, `official.rs` checks, between two start-time reads, + that the process runs in the host namespaces, is not traced, has sealed + itself (`cfc_client::seal_official_process`: inherited descriptors closed, + non-dumpable), runs one of the root-sealed `[ipc] official_clients` + binaries by device and inode, holds this connection's client end itself + (`UNIX_DIAG` names it) and mapped no executable file from outside sealed + directories. The check runs on the blocking pool. +3. **polkit.** `SetPaused` (pause and resume) and `ApplyRules` from the app + or tray also need `CheckAuthorization` for + `org.projectcolony.firewall.pause` or `org.projectcolony.firewall.import-rules` + (`polkit.rs`, one system-bus connection per call, 120 s timeout, the + dialog cancelled on expiry). Root is never asked. + +Every other peer is read-only. Its prompt subscription is not counted in the +router's census, so `no_ui_action` still applies when only such peers +listen, and it never enters a prompt's audience. Prompt ownership comes on +top: the daemon records which answering subscriber uids actually received +each prompt and refuses a verdict from anyone else, so one desktop session +cannot answer another's. Root is exempt. Values arriving over the wire are decoded strictly. An unspecified or out-of-range action or duration is an `InvalidArgument` error, not a silent fall-through to the zero value - which happened to be Allow. -Every mutating RPC and every Deny/Reject verdict is logged to the journal -with the calling uid and pid. See [HARDENING.md](HARDENING.md). +Every change RPC (allowed or refused, with the official image and the polkit +action) and every Deny/Reject verdict is logged to the journal with the +calling uid and pid. See [HARDENING.md](HARDENING.md). ## Threading model diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 5703d59..13a2718 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -9,15 +9,16 @@ desktop, not what's theoretically pure. 1. Start in `profile = "balanced"`, leave the UI running. 2. Click through prompts for a week. Save persistent rules as you go. -3. Run `cfc rules bootstrap-defaults` to install common system rules. +3. Run `sudo cfc rules bootstrap-defaults` to install common system rules. 4. Once the prompt rate drops to maybe 1-2 a day, switch to `profile = "strict"` if you prefer a shorter prompt timeout. 5. Audit `cfc rules list` monthly. Remove rules for apps you no longer use, and check `cfc log --since 30d` for destinations you did not expect. -On a headless machine, substitute `cfc prompts` for "leave the UI -running" throughout - it subscribes the same way the GUI does. +On a headless machine, substitute `sudo cfc prompts` for "leave the UI +running" throughout - it subscribes the same way the GUI does. Without sudo +it only watches (see [the control socket](#the-control-socket-and-who-can-talk-to-it)). ## Choosing a profile @@ -87,7 +88,7 @@ User-side conveniences that hit the network constantly: You can install the system service rules with one command: ```sh -cfc rules bootstrap-defaults +sudo cfc rules bootstrap-defaults ``` This is idempotent: it skips the rules it installed earlier and identical @@ -113,7 +114,7 @@ Allow, pause or prompt can admit it. Replace these rules explicitly with executable or numeric scopes; a legacy hostname Allow no longer grants access. The editor requires the old hostname to be removed before saving a replacement. Such a rule cannot be disabled either, since a toggle sends the hostname back -and the daemon refuses it: edit or delete it (`cfc rules remove `). Its +and the daemon refuses it: edit or delete it (`sudo cfc rules remove `). Its refusals are logged as the default policy, so the daemon names every enabled legacy hostname rule in a warning at startup. @@ -291,38 +292,104 @@ connect - the UI will report a permission error. Create the group with the shipped `sysusers.d` fragment or by hand (`groupadd -r colony-firewall`), then add yourself and restart. -**Layer 2 - peer credentials.** Every connection carries `SO_PEERCRED`, +**Layer 2 - who the caller is.** Every connection carries `SO_PEERCRED`, and the daemon checks the caller per RPC: -| RPC class | RPCs | Requires | -|-----------|-----------------------------|-----------------------------| -| Mutating | `UpsertRule`, `ApplyRules`, `DeleteRule`, `SetPaused`, `SubmitVerdict` | uid 0, **or** a socket that is genuinely group-gated | -| Read-only | `ListRules`, `GetStatus`, `ListEvents`, `StreamConnections`, `StreamPrompts` | Only layer 1 | - -`require_group = false` in `[ipc]` turns the mutating check off. Leave it -on unless you are gating the socket some other way (filesystem ACLs); -with it off, any process that manages to connect can rewrite your rules. - -**Say it plainly: every member of the group is fully trusted.** There is -no in-band authentication, no per-user identity, and no password. Group -membership grants the ability to allow or deny any traffic on this host, -which is root-equivalent control over the firewall. This is not a -multi-user privilege boundary - add only administrators of the machine. - -**The one exception is prompt ownership.** A prompt is about a process, -and that process has an owner uid. Delivery is scoped to it: a -`StreamPrompts` subscription is handed a prompt only when the subscriber's -peer uid matches the owner, and `SubmitVerdict` refuses a caller the prompt -was not handed to. So another logged-in user's session is neither shown the -prompt nor able to answer it - it never even learns the prompt id. - -Exactly what that does and does not promise: +| RPC | root (`sudo cfc`) | installed app or tray | any other program | +|-----|-------------------|-----------------------|-------------------| +| `ListRules`, `GetStatus`, `ListEvents`, `StreamConnections` | yes | yes | yes | +| `StreamPrompts` | yes, counts as a UI | yes, counts as a UI | sees its prompts, does **not** count as a UI | +| `SubmitVerdict`, `UpsertRule`, `DeleteRule` | yes | yes, no password | refused | +| `SetPaused` (pause **and** resume), `ApplyRules` (import, replace, bundles) | yes, no password | after polkit authorization | refused | + +There is no RPC that changes `[default_policy]`: that is root editing +`daemon.toml` and sending `SIGHUP`. + +**Group membership no longer grants control.** It lets your session connect +and read, and lets the installed app and tray connect. Everything else of +yours, a non-root `cfc` included, is read-only: a change is refused with the +reason, a read-only `cfc prompts` does not stop `no_ui_action` from applying, +and it never enters a prompt's audience, so it cannot answer even a prompt +about its own process. Any process running as the daemon's own uid is treated +like root; in production that *is* root. + +**How the app and tray are recognised.** Each request from a non-root peer +that wants to change something is checked against the calling process, all +of it between two reads of that process's start time (so a reused pid fails): + +- it runs in the host's mount and user namespaces; +- it is not traced, and its effective uid is the connection's; +- it ran the app's sealing prologue at startup: every inherited descriptor + closed and the process made non-dumpable, which the kernel shows by giving + its `/proc` files to root; +- its running image is, by device and inode, one of `[ipc] official_clients` + (default `/usr/bin/colony-firewall` and `/usr/bin/colony-firewall-tray`), + and that file is root-owned, unwritable by group and other, in root-owned + directories nobody else can write. Re-checked every time, so after an + upgrade a process still running the replaced binary is read-only until it + is restarted; +- it holds the client end of this very connection itself, on a descriptor + above stderr (found through sock_diag's `UNIX_DIAG`); +- every executable file it mapped comes from such sealed directories, so an + `LD_PRELOAD` or `LD_AUDIT` library from your home directory makes it + read-only, with the library named. + +The prologue and the descriptor check are what stop the obvious trick: +connect, write a whole request into the socket, then `exec` the installed app +with the socket inherited. Non-dumpable also stops later same-user `ptrace`, +`/proc//mem` and `pidfd_getfd` on the app. + +`require_group = true` (the default) also requires the app or tray to be run +by a proved group member. `require_group = false` waives that for the app +and tray only; it never makes anything else writable. + +**polkit for whole-firewall changes.** Pause, resume and rule import change +everything at once, so even the app and tray need an administrator password +for them: the daemon asks polkit (`org.projectcolony.firewall.pause`, +`org.projectcolony.firewall.import-rules`, both `auth_admin_keep`, so one +password covers a few minutes) and your session's polkit agent shows the +dialog. Without an agent (start one, e.g. `hyprpolkitagent` or +`polkit-gnome`) or without polkit, the request is refused with that reason +and `sudo cfc pause` still works. The daemon waits 120 s for an answer, then +cancels the dialog. Answering a prompt and editing one rule never ask. + +**What this still trusts.** The check is about the *process*, so code that +runs inside the installed app is the app: + +- a preloaded payload that copies itself into anonymous executable memory and + unmaps its file is not seen (anonymous executable mappings cannot be + refused: GPU drivers JIT into them); +- synthetic input into the GUI under X11 or XWayland can click its buttons; +- a user-installed Vulkan layer, GTK or input-method module, or a global + `LD_PRELOAD` (MangoHud, gamemode) loaded from your home directory makes the + app read-only rather than trusted. The refusal names the library. + +The kernel-enforced next step would be making the two binaries setgid to a +dedicated empty group and trusting the connect-time `SO_PEERCRED` gid: glibc +then ignores `LD_PRELOAD`/`LD_AUDIT` (secure execution) and the process is +non-dumpable from `exec`. That costs packaging work in every channel and is +not done yet. + +Side effects of the sealing prologue: the app and tray write no core dumps, +attaching a debugger to them needs root, and a developer build run from +`target/` is never official (its directory is not root-sealed). Run the +daemon as your own user to test writes from a development build. + +**Prompt ownership.** A prompt is about a process, and that process has an +owner uid. Delivery is scoped to it: a `StreamPrompts` subscription is handed +a prompt only when the subscriber's peer uid matches the owner, and +`SubmitVerdict` refuses a caller the prompt was not handed to. So another +logged-in user's session is neither shown the prompt nor able to answer it - +it never even learns the prompt id. + +Exactly what that does and does not promise (for callers that may answer at +all, see above): | Prompt is about a process owned by | Delivered to | Answerable by | |------------------------------------|------------------------|------------------------| -| uid 1000 | uid 1000, root | uid 1000, root | +| uid 1000 | uid 1000, root | uid 1000's app or tray, root | | uid 0 (a system daemon) | root only | root only | -| nobody - attribution failed | every subscriber | every subscriber that received it | +| nobody - attribution failed | every subscriber | every app, tray or root subscriber that received it | Two deliberate consequences: @@ -332,7 +399,7 @@ Two deliberate consequences: *root-owned* process is not shown to an ordinary user's UI. With no root subscriber connected there is no audience for it, so the daemon answers it immediately with `no_ui_action` rather than stalling the packet until - `prompt_timeout_secs` expires. Run the CLI as root if you want to be + `prompt_timeout_secs` expires. Run `sudo cfc prompts` if you want to be asked about system daemons. - **Unattributed flows are offered to everyone.** When the process exited before `/proc` could be read the daemon has no owner uid to match. It @@ -341,7 +408,7 @@ Two deliberate consequences: policy in exactly the case where a human should look. This is prompt-level isolation between sessions, not a privilege boundary: -every group member can still write rules that affect the whole host. +any group member's app can still write rules that affect the whole host. ## What hot-reloads and what needs a restart @@ -519,7 +586,8 @@ uses it only on the loopback rule (`oifname "lo"`). Order of operations: -1. `cfc pause --for 15m`: unmatched outbound flows pass instead of being +1. `sudo cfc pause --for 15m` (or Pause in the app, which asks for an + administrator password): unmatched outbound flows pass instead of being denied while you debug, explicit rules still apply, and it resumes on its own. Every profile denies unmatched flows, so switching profile changes nothing. @@ -540,7 +608,7 @@ cfc rules export --out ~/cfc-rules-$(date +%F).json Restore with: ```sh -cfc rules import --replace ~/cfc-rules-2026-05-25.json +sudo cfc rules import --replace ~/cfc-rules-2026-05-25.json ``` `--replace` makes the daemon's rule set match the file: every rule in the file diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 98d83e9..a5eba34 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -229,7 +229,9 @@ The GUI will not connect, or `cfc` prints: ``` permission denied on /run/colony-firewall/cfc.sock - add your user to the colony-firewall group (sudo usermod -aG colony-firewall $USER) then log -out and back in, or run as root +out and back in, or run as root. The group gives read access and lets the +Colony Firewall app and tray connect; firewall changes come from the app, +the tray or sudo cfc ``` The control socket is `root:colony-firewall` mode 0660, so the kernel @@ -281,6 +283,66 @@ Two neighbouring errors that are *not* this one, and say so: Every one of these exits 4 ("daemon unreachable"), so scripts can tell them apart from a bad argument (2) or a missing rule (3). +## A change is refused: "read-only access" + +Since 0.8.0 only root and the installed Colony Firewall app and tray can +change the firewall. Everything else of yours, `cfc` without sudo included, +is read-only, and the daemon says why (exit 1): + +``` +read-only access: . Firewall changes are accepted only from the +installed Colony Firewall app and tray, or from root (sudo cfc ...). +``` + +- **From `cfc`**: run it with `sudo`. A non-root `cfc prompts` still shows + prompts but cannot answer them, and does not count as a connected UI. +- **"this colony-firewall is not the installed /usr/bin/colony-firewall + (restart it after an upgrade)"** or **"the caller did not seal itself at + startup"**: the app or tray was upgraded under you, or is a 0.7 build. + Quit and start it again (the tray from your session's autostart or by + hand, `colony-firewall-tray &`). +- **"the caller loaded /home/…/something.so"**: a library from outside the + root-owned system directories is mapped into the app, usually a global + `LD_PRELOAD` (MangoHud, gamemode) or a user-installed Vulkan layer or + GTK/input-method module. Start the app without it. The daemon refuses to + trust a process that runs code it cannot vouch for. +- **"the caller is being traced"**: a debugger or `strace` is attached. +- **"the caller runs in a private mnt (or user) namespace"**: the app was + started inside a sandbox or container wrapper. Start it directly. +- **"the connection's client end is unknown (unix_diag: …)"**: the kernel + has no `unix_diag` support (module not loaded). `sudo modprobe unix_diag`; + until then the app and tray are read-only and `sudo cfc` works. +- **"mutating RPCs require uid 0 or membership of group 'colony-firewall'"**: + you started the app from a session that predates joining the group. Log + out and back in. + +The journal names the caller and the reason for every refusal: + +```sh +journalctl -u colony-firewalld -g 'refusing a firewall change' +``` + +## Pause, resume or import asks for a password, or fails + +Pause, resume and rule import change the whole firewall at once, so the app +and tray need an administrator password for them (polkit, kept for a few +minutes). Root (`sudo cfc pause`) is never asked. What the refusals mean: + +- **"authorization dialog dismissed"**: you cancelled it. +- **"no polkit authentication agent answered in your session"**: nothing in + your session shows polkit dialogs. Start one (`hyprpolkitagent`, + `polkit-gnome-authentication-agent-1`, `lxqt-policykit-agent`; most full + desktops already run one) or use `sudo cfc pause`. +- **"polkit is not installed or not running"** or **"the system D-Bus is + unreachable"**: install polkit, or use `sudo cfc`. +- **"not authorized by polkit policy"**: a local polkit rule denies + `org.projectcolony.firewall.pause` or `org.projectcolony.firewall.import-rules` + for you. `pkaction --verbose --action-id org.projectcolony.firewall.pause` + shows the defaults; a missing action means the policy file is not + installed in `/usr/share/polkit-1/actions/`. +- **"authorization timed out after 120 s"**: the dialog was left open; the + daemon closed it. + ## Loopback and the local resolver The snippet's `output` hook matches loopback traffic too. On systems using @@ -454,11 +516,12 @@ Kerberos, reverse DNS) is the exception; see Then pick one of three fixes: -**1. Answer prompts from the terminal.** This is what `cfc prompts` is -for - it subscribes just like the GUI does, so the daemon starts asking: +**1. Answer prompts from the terminal.** This is what `sudo cfc prompts` +is for - it subscribes just like the GUI does, so the daemon starts asking +(without sudo it only watches, and the daemon keeps applying `no_ui_action`): ```sh -cfc prompts +sudo cfc prompts ``` Keys are `a` allow, `d` deny, `r` reject, `s` skip (let it time out), `q` @@ -470,14 +533,14 @@ set. For a bounded unattended window - during a package install, say - `--auto-allow` or `--auto-deny` answer everything without asking, and `--count N` exits after N prompts. -**2. Pre-seed rules and accept the fallback.** `cfc rules -bundle add system` (also spelled `cfc rules bootstrap-defaults`) covers +**2. Pre-seed rules and accept the fallback.** `sudo cfc rules +bundle add system` (also spelled `sudo cfc rules bootstrap-defaults`) covers the usual system services. `cfc rules bundle list` shows the others — `web` for installed browsers, `dev` for git/cargo/docker, `updates` for apt/dnf/flatpak — each scoped to a specific executable, never to a bare port. Entries whose program is not installed here are skipped and reported. Add your own with -`cfc rules add`. Anything you did not anticipate still hits +`sudo cfc rules add`. Anything you did not anticipate still hits `no_ui_action`. **3. Change the fallback — deliberately.** `no_ui_action = "Allow"` in @@ -488,7 +551,7 @@ unanticipated connection — including a payload phoning home — goes out unasked. Prefer (1) or (2). If you do set it, send `SIGHUP` and it takes effect without a restart. -Note that `cfc prompts` and the GUI can both be connected at once, and +Note that `sudo cfc prompts` and the GUI can both be connected at once, and both see the prompts addressed to you. Delivery is scoped by the uid that owns the connecting process: you receive prompts for your own processes, root receives everything, and traffic the daemon could not diff --git a/scripts/smoke-test.sh b/scripts/smoke-test.sh index 0d303c0..c484db2 100755 --- a/scripts/smoke-test.sh +++ b/scripts/smoke-test.sh @@ -78,10 +78,10 @@ profile = "balanced" [storage] path = "${DB}" [ipc] -# This test runs unprivileged against a socket in a temp dir, so it can be -# neither root-owned nor gated by the colony-firewall group. Without this -# the daemon (correctly) refuses every mutating RPC and the test can only -# exercise the read-only half of the CLI. Production keeps the default. +# This test runs the daemon and cfc as the same unprivileged user, and a +# client with the daemon's own uid has full control, so writes pass without +# this. The key only waives the group check for the installed app and tray; +# it is kept as a harmless reminder that this socket is not group-gated. require_group = false EOF From b94acd71079b2b16cdb8d504c97c755202a789de Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 11:57:00 +0200 Subject: [PATCH 113/125] docs(daemon): name the handed-back socket gap and the group refusal plainly A same-user program that connects, execs the installed app and keeps a copy of the connection can pass it back into the sealed app over D-Bus while the copy writes a request, so the descriptor check sees the app holding it. The module docs, HARDENING and TODO now say so, with the setgid follow-up that would close it. The group refusal now says that changes need root or the app or tray run by a group member, instead of implying membership is enough. --- TODO.md | 2 +- crates/cfc-daemon/src/ipc.rs | 16 +++++++++++++++- crates/cfc-daemon/src/official.rs | 13 ++++++++++--- docs/HARDENING.md | 13 ++++++++++--- docs/TROUBLESHOOTING.md | 6 +++--- 5 files changed, 39 insertions(+), 11 deletions(-) diff --git a/TODO.md b/TODO.md index 62611cd..263b5b5 100644 --- a/TODO.md +++ b/TODO.md @@ -253,7 +253,7 @@ What defeats it completely: | **DNS tunnelling** | the resolver must be allowed for anything to work. CFC *observes* answers; it does not inspect or block queries. | | **Inherited or passed socket descriptors** | Existing connection authorization is not rechecked for each sending executable; socket attribution is ambiguous when ownership is shared. | | **CAP_NET_RAW packet sockets** | Packet-layer egress can bypass the IP OUTPUT hook. Layer-2 confinement is outside the shipped rules. | -| **Code inside the official app or tray** | the daemon accepts changes only from root and the installed, sealed app and tray, checking the running process (image, prologue, connection, tracer, file-backed executable mappings). Code already running inside them is not seen: a self-unmapping `LD_PRELOAD` payload living in anonymous memory, or synthetic X11/XWayland input clicking the GUI. Setgid binaries trusted by connect-time gid would close the first; not done. | +| **Code inside the official app or tray** | the daemon accepts changes only from root and the installed, sealed app and tray, checking the running process (image, prologue, connection, tracer, file-backed executable mappings). Code already running inside them is not seen: a self-unmapping `LD_PRELOAD` payload living in anonymous memory, or synthetic X11/XWayland input clicking the GUI. Also a connection opened before exec'ing the app and handed back into it over D-Bus while a kept copy writes the request (a race, repeatable). Setgid binaries trusted by connect-time gid would close the first and the last; not done. | | **Prompt fatigue** | demonstrated on this machine: ten Firefox prompts in a row, all denied, browser lost. A malicious installer generating thirty prompts trains the user to click Allow. | And one tradeoff worth stating plainly: the ruleset is **fail-closed for diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index f3ccd04..0298b30 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -573,7 +573,8 @@ impl FirewallService { match gate(peer.uid, self.own_uid, group_ok) { Gate::Privileged => Ok(None), Gate::DenyGroup => Err(Status::permission_denied(format!( - "mutating RPCs require uid 0 or membership of group '{}'", + "firewall changes require root, or the installed Colony Firewall app or \ + tray run by a member of group '{}'", self.auth.group ))), Gate::NeedOfficial => { @@ -2016,6 +2017,19 @@ mod tests { status.message() ); assert!(!svc.stats.is_paused()); + + let status = svc + .set_paused(request( + SetPausedRequest::default(), + peer(1001, 1001, OFFICIAL_PID), + )) + .await + .unwrap_err(); + assert!( + status.message().contains("member of group 'cfc-test'"), + "{}", + status.message() + ); } /// A pending prompt about uid 1000's process, with an answering UI so it diff --git a/crates/cfc-daemon/src/official.rs b/crates/cfc-daemon/src/official.rs index 9a67030..1ad83e3 100644 --- a/crates/cfc-daemon/src/official.rs +++ b/crates/cfc-daemon/src/official.rs @@ -22,11 +22,18 @@ //! 8. its start time still matches (the "after" read), so none of the above //! was read from a process that replaced it under the same pid. //! -//! Steps 4 and 6 close the exec-after-connect route: connect, write a whole -//! request, then exec the official binary with the socket inherited. That -//! process has not run the prologue yet, or has closed the inherited +//! Steps 4 and 6 close the plain exec-after-connect route: connect, write a +//! whole request, then exec the official binary with the socket inherited. +//! That process has not run the prologue yet, or has closed the inherited //! connection by the time it is sealed. //! +//! They do not close a variant: a helper that kept a copy of that socket can +//! hand it back to the sealed process (the app and tray accept D-Bus messages +//! from any same-user process, and D-Bus carries descriptors) while it +//! writes a request on it. Step 6 then sees the process holding the +//! connection. Binding each request to its sender (SO_PASSCRED) or trusting +//! a connect-time credential (setgid binaries) would; see docs/HARDENING.md. +//! //! What it cannot see: code already running inside the official image that //! moved itself into anonymous memory (anonymous executable mappings are //! not judged, GPU drivers JIT into them), and synthetic input to the GUI diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 13a2718..c1a608d 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -360,15 +360,22 @@ runs inside the installed app is the app: unmaps its file is not seen (anonymous executable mappings cannot be refused: GPU drivers JIT into them); - synthetic input into the GUI under X11 or XWayland can click its buttons; +- a socket handed back into the app: a same-user program that started the + app itself (connect first, then exec the installed binary, keeping a copy + of the connection in a child) can pass that copy into the sealed app as a + D-Bus message attachment while the child writes a request on it. The + daemon then sees the app holding the connection and accepts the request. + This is a race, but a repeatable one; - a user-installed Vulkan layer, GTK or input-method module, or a global `LD_PRELOAD` (MangoHud, gamemode) loaded from your home directory makes the app read-only rather than trusted. The refusal names the library. The kernel-enforced next step would be making the two binaries setgid to a dedicated empty group and trusting the connect-time `SO_PEERCRED` gid: glibc -then ignores `LD_PRELOAD`/`LD_AUDIT` (secure execution) and the process is -non-dumpable from `exec`. That costs packaging work in every channel and is -not done yet. +then ignores `LD_PRELOAD`/`LD_AUDIT` (secure execution), the process is +non-dumpable from `exec`, and a connection made before that `exec` carries +the wrong gid, which also closes the handed-back socket above. That costs +packaging work in every channel and is not done yet. Side effects of the sealing prologue: the app and tray write no core dumps, attaching a debugger to them needs root, and a developer build run from diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index a5eba34..b468dcf 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -312,9 +312,9 @@ installed Colony Firewall app and tray, or from root (sudo cfc ...). - **"the connection's client end is unknown (unix_diag: …)"**: the kernel has no `unix_diag` support (module not loaded). `sudo modprobe unix_diag`; until then the app and tray are read-only and `sudo cfc` works. -- **"mutating RPCs require uid 0 or membership of group 'colony-firewall'"**: - you started the app from a session that predates joining the group. Log - out and back in. +- **"firewall changes require root, or the installed Colony Firewall app or + tray run by a member of group 'colony-firewall'"**: you started the app + from a session that predates joining the group. Log out and back in. The journal names the caller and the reason for every refusal: From 583049a6ca9e1be2eb620e8b276af635f66a63b6 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:04:04 +0200 Subject: [PATCH 114/125] feat(rules)!: a program Deny beats every generic Allow, and /0 adds no specificity A Deny or Reject rule that names a program (exe_path or exe_sha256) now wins over every Allow rule that names none, whatever their predicate counts, so "deny --exe /opt/agent" is no longer overridden by "allow --protocol tcp --dst-port 443". Rules that name a program keep their specificity order among themselves, and everything else keeps the old ordering. The override lives in RuleSet::lookup, not in the sort key: the relation is not a total order, so no comparator can express it. The walk holds the first matching generic Allow and lets only a lower program rule change the answer. Engine::process_wide_action and deny_still_possible_for walk the same way, so the in-kernel connect hooks and the packet path agree. A /0 network no longer counts in RuleScope::specificity. reject_unscoped still accepts a /0-only scope, so stored rules are not quarantined. --- CHANGELOG.md | 11 + crates/cfc-core/src/rule.rs | 437 ++++++++++++++++++++++++++++-- crates/cfc-daemon/src/convert.rs | 27 +- crates/cfc-daemon/src/decision.rs | 113 ++++++++ crates/cfc-daemon/src/storage.rs | 12 + docs/ARCHITECTURE.md | 23 +- docs/HARDENING.md | 9 + docs/ROADMAP.md | 4 +- docs/TROUBLESHOOTING.md | 14 + 9 files changed, 625 insertions(+), 25 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 8728f35..017cd6d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -8,6 +8,17 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). ### Changed +- **Breaking: a Deny or Reject rule scoped to a program wins over every + Allow rule that names no program**, whatever their predicate counts. + `deny --exe /opt/agent` used to lose to `allow --protocol tcp + --dst-port 443` because the Allow carried more predicates; it now + refuses the agent there too, and the in-kernel connect hooks refuse it + outright. Rules that name a program keep their specificity order among + themselves, so `allow --exe X --dst-port 443` still beats `deny --exe X`. + A `/0` network (`--dst-net 0.0.0.0/0`, `::/0`) no longer counts as a + predicate when rules are ranked; it still limits a rule to one address + family, and stored rules that carry only a `/0` keep loading. Some flows + change verdict on upgrade: review `cfc rules list`. - New loopback flows now go through the queue instead of being accepted outright (`oifname "lo" accept` is gone from the outbound table). While the daemon runs, explicit rules apply to them, so a loopback Deny that 0.7.0 diff --git a/crates/cfc-core/src/rule.rs b/crates/cfc-core/src/rule.rs index a726aae..cfe6f2b 100644 --- a/crates/cfc-core/src/rule.rs +++ b/crates/cfc-core/src/rule.rs @@ -146,16 +146,22 @@ impl RuleScope { /// a scope constraining *only* the direction now counts as constraining /// nothing, and `reject_unscoped` refuses it - "allow every inbound flow /// from anyone" was never a rule this store should hold. + /// + /// A `/0` network is not counted either, for the same reason: it narrows + /// nothing within its address family, so `--dst-net 0.0.0.0/0` must not + /// lift a rule above a genuinely narrower one. Matching is unchanged + /// (an IPv4 `/0` still excludes IPv6 flows), only the ranking. pub fn specificity(&self) -> u8 { + let narrows = |net: &Option| net.is_some_and(|n| n.prefix_len() > 0); [ - self.src_net.is_some(), + narrows(&self.src_net), self.src_port.is_some(), self.exe_path.is_some(), self.exe_sha256.is_some(), self.parent_exe.is_some(), self.uid.is_some(), self.dst_host.is_some(), - self.dst_net.is_some(), + narrows(&self.dst_net), self.dst_port.is_some(), self.protocol.is_some(), ] @@ -164,6 +170,15 @@ impl RuleScope { .count() as u8 } + /// True when this scope names a program: an `exe_path` or `exe_sha256` + /// predicate. + /// + /// The line [`RuleSet::lookup`] draws for precedence: a Deny or Reject + /// that names a program wins over every Allow that names none. + pub fn names_program(&self) -> bool { + self.exe_path.is_some() || self.exe_sha256.is_some() + } + /// True when this scope says anything at all about *where* a connection /// goes. /// @@ -562,14 +577,18 @@ fn action_rank(action: Action) -> u8 { impl RuleSet { /// Sort rules into deterministic precedence order: /// - /// 1. specificity DESC — more `Some(..)` scope predicates first; - /// 2. action severity — Deny, then Reject, before Allow on ties; - /// 3. `created_at` ASC — oldest rule first; - /// 4. `id` ASC — final total-order tiebreak. + /// 1. specificity DESC - more scope predicates first (see + /// [`RuleScope::specificity`]); + /// 2. action severity - Deny, then Reject, before Allow on ties; + /// 3. `created_at` ASC - oldest rule first; + /// 4. `id` ASC - final total-order tiebreak. + /// + /// [`RuleSet::lookup`] applies one override on top of this order (a + /// program Deny beats a generic Allow); it cannot live in the sort key. /// /// Must be called whenever the set is (re)built or a rule is inserted, - /// replaced, or toggled, so `lookup`'s first-match walk is stable across - /// daemon restarts regardless of storage iteration order. + /// replaced, or toggled, so `lookup`'s walk is stable across daemon + /// restarts regardless of storage iteration order. pub fn sort_deterministic(&mut self) { self.rules.sort_by_key(|r| { ( @@ -583,10 +602,23 @@ impl RuleSet { /// Find the winning enabled, non-expired rule for `(conn, proc)`. /// - /// Precedence contract: most-specific scope wins; deny beats allow at - /// equal specificity; oldest rule first on remaining ties. This holds - /// because the set is kept in [`RuleSet::sort_deterministic`] order and - /// `lookup` returns the first match of that walk. + /// Precedence contract: + /// + /// 1. A Deny or Reject rule that names a program + /// ([`RuleScope::names_program`]) wins over every Allow rule that names + /// none, whatever their predicate counts. "Deny this program" means + /// that program, even where a broader "allow HTTPS" ranks higher. + /// 2. Otherwise the most-specific scope wins; deny beats allow at equal + /// specificity; oldest rule first on remaining ties. Rules that name a + /// program keep this order among themselves, so `allow --exe X + /// --dst-port 443` still beats `deny --exe X`. + /// + /// The set is kept in [`RuleSet::sort_deterministic`] order, which is the + /// second point alone. The first is applied here, during the walk, and not + /// in the sort key: the relation is not a total order (a program Allow of + /// specificity 3 beats a program Deny of 2, which beats a generic Allow of + /// 5, which beats a generic Deny of 4, which beats the program Allow), so + /// no comparator could express it. /// /// `now_unix_ms` is the current wall-clock time; rules whose /// `Duration::Seconds(..)` window has elapsed are skipped (see @@ -600,11 +632,19 @@ impl RuleSet { now_unix_ms: i64, ) -> Match<'_> { let mut undecidable = None; + // The first definite Allow that names no program. Held, not returned: + // a program Deny ranked below it may still override it. + let mut generic_allow: Option<&Rule> = None; for rule in self .rules .iter() .filter(|r| r.enabled && !r.is_expired(now_unix_ms)) { + // Once a generic Allow is held, only a program rule can change + // the answer: anything else ranks below it and loses by order. + if generic_allow.is_some() && !rule.scope.names_program() { + continue; + } // The connection half first, so a rule's own destination // predicates can exclude it before its process half is ever // questioned. Without that order an undecidable rule would abstain @@ -620,22 +660,46 @@ impl RuleSet { // application's intended hostname or enumerates every alias. // Keep a legacy name at its original priority as uncertainty. if missing_process || rule.scope.dst_host.is_some() { - let first = *undecidable.get_or_insert(rule); - // A possible Allow keeps its precedence. Only a sequence - // of possible closed actions may yield to a definite Deny. if rule.action == Action::Allow { - return Match::Undecidable(first); + // Under a held generic Allow, a possible program Allow + // cannot change the answer: if it matched, the generic + // Allow would still win. + if generic_allow.is_some() { + continue; + } + // A possible Allow keeps its precedence. + return Match::Undecidable(undecidable.unwrap_or(rule)); } + // Only a sequence of possible closed actions may yield to a + // definite Deny. + undecidable.get_or_insert(rule); continue; } - if rule.action == Action::Allow { + let winner = match generic_allow { + // A lower program Allow cannot turn the answer into a refusal. + Some(held) if rule.action == Action::Allow => held, + Some(_) => rule, + None if rule.action == Action::Allow && !rule.scope.names_program() => { + if let Some(first) = undecidable { + return Match::Undecidable(first); + } + generic_allow = Some(rule); + continue; + } + None => rule, + }; + if winner.action == Action::Allow { if let Some(first) = undecidable { return Match::Undecidable(first); } } - return Match::Rule(rule); + return Match::Rule(winner); + } + match (undecidable, generic_allow) { + (Some(first), _) => Match::Undecidable(first), + (None, Some(held)) => Match::Rule(held), + (None, None) => Match::None, } - undecidable.map_or(Match::None, Match::Undecidable) } } @@ -1186,6 +1250,341 @@ mod tests { assert_eq!(hit_fwd.name, "deny-curl"); } + fn scoped(name: &str, action: Action, scope: RuleScope) -> Rule { + Rule::new(name, action, scope) + } + + fn sorted(rules: Vec) -> RuleSet { + let mut set = RuleSet { rules }; + set.sort_deterministic(); + set + } + + fn winner<'a>(set: &'a RuleSet, conn: &Connection, proc: &Process) -> Option<&'a str> { + set.lookup(conn, proc, now()) + .rule() + .map(|r| r.name.as_str()) + } + + #[test] + fn a_program_deny_beats_a_more_specific_generic_allow() { + // The scenario that made the override necessary: "deny this agent" + // lost to "allow HTTPS" because the allow carried two predicates. + let set = sorted(vec![ + scoped( + "deny-telemetry", + Action::Deny, + RuleScope { + exe_path: Some("/opt/telemetry-agent".into()), + ..RuleScope::any() + }, + ), + scoped( + "allow-https", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert_eq!(set.rules[0].name, "allow-https", "the sort is unchanged"); + let conn = mk_conn(); + assert_eq!( + winner(&set, &conn, &mk_proc("/opt/telemetry-agent")), + Some("deny-telemetry") + ); + assert_eq!( + winner(&set, &conn, &mk_proc("/usr/bin/curl")), + Some("allow-https") + ); + } + + #[test] + fn a_program_reject_beats_a_three_predicate_generic_allow() { + let set = sorted(vec![ + scoped( + "reject-x", + Action::Reject, + RuleScope { + exe_sha256: Some("aa".repeat(32)), + ..RuleScope::any() + }, + ), + scoped( + "allow-net", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + dst_net: Some("1.2.3.0/24".parse().unwrap()), + ..RuleScope::any() + }, + ), + ]); + let hashed = Process { + sha256: Some("aa".repeat(32)), + ..mk_proc("/usr/bin/x") + }; + assert_eq!(winner(&set, &mk_conn(), &hashed), Some("reject-x")); + } + + #[test] + fn a_more_specific_program_allow_still_beats_a_program_deny() { + let set = sorted(vec![ + scoped( + "deny-x", + Action::Deny, + RuleScope { + exe_path: Some("/usr/bin/x".into()), + ..RuleScope::any() + }, + ), + scoped( + "allow-x-443", + Action::Allow, + RuleScope { + exe_path: Some("/usr/bin/x".into()), + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + let x = mk_proc("/usr/bin/x"); + assert_eq!(winner(&set, &mk_conn(), &x), Some("allow-x-443")); + let http = Connection { + dst_port: 80, + ..mk_conn() + }; + assert_eq!(winner(&set, &http, &x), Some("deny-x")); + } + + #[test] + fn a_program_allow_below_a_generic_allow_leaves_it_the_answer() { + let set = sorted(vec![ + scoped( + "allow-x", + Action::Allow, + RuleScope { + exe_path: Some("/usr/bin/x".into()), + ..RuleScope::any() + }, + ), + scoped( + "allow-https", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert_eq!( + winner(&set, &mk_conn(), &mk_proc("/usr/bin/x")), + Some("allow-https") + ); + } + + #[test] + fn a_generic_deny_keeps_the_old_order_against_generic_allows() { + // The override is about rules that name a program. Between rules that + // name none, specificity still decides, whatever the action. + let set = sorted(vec![ + scoped( + "deny-443", + Action::Deny, + RuleScope { + dst_port: Some(443), + ..RuleScope::any() + }, + ), + scoped( + "allow-tcp-443", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert_eq!( + winner(&set, &mk_conn(), &mk_proc("/usr/bin/curl")), + Some("allow-tcp-443") + ); + } + + #[test] + fn an_unknown_process_under_a_program_deny_and_a_generic_allow_is_undecidable() { + // The program Deny could be the one that overrides the Allow, and the + // process cannot be checked against it. + let set = sorted(vec![ + scoped( + "deny-x", + Action::Deny, + RuleScope { + exe_path: Some("/usr/bin/x".into()), + ..RuleScope::any() + }, + ), + scoped( + "allow-https", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert!(matches!( + set.lookup(&mk_conn(), &Process::unknown(0), now()), + Match::Undecidable(r) if r.name == "deny-x" + )); + // A known other program is decided: the Deny cannot be about it. + assert_eq!( + winner(&set, &mk_conn(), &mk_proc("/usr/bin/curl")), + Some("allow-https") + ); + } + + #[test] + fn lookup_is_independent_of_insertion_order_with_the_override() { + // Four rules whose pairwise precedence is a cycle: program allow (3) + // beats program deny (2) by specificity, which beats generic allow (5) + // by the override, which beats generic deny (4) by specificity, which + // beats the program allow (3) by specificity. The answer must still + // not depend on the order the rules arrived in. + let x = "/usr/bin/x"; + let net = || Some("1.2.3.0/24".parse().unwrap()); + let src = || Some("192.168.1.0/24".parse().unwrap()); + let rules = [ + scoped( + "program-allow", + Action::Allow, + RuleScope { + exe_path: Some(x.into()), + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + ..RuleScope::any() + }, + ), + scoped( + "program-deny", + Action::Deny, + RuleScope { + exe_path: Some(x.into()), + protocol: Some(Protocol::Tcp), + ..RuleScope::any() + }, + ), + scoped( + "generic-allow", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + dst_net: net(), + src_net: src(), + src_port: Some(54321), + ..RuleScope::any() + }, + ), + scoped( + "generic-deny", + Action::Deny, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + dst_net: net(), + src_net: src(), + ..RuleScope::any() + }, + ), + ]; + let proc = mk_proc(x); + let conn = mk_conn(); + let mut answers = std::collections::BTreeSet::new(); + for a in 0..4 { + for b in (0..4).filter(|b| *b != a) { + for c in (0..4).filter(|c| *c != a && *c != b) { + let d = 6 - a - b - c; + let set = sorted([a, b, c, d].map(|i| rules[i].clone()).to_vec()); + answers.insert(winner(&set, &conn, &proc).map(str::to_owned)); + } + } + } + assert_eq!( + answers.into_iter().collect::>(), + [Some("generic-allow".to_owned())] + ); + + // Without the program allow, the program deny overrides the generic + // allow above it. + let set = sorted(rules[1..].to_vec()); + assert_eq!(winner(&set, &conn, &proc), Some("program-deny")); + } + + #[test] + fn slash_zero_networks_add_no_specificity() { + let any_v4 = RuleScope { + dst_net: Some("0.0.0.0/0".parse().unwrap()), + dst_port: Some(443), + ..RuleScope::any() + }; + assert_eq!(any_v4.specificity(), 1); + let any_v6_source = RuleScope { + src_net: Some("::/0".parse().unwrap()), + ..RuleScope::any() + }; + assert_eq!(any_v6_source.specificity(), 0); + let half = RuleScope { + dst_net: Some("0.0.0.0/1".parse().unwrap()), + ..RuleScope::any() + }; + assert_eq!(half.specificity(), 1); + + // Only the ranking changed: an IPv4 `/0` still excludes IPv6 flows. + let v6 = Connection { + dst_ip: "2001:db8::1".parse().unwrap(), + src_ip: "2001:db8::2".parse().unwrap(), + ..mk_conn() + }; + let proc = mk_proc("/usr/bin/curl"); + assert!(any_v4.matches(&mk_conn(), &proc)); + assert!(!any_v4.matches(&v6, &proc)); + } + + #[test] + fn a_slash_zero_rule_does_not_outrank_by_count() { + // Both now count one predicate, so the tie-break decides: closed wins. + let set = sorted(vec![ + scoped( + "allow-all-v4-tcp", + Action::Allow, + RuleScope { + dst_net: Some("0.0.0.0/0".parse().unwrap()), + protocol: Some(Protocol::Tcp), + ..RuleScope::any() + }, + ), + scoped( + "deny-443", + Action::Deny, + RuleScope { + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert_eq!( + winner(&set, &mk_conn(), &mk_proc("/usr/bin/curl")), + Some("deny-443") + ); + } + #[test] fn equal_specificity_and_action_oldest_wins() { let mut scope = RuleScope::any(); diff --git a/crates/cfc-daemon/src/convert.rs b/crates/cfc-daemon/src/convert.rs index ba52eef..c9611f9 100644 --- a/crates/cfc-daemon/src/convert.rs +++ b/crates/cfc-daemon/src/convert.rs @@ -336,8 +336,12 @@ pub fn rule_to_pb(r: &Rule) -> pb::RuleInfo { /// matches EVERYTHING"), which is the strongest argument for enforcing it here: /// three clients independently decided it was dangerous, and the one boundary /// they all pass through did not check. +/// +/// A `/0` network adds no specificity but still counts as a scope here: it +/// limits the rule to one address family, and refusing it would quarantine +/// rules stored before `/0` stopped ranking (storage runs this at load). pub fn reject_unscoped(scope: &RuleScope) -> Result<(), String> { - if scope.specificity() == 0 { + if scope.specificity() == 0 && scope.src_net.is_none() && scope.dst_net.is_none() { return Err( "rule scope constrains nothing, so it would match every process and \ every destination; scope it to at least one of exe_path, uid, \ @@ -582,6 +586,27 @@ mod tests { assert!(rule_from_pb(&pb).is_ok()); } + #[test] + fn a_slash_zero_only_scope_is_still_storable() { + // `/0` stopped adding specificity, but it still limits a rule to one + // address family, so the gate keeps accepting it: refusing it here + // would also quarantine such rules already on disk. + for net in ["0.0.0.0/0", "::/0"] { + let dst = RuleScope { + dst_net: Some(net.parse().unwrap()), + ..RuleScope::any() + }; + assert_eq!(dst.specificity(), 0); + assert_eq!(reject_unscoped(&dst), Ok(()), "{net}"); + let src = RuleScope { + src_net: Some(net.parse().unwrap()), + ..RuleScope::any() + }; + assert_eq!(reject_unscoped(&src), Ok(()), "{net}"); + } + assert!(reject_unscoped(&RuleScope::any()).is_err()); + } + #[test] fn an_out_of_range_port_is_refused_rather_than_wrapped() { // `as u16` wrapped: a scope asking for 65979 became a rule scoped to diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index 0aecd29..ae5feac 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -209,11 +209,18 @@ impl Engine { /// [`RuleScope::undecidable_for`] - also ends the walk with `None`, for /// the same reason: it might have been the one that mattered. /// + /// An Allow that names no program is walked past, because `lookup` lets a + /// program Deny or Reject below it win anyway. The first other rule that + /// applies then answers as above if it is such a refusal. If it is + /// anything else and a generic Allow was passed, the answer depends on + /// the destination (that Allow, or what lies below it), so `None`. + /// /// `None` means "ask the packet path", which is always a safe answer: it /// is what happened before this existed. pub fn process_wide_action(&self, proc: &Process) -> Option { let now_unix_ms = chrono::Utc::now().timestamp_millis(); let rules = self.inner.rules.read(); + let mut saw_generic_allow = false; for rule in rules .rules .iter() @@ -241,6 +248,13 @@ impl Engine { if !rule.scope.matches_process(proc) { continue; } + if is_generic_allow(rule) { + saw_generic_allow = true; + continue; + } + if saw_generic_allow && !is_program_refusal(rule) { + return None; + } return (!rule.scope.constrains_destination()).then_some(rule.action); } None @@ -274,15 +288,22 @@ impl Engine { /// * a decidable match ends the walk exactly as `process_wide_action` /// does: its action (or the packet path, for a destination-scoped /// rule) is the whole answer, and nothing below it can matter. + /// * a decidable Allow that names no program does not end it: a program + /// Deny below it still wins in `lookup`, so from there on only rules + /// that name a program are looked at. pub fn deny_still_possible_for(&self, proc: &Process) -> bool { let now_unix_ms = chrono::Utc::now().timestamp_millis(); let rules = self.inner.rules.read(); + let mut saw_generic_allow = false; for rule in rules .rules .iter() .filter(|r| r.enabled && !r.is_expired(now_unix_ms)) .filter(|r| r.scope.direction != Some(cfc_core::Direction::Inbound)) { + if saw_generic_allow && !rule.scope.names_program() { + continue; + } if rule.scope.undecidable_for(proc) { if matches!( rule.action, @@ -296,6 +317,10 @@ impl Engine { if !rule.scope.matches_process(proc) { continue; } + if is_generic_allow(rule) { + saw_generic_allow = true; + continue; + } return matches!( rule.action, cfc_core::Action::Deny | cfc_core::Action::Reject @@ -498,6 +523,18 @@ impl Engine { } } +/// An Allow that names no program: the rule `lookup` lets a program refusal +/// override, whatever its rank. +fn is_generic_allow(rule: &Rule) -> bool { + rule.action == cfc_core::Action::Allow && !rule.scope.names_program() +} + +/// A Deny or Reject that names a program: the rule that overrides a generic +/// Allow. +fn is_program_refusal(rule: &Rule) -> bool { + rule.action != cfc_core::Action::Allow && rule.scope.names_program() +} + #[cfg(test)] mod tests { use super::*; @@ -557,6 +594,13 @@ mod tests { Rule::new(format!("allow-{port}"), Action::Allow, scope) } + /// Two predicates, so it outranks a one-predicate rule of any action. + fn allow_tcp_port_rule(port: u16) -> Rule { + let mut rule = allow_port_rule(port); + rule.scope.protocol = Some(Protocol::Tcp); + rule + } + fn deny_port_rule(port: u16) -> Rule { let mut scope = RuleScope::any(); scope.dst_port = Some(port); @@ -894,6 +938,75 @@ mod tests { assert_eq!(engine.process_wide_action(&proc("/usr/bin/wget")), None); } + #[test] + fn process_wide_action_applies_the_program_deny_over_a_generic_allow() { + // `lookup` lets the program Deny beat the higher-ranked generic Allow, + // so the connect hooks may refuse X outright: the packet path would + // refuse every flow of X too. Both must agree. + let engine = engine_with(vec![ + exe_rule("deny-x", "/usr/bin/x", Action::Deny), + allow_tcp_port_rule(443), + ]); + let x = proc("/usr/bin/x"); + assert_eq!(engine.process_wide_action(&x), Some(Action::Deny)); + assert!(engine.deny_still_possible_for(&x)); + assert!(matches!( + engine.peek(&conn(443), &x), + Decision::Resolved(v) if v.action == Action::Deny + )); + + let y = proc("/usr/bin/y"); + assert_eq!(engine.process_wide_action(&y), None); + assert!(!engine.deny_still_possible_for(&y)); + } + + #[test] + fn deny_still_possible_sees_a_program_deny_below_a_generic_allow() { + // The program Deny is pinned to a digest this process lacks: if it + // matches, it overrides the generic Allow ranked above it. + let mut hashed = RuleScope::any(); + hashed.exe_path = Some(PathBuf::from("/usr/bin/x")); + hashed.exe_sha256 = Some("aa".repeat(32)); + let mut generic = RuleScope::any(); + generic.uid = Some(1000); + generic.protocol = Some(Protocol::Tcp); + generic.dst_port = Some(443); + let engine = engine_with(vec![ + Rule::new("deny-x-pinned", Action::Deny, hashed), + Rule::new("allow-user-https", Action::Allow, generic), + ]); + let no_hash = proc("/usr/bin/x"); + assert_eq!(engine.process_wide_action(&no_hash), None); + assert!(engine.deny_still_possible_for(&no_hash)); + } + + #[test] + fn a_generic_allow_above_a_generic_deny_keeps_process_wide_none() { + // No rule names a program, so the override does not apply and the + // old order stands: the generic Allow wins, and a lower destination- + // free generic Deny is never reachable. + let mut user = RuleScope::any(); + user.uid = Some(1000); + let engine = engine_with(vec![ + allow_tcp_port_rule(443), + Rule::new("deny-user", Action::Deny, user), + ]); + let p = proc("/usr/bin/curl"); + assert_eq!(engine.process_wide_action(&p), None); + assert!(!engine.deny_still_possible_for(&p)); + } + + #[test] + fn a_program_allow_under_a_generic_allow_is_no_process_wide_answer() { + let engine = engine_with(vec![ + exe_rule("allow-x", "/usr/bin/x", Action::Allow), + allow_tcp_port_rule(443), + ]); + let x = proc("/usr/bin/x"); + assert_eq!(engine.process_wide_action(&x), None); + assert!(!engine.deny_still_possible_for(&x)); + } + #[test] fn a_destination_scoped_rule_is_never_a_process_wide_answer() { // The rule matches this process, but only for port 443. Precomputing diff --git a/crates/cfc-daemon/src/storage.rs b/crates/cfc-daemon/src/storage.rs index b1deb09..9b8ad08 100644 --- a/crates/cfc-daemon/src/storage.rs +++ b/crates/cfc-daemon/src/storage.rs @@ -828,6 +828,18 @@ mod tests { }, ); assert_eq!(super::quarantine_reason(&scoped), None); + + // `/0` adds no specificity since 0.8.0 but is still a scope: a rule + // stored with only `--dst-net 0.0.0.0/0` keeps loading. + let slash_zero = Rule::new( + "every ipv4 destination", + cfc_core::Action::Deny, + cfc_core::RuleScope { + dst_net: Some("0.0.0.0/0".parse().unwrap()), + ..cfc_core::RuleScope::any() + }, + ); + assert_eq!(super::quarantine_reason(&slash_zero), None); } #[test] diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 06aaf2d..50d9f98 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -251,15 +251,30 @@ migration cannot recover that intent. This is a policy-entry contract; it does not pin an inode, follow aliases at exec time, or attest future pathname changes. Legacy alias intent loss is not repaired by this validation. -`RuleSet` is kept sorted so that lookup is a linear scan that returns the -first match, and the order does not depend on what SQLite happened to -return: +`RuleSet` is kept sorted so that lookup is a linear scan, and the order does +not depend on what SQLite happened to return: -1. specificity descending (how many scope predicates are set) +1. specificity descending (how many scope predicates are set; a `/0` + network is not counted, since it narrows nothing within its address + family) 2. Deny, then Reject, then Allow 3. oldest `created_at` first 4. `id`, as a total-order tiebreak +One override is applied during the scan rather than in the sort: a Deny or +Reject rule that names a program (`exe_path` or `exe_sha256`) wins over +every Allow rule that names none, whatever their predicate counts. Rules that +name a program keep their specificity order among themselves, so +`allow --exe X --dst-port 443` still beats `deny --exe X`. Everything else +keeps the order above. The scan holds the first matching generic Allow +instead of returning it, and only a lower program rule can still change the +answer: a program Deny or Reject replaces it, a program Allow leaves it. The +relation is not a total order (program Allow 3 > program Deny 2 > generic +Allow 5 > generic Deny 4 > program Allow 3), so no sort key could express +it. The in-kernel precompute (`Engine::process_wide_action` and +`deny_still_possible_for`) walks the same way, so the connect hooks and the +packet path agree. + Disabled and expired rules are filtered at lookup, so a `Seconds(n)` rule stops matching the instant it expires rather than when the reaper next runs. A 30-second maintenance task flushes hit counts to disk and deletes expired diff --git a/docs/HARDENING.md b/docs/HARDENING.md index c1a608d..8f60d59 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -175,6 +175,15 @@ Two caveats, both real: protocol` is much safer than `exe` alone - if a process is later compromised, the attacker still can't pivot to arbitrary destinations. +**A program Deny beats a generic Allow.** A Deny or Reject rule scoped to +a program (`--exe` or `--sha256`) wins over every Allow rule that names no +program, however many other predicates that Allow carries: `deny --exe +/opt/agent` holds even next to `allow --protocol tcp --dst-port 443`. Among +rules that name a program the more specific one still wins, so an +`allow --exe X --dst-port 443` carves an exception out of `deny --exe X`. +A `/0` network (`--dst-net 0.0.0.0/0`) does not count as a predicate when +rules are ranked. + **Watch the hit counter.** `cfc rules list` shows `hits` per rule. A rule with zero hits after weeks of use is probably obsolete or wrong. diff --git a/docs/ROADMAP.md b/docs/ROADMAP.md index 5ddd398..f78cb91 100644 --- a/docs/ROADMAP.md +++ b/docs/ROADMAP.md @@ -75,7 +75,9 @@ were wrong or missing once the happy path worked. ### Rule semantics - [x] Deterministic precedence (specificity, then Deny > Reject > - Allow, then created_at, then id) + Allow, then created_at, then id; a program-scoped Deny or Reject + beats any Allow that names no program, and `/0` adds no + specificity) - [x] `Duration` enforced at lookup; expired rules reaped periodically - [x] `Once` / `UntilRestart` purged at startup; persisting `Once` refused - [x] Forward-compatible rule serialization + frozen v0.1.0 fixtures diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index b468dcf..f261d06 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -668,6 +668,20 @@ journalctl -u colony-firewalld -g 'fails the API boundary' Remove it with `cfc rules remove ` (the full id) and re-create it in a form the daemon accepts. `cfc rules import --replace` also deletes such rows. +## An Allow rule no longer lets a program through + +Since 0.8.0 a Deny or Reject rule scoped to a program (`--exe` or +`--sha256`) wins over every Allow rule that names no program, however many +predicates that Allow has. A broad `allow --protocol tcp --dst-port 443` +therefore no longer admits a program that `deny --exe` refuses, and the +connect hooks may refuse that program outright. To let it through on some +destinations, add an Allow scoped to the same program and those +destinations (`--exe X --dst-port 443`): among program rules the more +specific one still wins. A `/0` network no longer counts when rules are +ranked either, so a rule that relied on `--dst-net 0.0.0.0/0` to outrank +another may now lose the tie to a Deny. `cfc rules list` shows the rules +involved; the hit counters show which one answers. + ## Where things live | Thing | Path | From 469bd761d402c75b5b35091142af521d33e40869 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:13:02 +0200 Subject: [PATCH 115/125] feat(daemon)!: prompt instead of refusing flows whose identity is incomplete A rule that tests an executable, uid or digest the daemon could not establish (unattributed sockets, ambiguous UDP, an expired attribution budget, a binary over the hashing cap) used to turn the flow into a silent Deny before pause, loopback or prompt logic ran. One "allow this program" rule therefore blocked every unattributed flow a generic rule allowed. RuleSet::lookup now walks past such rules and remembers whether they allow or refuse. The definite winner answers when every resolution agrees with it; otherwise the result is Undecidable and the engine returns NeedsPrompt naming the rule. The prompt carries PromptEvent.undecided_rule_id, follows no_ui_action with no answering UI and the existing caps, and is not lifted by pause or the loopback allowance. The user's answer now applies instead of being overridden by the per-packet re-check. Events and the live feed record the undecided rule in rule_id with a non-rule source. Legacy hostname rules that cannot be decided are still refused. UDP attribution failures log their reason at debug level. --- README.md | 3 +- crates/cfc-cli/src/prompts.rs | 1 + crates/cfc-cli/tests/cli_e2e.rs | 1 + crates/cfc-core/src/rule.rs | 240 +++++++++++++++++---- crates/cfc-daemon/examples/prompt_demo.rs | 1 + crates/cfc-daemon/src/convert.rs | 76 ++++++- crates/cfc-daemon/src/decision.rs | 102 ++++++--- crates/cfc-daemon/src/ipc.rs | 25 +-- crates/cfc-daemon/src/nfqueue.rs | 221 ++++++++++++++++--- crates/cfc-daemon/src/process_resolve.rs | 20 +- crates/cfc-daemon/src/prompts.rs | 37 ++++ crates/cfc-daemon/tests/ipc_integration.rs | 20 +- crates/cfc-proto/proto/cfc.proto | 11 + crates/cfc-ui/src/main.rs | 1 + crates/cfc-ui/src/views/prompts.rs | 1 + docs/ARCHITECTURE.md | 31 ++- docs/HARDENING.md | 18 +- docs/TROUBLESHOOTING.md | 24 +++ 18 files changed, 692 insertions(+), 141 deletions(-) diff --git a/README.md b/README.md index cbcea35..75fd5f2 100644 --- a/README.md +++ b/README.md @@ -308,7 +308,8 @@ by default, so a `--network=host` container has it; drop it with `--cap-drop NET_RAW` for workloads CFC should govern. Executables over 64 MiB (Chromium, Electron apps, VS Code) are never hashed. -A hash-pinned rule naming one refuses its flows, and outside a root-owned +A hash-pinned rule naming one cannot be decided, so its flows are prompted +(or take `no_ui_action` with no UI connected), and outside a root-owned path an "Allow always" for one cannot be saved, so it prompts for every new flow. See [docs/HARDENING.md](docs/HARDENING.md#rule-design-principles). The complete list of non-goals is in diff --git a/crates/cfc-cli/src/prompts.rs b/crates/cfc-cli/src/prompts.rs index c4f637f..9f1ddf2 100644 --- a/crates/cfc-cli/src/prompts.rs +++ b/crates/cfc-cli/src/prompts.rs @@ -948,6 +948,7 @@ mod tests { process: Some(process()), deadline_unix_ms: 1_700_000_030_000, binds_to_hash: false, + undecided_rule_id: String::new(), }; let v = serde_json::to_value(to_json(&ev, None, None)).unwrap(); assert_eq!(v["prompt_id"], "17"); diff --git a/crates/cfc-cli/tests/cli_e2e.rs b/crates/cfc-cli/tests/cli_e2e.rs index d2dbf90..1a7555b 100644 --- a/crates/cfc-cli/tests/cli_e2e.rs +++ b/crates/cfc-cli/tests/cli_e2e.rs @@ -88,6 +88,7 @@ impl Firewall for FakeDaemon { // Far enough out that a slow CI box cannot expire it. deadline_unix_ms: chrono::Utc::now().timestamp_millis() + 60_000, binds_to_hash: false, + undecided_rule_id: String::new(), }; let _ = tx.send(Ok(ev)).await; // Hold the stream open; the CLI is expected to leave on its own diff --git a/crates/cfc-core/src/rule.rs b/crates/cfc-core/src/rule.rs index cfe6f2b..ba7536c 100644 --- a/crates/cfc-core/src/rule.rs +++ b/crates/cfc-core/src/rule.rs @@ -623,18 +623,29 @@ impl RuleSet { /// `now_unix_ms` is the current wall-clock time; rules whose /// `Duration::Seconds(..)` window has elapsed are skipped (see /// [`Rule::is_expired`]). - /// Missing identity cannot yield to a lower Allow. A definite refusal - /// may still answer after earlier possible refusals. See [`Match`]. + /// + /// A rule this process cannot be checked against (see + /// [`RuleScope::undecidable_for`]) or a legacy hostname rule is walked + /// past and remembered. The definite winner then answers only if every + /// way those rules could resolve gives the same action; otherwise the + /// answer is [`Match::Undecidable`]. See [`Match`]. pub fn lookup( &self, conn: &crate::Connection, proc: &crate::Process, now_unix_ms: i64, ) -> Match<'_> { - let mut undecidable = None; + // The rule reported when the outcome is open. A legacy hostname rule + // takes the slot from an identity one: the caller refuses it rather + // than prompting. + let mut undecided: Option<&Rule> = None; + // Whether an undecided Allow, or an undecided Deny/Reject, was passed. + let mut maybe_allow = false; + let mut maybe_closed = false; // The first definite Allow that names no program. Held, not returned: // a program Deny ranked below it may still override it. let mut generic_allow: Option<&Rule> = None; + let mut winner = None; for rule in self .rules .iter() @@ -660,45 +671,41 @@ impl RuleSet { // application's intended hostname or enumerates every alias. // Keep a legacy name at its original priority as uncertainty. if missing_process || rule.scope.dst_host.is_some() { + if undecided + .is_none_or(|u| u.scope.dst_host.is_none() && rule.scope.dst_host.is_some()) + { + undecided = Some(rule); + } if rule.action == Action::Allow { - // Under a held generic Allow, a possible program Allow - // cannot change the answer: if it matched, the generic - // Allow would still win. - if generic_allow.is_some() { - continue; - } - // A possible Allow keeps its precedence. - return Match::Undecidable(undecidable.unwrap_or(rule)); + maybe_allow = true; + } else { + maybe_closed = true; } - // Only a sequence of possible closed actions may yield to a - // definite Deny. - undecidable.get_or_insert(rule); continue; } - let winner = match generic_allow { + winner = Some(match generic_allow { // A lower program Allow cannot turn the answer into a refusal. Some(held) if rule.action == Action::Allow => held, - Some(_) => rule, None if rule.action == Action::Allow && !rule.scope.names_program() => { - if let Some(first) = undecidable { - return Match::Undecidable(first); - } generic_allow = Some(rule); continue; } - None => rule, - }; - if winner.action == Action::Allow { - if let Some(first) = undecidable { - return Match::Undecidable(first); - } - } - return Match::Rule(winner); + _ => rule, + }); + break; } - match (undecidable, generic_allow) { - (Some(first), _) => Match::Undecidable(first), - (None, Some(held)) => Match::Rule(held), - (None, None) => Match::None, + let Some(winner) = winner.or(generic_allow) else { + return undecided.map_or(Match::None, Match::Undecidable); + }; + // Every resolution of the undecided rules gives the winner's action. + let settled = if winner.action == Action::Allow { + !maybe_closed + } else { + !maybe_allow + }; + match undecided { + Some(first) if !settled => Match::Undecidable(first), + _ => Match::Rule(winner), } } } @@ -706,8 +713,10 @@ impl RuleSet { /// What [`RuleSet::lookup`] found. /// /// Three outcomes, not two, and the third is the whole point. Precedence is -/// ordered, so a rule that cannot be decided must not be walked past: the -/// rules beneath it are the ones its author wrote it to override. +/// ordered, so a rule that cannot be decided must not be ignored: the rules +/// beneath it are the ones its author wrote it to override. When it and the +/// rule that would otherwise answer agree on the action, that rule answers; +/// when they disagree, nobody does. /// /// The case in the field is a `deny` scoped to `exe_sha256` over a binary the /// daemon cannot hash - over 64 MiB, unreadable, or a process whose image is @@ -722,9 +731,11 @@ impl RuleSet { pub enum Match<'a> { /// This rule answered. Rule(&'a Rule), - /// This rule could apply, but its process identity or hostname cannot be decided. - /// The caller must refuse conservatively; a prompt or permissive fallback - /// cannot establish the missing identity. + /// This rule could apply, but the process identity (exe, uid or digest) + /// or a legacy hostname cannot be decided, and the rules that could apply + /// disagree. The caller asks the user, with the identity shown as unknown; + /// a legacy hostname rule (`dst_host`) is refused instead, since no answer + /// can establish the name. Undecidable(&'a Rule), /// No rule is about this flow. None, @@ -759,7 +770,7 @@ mod tests { } #[test] - fn legacy_hostname_policy_never_authorizes_or_yields_to_a_lower_allow() { + fn legacy_hostname_policy_never_yields_to_a_disagreeing_lower_allow() { for action in [Action::Allow, Action::Deny, Action::Reject] { let mut named = RuleScope::any(); named.exe_path = Some("/usr/bin/curl".into()); @@ -789,10 +800,21 @@ mod tests { let mut conn = mk_conn(); conn.dst_host = host.map(str::to_owned); conn.dst_host_verified = verified; - assert!( - matches!(set.lookup(&conn, &proc, now()), Match::Undecidable(r) if r.name == "legacy-name"), - "{action:?} {host:?} verified={verified}" - ); + // A legacy Allow over a lower Allow: both resolutions + // allow, so the lower rule answers. A legacy refusal over + // it stays open, whatever the PTR says. + let result = set.lookup(&conn, &proc, now()); + if action == Action::Allow { + assert!( + matches!(result, Match::Rule(r) if r.name == "lower-allow"), + "{host:?} verified={verified}" + ); + } else { + assert!( + matches!(result, Match::Undecidable(r) if r.name == "legacy-name"), + "{action:?} {host:?} verified={verified}" + ); + } assert!(!set.rules[0].scope.matches(&conn, &proc)); } } @@ -1416,6 +1438,142 @@ mod tests { ); } + fn exe(path: &str) -> RuleScope { + RuleScope { + exe_path: Some(path.into()), + ..RuleScope::any() + } + } + + #[test] + fn an_undecidable_allow_yields_to_a_definite_allow_below() { + // One "allow this program" rule used to make every unattributed flow + // undecidable, so ICMP that a generic rule allows was refused. + let set = sorted(vec![ + scoped("allow-firefox", Action::Allow, exe("/usr/bin/firefox")), + scoped( + "allow-icmp", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Icmp), + ..RuleScope::any() + }, + ), + ]); + let ping = Connection { + protocol: Protocol::Icmp, + src_port: 0, + dst_port: 0, + ..mk_conn() + }; + assert_eq!( + winner(&set, &ping, &Process::unknown(0)), + Some("allow-icmp") + ); + } + + #[test] + fn undecidable_allow_over_a_definite_deny_is_undecidable() { + let set = sorted(vec![ + scoped( + "allow-firefox", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + ..exe("/usr/bin/firefox") + }, + ), + scoped( + "deny-https", + Action::Deny, + RuleScope { + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert!(matches!( + set.lookup(&mk_conn(), &Process::unknown(0), now()), + Match::Undecidable(r) if r.name == "allow-firefox" + )); + } + + #[test] + fn undecidable_deny_over_a_definite_allow_is_undecidable() { + let set = sorted(vec![ + scoped("deny-x", Action::Deny, exe("/usr/bin/x")), + scoped( + "allow-https", + Action::Allow, + RuleScope { + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert!(matches!( + set.lookup(&mk_conn(), &Process::unknown(0), now()), + Match::Undecidable(r) if r.name == "deny-x" + )); + // With no rule that disagrees, the refusal is settled either way. + let deny_only = sorted(vec![ + scoped("deny-x", Action::Deny, exe("/usr/bin/x")), + scoped( + "deny-https", + Action::Deny, + RuleScope { + dst_port: Some(443), + ..RuleScope::any() + }, + ), + ]); + assert_eq!( + winner(&deny_only, &mk_conn(), &Process::unknown(0)), + Some("deny-https") + ); + } + + #[test] + fn an_undecidable_program_allow_under_a_generic_allow_keeps_a_program_deny_open() { + // allow --sha256 H --dst-port 443 ranks above deny --exe /usr/bin/y. + // If the image hashes to H, that Allow ends the walk under the generic + // Allow and the answer is Allow; if not, the program Deny wins. With + // no digest, neither can be claimed. + let set = sorted(vec![ + scoped( + "allow-net", + Action::Allow, + RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + dst_net: Some("1.2.3.0/24".parse().unwrap()), + ..RuleScope::any() + }, + ), + scoped( + "allow-h", + Action::Allow, + RuleScope { + exe_sha256: Some("aa".repeat(32)), + dst_port: Some(443), + protocol: Some(Protocol::Tcp), + ..RuleScope::any() + }, + ), + scoped("deny-y", Action::Deny, exe("/usr/bin/y")), + ]); + let unhashed = mk_proc("/usr/bin/y"); + assert!(matches!( + set.lookup(&mk_conn(), &unhashed, now()), + Match::Undecidable(r) if r.name == "allow-h" + )); + let mut hashed = mk_proc("/usr/bin/y"); + hashed.sha256 = Some("aa".repeat(32)); + assert_eq!(winner(&set, &mk_conn(), &hashed), Some("allow-net")); + hashed.sha256 = Some("cc".repeat(32)); + assert_eq!(winner(&set, &mk_conn(), &hashed), Some("deny-y")); + } + #[test] fn an_unknown_process_under_a_program_deny_and_a_generic_allow_is_undecidable() { // The program Deny could be the one that overrides the Allow, and the diff --git a/crates/cfc-daemon/examples/prompt_demo.rs b/crates/cfc-daemon/examples/prompt_demo.rs index e005199..3a18a81 100644 --- a/crates/cfc-daemon/examples/prompt_demo.rs +++ b/crates/cfc-daemon/examples/prompt_demo.rs @@ -132,6 +132,7 @@ async fn main() -> anyhow::Result<()> { prompt_id, connection, process, + undecided: None, }) .await .is_err() diff --git a/crates/cfc-daemon/src/convert.rs b/crates/cfc-daemon/src/convert.rs index c9611f9..8863252 100644 --- a/crates/cfc-daemon/src/convert.rs +++ b/crates/cfc-daemon/src/convert.rs @@ -404,10 +404,9 @@ pub fn verdict_to_pb_action(v: &Verdict) -> pb::Action { /// Flattens a decided flow into the row shape persisted by /// [`crate::storage::RuleStore::insert_events`]. pub fn event_row_from_observed( - conn: &Connection, - proc: &Process, - verdict: &Verdict, + obs: &crate::nfqueue::ObservedConnection, ) -> crate::storage::EventRow { + let (conn, proc, verdict) = (&obs.connection, &obs.process, &obs.verdict); crate::storage::EventRow { id: 0, ts_unix_ms: conn.timestamp.timestamp_millis(), @@ -422,10 +421,7 @@ pub fn event_row_from_observed( uid: proc.uid, action: action_db_str(verdict.action).to_string(), source: verdict_source_db_str(&verdict.source).to_string(), - rule_id: match verdict.source { - cfc_core::VerdictSource::Rule(id) => Some(id.to_string()), - _ => None, - }, + rule_id: obs.rule_id().map(|id| id.to_string()), } } @@ -921,7 +917,7 @@ mod tests { let rule_id = uuid::Uuid::new_v4(); let verdict = cfc_core::Verdict::deny_from_rule(rule_id); - let row = event_row_from_observed(&conn, &proc, &verdict); + let row = event_row_from_observed(&observed(conn, proc, verdict, None)); assert_eq!(row.action, "Deny"); assert_eq!(row.source, "rule"); assert_eq!(row.rule_id.as_deref(), Some(rule_id.to_string().as_str())); @@ -948,13 +944,75 @@ mod tests { 2, ); let proc = cfc_core::Process::unknown(7); - let row = event_row_from_observed(&conn, &proc, &cfc_core::Verdict::default_allow()); + let row = event_row_from_observed(&observed( + conn, + proc, + cfc_core::Verdict::default_allow(), + None, + )); assert_eq!(row.uid, None); assert_eq!(row.source, "default"); assert_eq!(row.rule_id, None); assert_eq!(event_row_to_pb(&row).uid, None); } + fn observed( + connection: cfc_core::Connection, + process: cfc_core::Process, + verdict: cfc_core::Verdict, + undecided: Option, + ) -> crate::nfqueue::ObservedConnection { + crate::nfqueue::ObservedConnection { + connection, + process, + verdict, + undecided, + } + } + + #[test] + fn an_undecided_outcome_records_the_rule_id() { + let conn = cfc_core::Connection::new( + Protocol::Udp, + Direction::Outbound, + IpAddr::V4(Ipv4Addr::new(10, 0, 0, 2)), + 5353, + IpAddr::V4(Ipv4Addr::new(1, 1, 1, 1)), + 53, + ); + let undecided = uuid::Uuid::new_v4(); + for verdict in [ + cfc_core::Verdict::from_policy(Action::Deny), + cfc_core::Verdict::default_allow(), + cfc_core::Verdict { + action: Action::Deny, + source: cfc_core::VerdictSource::Timeout, + }, + cfc_core::Verdict { + action: Action::Allow, + source: cfc_core::VerdictSource::UserPrompt, + }, + ] { + let row = event_row_from_observed(&observed( + conn.clone(), + cfc_core::Process::unknown(0), + verdict, + Some(undecided), + )); + assert_ne!(row.source, "rule"); + assert_eq!(row.rule_id.as_deref(), Some(undecided.to_string().as_str())); + } + // A rule that answered is recorded as itself, not as undecided. + let answered = uuid::Uuid::new_v4(); + let row = event_row_from_observed(&observed( + conn, + cfc_core::Process::unknown(0), + cfc_core::Verdict::deny_from_rule(answered), + Some(undecided), + )); + assert_eq!(row.rule_id.as_deref(), Some(answered.to_string().as_str())); + } + #[test] fn event_row_to_pb_tolerates_unknown_action_strings() { let row = crate::storage::EventRow { diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index ae5feac..ff57289 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -50,8 +50,15 @@ struct EngineInner { pub enum Decision { /// A persistent rule matched. Return the verdict immediately. Resolved(Verdict), - /// No rule matched. Caller should prompt the user. - NeedsPrompt { fallback: Verdict }, + /// No rule answered. Caller should prompt the user, and apply `fallback` + /// (no_ui_action) when nobody can be asked. + NeedsPrompt { + fallback: Verdict, + /// A rule that may apply but cannot be decided, because the process + /// identity (exe, uid or digest) is incomplete. `None` when no rule is + /// about this flow. + undecided: Option, + }, } impl Engine { @@ -159,30 +166,33 @@ impl Engine { /// must not credit the rule it did not follow. pub fn peek(&self, conn: &Connection, proc: &Process) -> Decision { let now_unix_ms = chrono::Utc::now().timestamp_millis(); - let rule_match = { - let rules = self.inner.rules.read(); - match rules.lookup(conn, proc, now_unix_ms) { - cfc_core::rule::Match::Rule(r) => Some((r.id, r.action)), - // Missing identity cannot authorize traffic or be overridden - // by pause, a permissive fallback, or a prompt response. - cfc_core::rule::Match::Undecidable(r) => { - tracing::debug!( - rule = %r.name, - exe = %proc.exe.display(), - "policy identity is incomplete; refusing this flow" - ); - return Decision::Resolved(Verdict::from_policy(cfc_core::Action::Deny)); - } - cfc_core::rule::Match::None => None, - } - }; - if let Some((rule_id, action)) = rule_match { + let rules = self.inner.rules.read(); + let undecided = match rules.lookup(conn, proc, now_unix_ms) { // Verbatim: a Reject rule must reach the datapath as Reject so // the refusal is actually injected, not silently downgraded. - return Decision::Resolved(Verdict::from_rule(action, rule_id)); - } + cfc_core::rule::Match::Rule(r) => { + return Decision::Resolved(Verdict::from_rule(r.action, r.id)); + } + // No answer can establish a legacy hostname, so it stays refused. + cfc_core::rule::Match::Undecidable(r) if r.scope.dst_host.is_some() => { + tracing::debug!(rule = %r.name, "legacy hostname rule cannot be decided; refusing this flow"); + return Decision::Resolved(Verdict::from_policy(cfc_core::Action::Deny)); + } + // Incomplete identity is uncertainty, not a refusal: ask the user. + cfc_core::rule::Match::Undecidable(r) => { + tracing::debug!( + rule = %r.name, + exe = %proc.exe.display(), + "process identity is incomplete; prompting for this flow" + ); + Some(r.id) + } + cfc_core::rule::Match::None => None, + }; + drop(rules); Decision::NeedsPrompt { fallback: self.fallback_verdict(), + undecided, } } @@ -1150,13 +1160,55 @@ mod tests { fn no_rules_returns_needs_prompt() { let engine = Engine::new(RuleSet::default(), shared(dp_allow())); match engine.evaluate(&conn(443), &proc("/usr/bin/curl")) { - Decision::NeedsPrompt { fallback } => { + Decision::NeedsPrompt { fallback, .. } => { assert_eq!(fallback.action, Action::Allow); } _ => panic!("expected NeedsPrompt"), } } + #[test] + fn incomplete_identity_needs_a_prompt_naming_the_rule() { + let mut scope = RuleScope::any(); + scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); + let rule = Rule::new("deny-curl", Action::Deny, scope); + let id = rule.id; + let engine = Engine::new(RuleSet { rules: vec![rule] }, shared(dp_allow())); + match engine.evaluate(&conn(443), &Process::unknown(0)) { + Decision::NeedsPrompt { + fallback, + undecided, + } => { + assert_eq!(undecided, Some(id)); + assert_eq!(fallback, Verdict::from_policy(Action::Allow)); + } + _ => panic!("expected NeedsPrompt"), + } + assert!(matches!( + engine.evaluate(&conn(443), &proc("/usr/bin/wget")), + Decision::NeedsPrompt { + undecided: None, + .. + } + )); + } + + #[test] + fn legacy_hostname_policy_still_refuses() { + let mut scope = RuleScope::any(); + scope.dst_host = Some("example.org".into()); + let engine = Engine::new( + RuleSet { + rules: vec![Rule::new("legacy", Action::Deny, scope)], + }, + shared(dp_allow()), + ); + match engine.evaluate(&conn(443), &proc("/usr/bin/curl")) { + Decision::Resolved(v) => assert_eq!(v, Verdict::from_policy(Action::Deny)), + _ => panic!("expected Resolved"), + } + } + #[test] fn matching_allow_rule_resolves_to_allow() { let mut rs = RuleSet::default(); @@ -1186,7 +1238,7 @@ mod tests { fn fallback_respects_default_policy_deny() { let engine = Engine::new(RuleSet::default(), shared(dp_deny())); match engine.evaluate(&conn(443), &proc("/usr/bin/curl")) { - Decision::NeedsPrompt { fallback } => { + Decision::NeedsPrompt { fallback, .. } => { assert_eq!(fallback.action, Action::Deny); } _ => panic!("expected NeedsPrompt"), @@ -1261,7 +1313,7 @@ mod tests { *policy.write().unwrap() = dp_deny(); assert_eq!(engine.fallback_verdict().action, Action::Deny); match engine.evaluate(&conn(443), &proc("/usr/bin/curl")) { - Decision::NeedsPrompt { fallback } => { + Decision::NeedsPrompt { fallback, .. } => { assert_eq!(fallback.action, Action::Deny); } _ => panic!("expected NeedsPrompt"), diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 0298b30..3dcefe9 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -1047,10 +1047,7 @@ impl Firewall for FirewallService { connection: Some(convert::connection_to_pb(&obs.connection)), process: Some(convert::process_to_pb(&obs.process)), verdict: convert::verdict_to_pb_action(&obs.verdict) as i32, - rule_id: match obs.verdict.source { - cfc_core::VerdictSource::Rule(id) => id.to_string(), - _ => String::new(), - }, + rule_id: obs.rule_id().map(|id| id.to_string()).unwrap_or_default(), }; if tx.send(Ok(ev)).await.is_err() { break; @@ -1450,11 +1447,7 @@ pub fn spawn_event_pipeline( if obs.verdict.action != cfc_core::Action::Allow { continue; } - feeder.push(convert::event_row_from_observed( - &obs.connection, - &obs.process, - &obs.verdict, - )); + feeder.push(convert::event_row_from_observed(&obs)); } Err(broadcast::error::RecvError::Lagged(n)) => { count_dropped( @@ -2052,6 +2045,7 @@ mod tests { 443, ), process, + undecided: None, }) .await .unwrap(); @@ -2435,6 +2429,7 @@ mod tests { action, source: cfc_core::VerdictSource::DefaultPolicy, }, + undecided: None, } } @@ -2448,11 +2443,7 @@ mod tests { // The worker queues a refusal itself and then publishes it; the // feeder must not record it a second time. let blocked = observed(80, cfc_core::Action::Deny); - sink.push(convert::event_row_from_observed( - &blocked.connection, - &blocked.process, - &blocked.verdict, - )); + sink.push(convert::event_row_from_observed(&blocked)); tx.send(blocked).unwrap(); // Well past the batch interval; paused time auto-advances. @@ -2518,11 +2509,7 @@ mod tests { let (tx, _rx) = broadcast::channel(64); let sink = spawn_event_pipeline(store.clone(), &tx, 1000); let blocked = observed(80, cfc_core::Action::Deny); - let row = convert::event_row_from_observed( - &blocked.connection, - &blocked.process, - &blocked.verdict, - ); + let row = convert::event_row_from_observed(&blocked); for _ in 0..3 { sink.push(row.clone()); } diff --git a/crates/cfc-daemon/src/nfqueue.rs b/crates/cfc-daemon/src/nfqueue.rs index 3dba252..14d7fce 100644 --- a/crates/cfc-daemon/src/nfqueue.rs +++ b/crates/cfc-daemon/src/nfqueue.rs @@ -187,6 +187,9 @@ pub struct PromptRequest { pub prompt_id: u64, pub connection: Connection, pub process: Process, + /// A rule that may apply but cannot be decided, because the process + /// identity is incomplete. See [`Decision::NeedsPrompt`]. + pub undecided: Option, } /// A resolved prompt flowing back from the router to the worker. @@ -202,6 +205,21 @@ pub struct ObservedConnection { pub connection: Connection, pub process: Process, pub verdict: Verdict, + /// The rule that could not be decided for this flow, when that is why it + /// was prompted. See [`Decision::NeedsPrompt`]. + pub undecided: Option, +} + +impl ObservedConnection { + /// The rule recorded with this flow: the one that answered, or else the + /// one that could not be decided. A rule id with a source other than + /// `rule` therefore means "this rule could not be decided". + pub fn rule_id(&self) -> Option { + match self.verdict.source { + cfc_core::VerdictSource::Rule(id) => Some(id), + _ => self.undecided, + } + } } /// Logs a refusal to the journal, then publishes to the bounded, lossy live @@ -215,6 +233,7 @@ pub fn publish_observation(tx: &broadcast::Sender, obs: Obse pid = obs.process.pid, uid = ?obs.process.uid, dst = %format_args!("{}:{}", obs.connection.dst_ip, obs.connection.dst_port), + undecided_rule = ?obs.undecided, "connection blocked" ); } @@ -734,6 +753,7 @@ impl Worker { // Per packet, not per prompt: a Reject response is derived from // the individual segment (its sequence numbers, its source // port), and parallel connections share one prompt. + let mut undecided = None; let verdict = match self.engine.peek(&packet.connection, &packet.process) { // A refusal decided since the prompt opened always wins. Decision::Resolved(current) if current.action != Action::Allow => current, @@ -750,6 +770,12 @@ impl Worker { { current } + // The user's answer applies to a flow whose identity is + // incomplete: no rule can decide it, that is why it was asked. + Decision::NeedsPrompt { undecided: u, .. } => { + undecided = u; + pv.verdict + } _ => pv.verdict, }; // Only a rule whose answer was applied gets the hit. @@ -762,6 +788,7 @@ impl Worker { connection: packet.connection, process: packet.process, verdict, + undecided, }, )?; } @@ -795,13 +822,15 @@ impl Worker { connection, process, verdict, + undecided: None, }, ), PacketOutcome::Prompt { connection, process, fallback, - } => self.park_for_prompt(msg, connection, process, fallback), + undecided, + } => self.park_for_prompt(msg, connection, process, fallback, undecided), } } @@ -815,6 +844,7 @@ impl Worker { connection: Connection, process: Process, fallback: Verdict, + undecided: Option, ) -> anyhow::Result<()> { let flow = FlowKey::for_flow(&connection, &process); if let Some(&prompt_id) = self.pending_flows.get(&flow) { @@ -834,7 +864,7 @@ impl Worker { parked = pending.packets.len(), "prompt already holds its packet cap; applying the fallback" ); - return self.deliver_fallback(msg, connection, process, fallback); + return self.deliver_fallback(msg, connection, process, fallback, undecided); } trace!( prompt_id, @@ -876,7 +906,7 @@ impl Worker { parked = self.waiters.len(), "prompt backlog at its cap; applying the fallback rather than parking" ); - return self.deliver_fallback(msg, connection, process, fallback); + return self.deliver_fallback(msg, connection, process, fallback, undecided); } let prompt_id = self.next_prompt_id; @@ -888,6 +918,7 @@ impl Worker { prompt_id, connection: connection.clone(), process: process.clone(), + undecided, }; match self.prompt_tx.try_send(req) { Ok(()) => { @@ -909,7 +940,7 @@ impl Worker { // Router saturated or gone: apply the default policy now // rather than stranding the packet. trace!("prompt channel unavailable ({e}); applying fallback"); - return self.deliver_fallback(msg, connection, process, fallback); + return self.deliver_fallback(msg, connection, process, fallback, undecided); } } Ok(()) @@ -929,6 +960,7 @@ impl Worker { connection: Connection, process: Process, fallback: Verdict, + undecided: Option, ) -> anyhow::Result<()> { self.deliver( msg, @@ -936,6 +968,7 @@ impl Worker { connection, process, verdict: fallback, + undecided, }, ) } @@ -953,11 +986,8 @@ impl Worker { if obs.verdict.action == Action::Allow { self.dns.enqueue(obs.connection.dst_ip); } else { - self.events.push(crate::convert::event_row_from_observed( - &obs.connection, - &obs.process, - &obs.verdict, - )); + self.events + .push(crate::convert::event_row_from_observed(&obs)); } record(&self.stats, obs.verdict.action); publish_observation(&self.observed_tx, obs); @@ -1101,13 +1131,15 @@ enum PacketOutcome { process: Process, verdict: Verdict, }, - /// No rule matched and prompting is enabled: ask the user, applying + /// No rule answered and prompting is enabled: ask the user, applying /// `fallback` if no prompt can be delivered. Stats are recorded when - /// the prompt resolves. + /// the prompt resolves. `undecided` names the rule that may apply but + /// could not be decided, if that is why. Prompt { connection: Connection, process: Process, fallback: Verdict, + undecided: Option, }, } @@ -1206,7 +1238,10 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack process: proc, verdict, }, - Decision::NeedsPrompt { fallback } => { + Decision::NeedsPrompt { + fallback, + undecided, + } => { // Inbound never asks. // // The decision the owner took, and it is not a shortcut: nothing @@ -1234,15 +1269,16 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack }; } // Preserve desktop IPC for unmatched local flows after explicit - // policy and incomplete-identity refusals have had their say. - if meta.loopback { + // policy has had its say. Neither this nor pause lifts a flow a + // rule may be about: its identity is incomplete, so it is asked. + if meta.loopback && undecided.is_none() { return PacketOutcome::Deliver { connection: conn, process: proc, verdict: Verdict::default_allow(), }; } - if deps.stats.is_paused() { + if deps.stats.is_paused() && undecided.is_none() { // Paused means "stop prompting", not "stop filtering": // rules above still applied; only unmatched flows pass // without a prompt. @@ -1258,6 +1294,7 @@ fn handle_packet(payload: &[u8], meta: &PacketMeta, deps: &PipelineDeps) -> Pack connection: conn, process: proc, fallback, + undecided, } } } @@ -1502,6 +1539,7 @@ mod tests { conn_to(1024 + i as u16, 1111), test_process(4242, "/usr/bin/curl"), Verdict::default_deny(), + None, ) .unwrap(); // Drain as we go: the harness channel holds 16, and a full channel @@ -1520,6 +1558,7 @@ mod tests { conn_to(9000, 1111), test_process(4242, "/usr/bin/curl"), Verdict::default_deny(), + None, ) .unwrap(); @@ -1572,6 +1611,7 @@ mod tests { conn_to(443, 1111 + i as u16), test_process(4242, "/usr/bin/curl"), Verdict::default_deny(), + None, ) .unwrap(); } @@ -1587,6 +1627,7 @@ mod tests { conn_to(443, 9999), test_process(4242, "/usr/bin/curl"), Verdict::default_deny(), + None, ) .unwrap(); @@ -1888,6 +1929,7 @@ mod tests { connection, process, fallback, + .. } => { assert_eq!(fallback.action, Action::Deny); assert_eq!(connection.pid, Some(4242)); @@ -2186,6 +2228,7 @@ mod tests { first, process.clone(), Verdict::default_deny(), + None, ) .unwrap(); h.worker() @@ -2194,6 +2237,7 @@ mod tests { second, process, Verdict::default_deny(), + None, ) .unwrap(); let request = h.prompt_rx.try_recv().unwrap(); @@ -2212,16 +2256,79 @@ mod tests { ); } + fn exe_rule(action: Action, exe: &str) -> Rule { + let mut scope = RuleScope::any(); + scope.exe_path = Some(PathBuf::from(exe)); + Rule::new(format!("{action:?} {exe}"), action, scope) + } + + /// Minimal IPv4/UDP packet: 1.2.3.4:5555 -> 5.6.7.8:`dst_port`. + fn udp_packet(dst_port: u16) -> Vec { + let mut pkt = tcp_packet(dst_port); + pkt.truncate(28); + pkt[9] = 17; + pkt + } + + /// The process could not be attributed: no socket owner was found, as for + /// ambiguous UDP or an expired attribution budget. + fn unattributed(env: &mut TestEnv) { + env.resolver.pid = None; + } + + fn expect_undecided_prompt(outcome: PacketOutcome, rule: &Rule, fallback: Action) { + match outcome { + PacketOutcome::Prompt { + undecided, + fallback: f, + .. + } => { + assert_eq!(undecided, Some(rule.id)); + assert_eq!(f.action, fallback); + assert_eq!(f.source, VerdictSource::DefaultPolicy); + } + other => panic!("expected an undecided prompt, got {other:?}"), + } + } + + #[test] + fn an_unattributed_flow_under_a_program_allow_is_prompted_not_dropped() { + let rule = exe_rule(Action::Allow, "/usr/bin/firefox"); + for policy in [dp_allow(), dp_deny()] { + let no_ui_action = policy.no_ui_action; + let mut env = TestEnv::new(vec![rule.clone()], policy); + unattributed(&mut env); + expect_undecided_prompt(env.handle(&tcp_packet(443), &NO_META), &rule, no_ui_action); + } + } + #[test] - fn missing_executable_cannot_lift_a_deny_when_paused() { + fn an_undecided_flow_is_prompted_while_paused() { + // Pause means "stop asking about flows no rule is about"; a program + // rule may be about this one. let mut rule = deny_port_rule(443); rule.scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); - let mut env = TestEnv::new(vec![rule], dp_allow()); - env.resolver.pid = None; + let mut env = TestEnv::new(vec![rule.clone()], dp_allow()); + unattributed(&mut env); env.stats.set_paused(true); - match env.handle(&tcp_packet(443), &NO_META) { - PacketOutcome::Deliver { verdict, .. } => assert_eq!(verdict.action, Action::Deny), - other => panic!("expected closed policy, got {other:?}"), + expect_undecided_prompt(env.handle(&tcp_packet(443), &NO_META), &rule, Action::Allow); + } + + #[test] + fn ambiguous_udp_under_a_program_deny_is_prompted() { + let rule = exe_rule(Action::Deny, "/usr/bin/x"); + let mut env = TestEnv::new(vec![rule.clone()], dp_deny()); + unattributed(&mut env); + expect_undecided_prompt(env.handle(&udp_packet(53), &NO_META), &rule, Action::Deny); + } + + #[test] + fn ambiguous_udp_with_no_rules_is_prompted() { + let mut env = TestEnv::new(vec![], dp_deny()); + unattributed(&mut env); + match env.handle(&udp_packet(53), &NO_META) { + PacketOutcome::Prompt { undecided, .. } => assert_eq!(undecided, None), + other => panic!("expected a prompt, got {other:?}"), } } @@ -2745,24 +2852,76 @@ mod tests { } } - #[test] - fn loopback_missing_identity_cannot_override_an_application_refusal() { - let mut scope = RuleScope::any(); - scope.exe_path = Some(PathBuf::from("/usr/bin/curl")); - let rule = Rule::new("local application refusal", Action::Deny, scope); - let mut h = LoopHarness::new(vec![], vec![rule], dp_deny()); + /// A [`LoopHarness`] whose resolver attributes nothing. + fn unattributed_harness(rules: Vec, policy: DefaultPolicy) -> LoopHarness { + let mut h = LoopHarness::new(vec![], rules, policy); h.worker().resolver = Box::new(StubResolver { pid: None, process: Process::unknown(0), socket_lookups: std::sync::atomic::AtomicUsize::new(0), }); + h + } + + #[test] + fn an_undecided_loopback_flow_is_prompted() { + let rule = exe_rule(Action::Deny, "/usr/bin/curl"); + let mut h = unattributed_harness(vec![rule.clone()], dp_deny()); h.stats.set_paused(true); let mut msg = FakeMsg::new(1, tcp_packet(53)); msg.outdev = 1; h.worker().handle_message(msg).unwrap(); - assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Drop)]); - assert_eq!(h.audited().len(), 1); - assert!(h.prompt_rx.try_recv().is_err()); + assert!(h.verdicts().is_empty(), "the packet waits for the answer"); + let req = h.prompt_rx.try_recv().expect("prompt dispatched"); + assert_eq!(req.undecided, Some(rule.id)); + } + + #[test] + fn resolve_prompt_applies_the_users_allow_to_an_undecided_flow() { + // Before, the per-packet re-check turned an undecided flow into a + // Deny whatever the user answered. + let rule = exe_rule(Action::Deny, "/usr/bin/curl"); + let mut h = unattributed_harness(vec![rule.clone()], dp_deny()); + h.worker() + .handle_message(FakeMsg::new(1, tcp_packet(443))) + .unwrap(); + let req = h.prompt_rx.try_recv().expect("prompt dispatched"); + h.send_verdict( + req.prompt_id, + Verdict { + action: Action::Allow, + source: VerdictSource::UserPrompt, + }, + ); + h.worker().drain_verdicts().unwrap(); + assert_eq!(h.verdicts(), vec![(1, NfqVerdict::Accept)]); + let observed = h.observed_rx.try_recv().unwrap(); + assert_eq!(observed.undecided, Some(rule.id)); + assert_eq!(observed.rule_id(), Some(rule.id)); + } + + #[test] + fn an_undecided_prompt_past_the_cap_takes_no_ui_action() { + let rule = exe_rule(Action::Deny, "/usr/bin/curl"); + let mut h = unattributed_harness(vec![rule.clone()], dp_deny()); + // Unknown identities never share a prompt, so every packet parks its + // own until the backlog cap. + for i in 0..MAX_PARKED_PROMPTS { + h.worker() + .handle_message(FakeMsg::new(i as u32, tcp_packet(443))) + .unwrap(); + while h.prompt_rx.try_recv().is_ok() {} + } + assert!(h.verdicts().is_empty()); + let overflow = MAX_PARKED_PROMPTS as u32; + h.worker() + .handle_message(FakeMsg::new(overflow, tcp_packet(443))) + .unwrap(); + assert_eq!(h.verdicts(), vec![(overflow, NfqVerdict::Drop)]); + let audited = h.audited(); + assert_eq!(audited.len(), 1); + assert_eq!(audited[0].source, "default"); + assert_eq!(audited[0].rule_id, Some(rule.id.to_string())); } #[test] @@ -3040,6 +3199,7 @@ mod tests { conn_to(443, 1111), test_process(4242, "/usr/bin/curl"), Verdict::default_deny(), + None, ) .unwrap(); let req = h.prompt_rx.try_recv().expect("prompt dispatched"); @@ -3162,6 +3322,7 @@ mod tests { conn, proc, Verdict::default_deny(), + None, ) .unwrap(); diff --git a/crates/cfc-daemon/src/process_resolve.rs b/crates/cfc-daemon/src/process_resolve.rs index fd640c3..2774170 100644 --- a/crates/cfc-daemon/src/process_resolve.rs +++ b/crates/cfc-daemon/src/process_resolve.rs @@ -55,7 +55,7 @@ use std::os::unix::fs::MetadataExt; use std::path::{Path, PathBuf}; use std::sync::LazyLock; use std::time::{Duration, Instant, SystemTime, UNIX_EPOCH}; -use tracing::trace; +use tracing::{debug, trace}; /// Per-lookup budget for the /proc slow path. const RESOLVE_BUDGET: Duration = Duration::from_millis(50); @@ -529,24 +529,33 @@ fn udp_inode_from_tables( uid: Option, deadline: Instant, ) -> Option { + // Each `None` below leaves the flow unattributed; it is then prompted + // with the identity shown as unknown. Named, so the journal says which. + let expired = || { + debug!("attribution budget expired"); + None + }; let mut entries = Vec::new(); for table in tables { if Instant::now() > deadline { - return None; + return expired(); } let contents = match fs::read_to_string(table) { Ok(contents) => contents, Err(e) if e.kind() == std::io::ErrorKind::NotFound => continue, - Err(_) => return None, + Err(e) => { + debug!(table, "table unreadable: {e}"); + return None; + } }; entries.extend(contents.lines().skip(1).filter_map(parse_table_line)); } if Instant::now() > deadline { - return None; + return expired(); } let inode = scan_table_entries(&entries, Protocol::Udp, local, remote, uid); if Instant::now() > deadline { - return None; + return expired(); } inode } @@ -606,6 +615,7 @@ fn scan_table_entries( && (endpoint_eq(e.remote, remote) || endpoint_is_zero(e.remote)) { if inode.is_some_and(|inode| inode != e.inode) { + debug!(local = ?local, remote = ?remote, "udp attribution ambiguous"); return None; } inode = Some(e.inode); diff --git a/crates/cfc-daemon/src/prompts.rs b/crates/cfc-daemon/src/prompts.rs index c3a6987..1127b0c 100644 --- a/crates/cfc-daemon/src/prompts.rs +++ b/crates/cfc-daemon/src/prompts.rs @@ -256,6 +256,7 @@ impl PromptRouter { // Said before the user answers, because "your allow will follow // the hash, not the path" changes what clicking Allow means. binds_to_hash: binding.hash_expected, + undecided_rule_id: req.undecided.map(|id| id.to_string()).unwrap_or_default(), }; self.inner.pending.lock().insert(prompt_id, binding); @@ -451,6 +452,7 @@ mod tests { 443, ), process: Process::unknown(1), + undecided: None, } } @@ -627,6 +629,41 @@ mod tests { assert!(router.submit("7", user_allow()).is_none()); } + #[tokio::test] + async fn an_undecided_prompt_carries_its_rule_id() { + let (tx, _rx) = std::sync::mpsc::channel(); + let router = PromptRouter::new(shared(dp(3600)), Stats::new(), tx); + let mut ui = router.subscribe(1000, true); + let rule = uuid::Uuid::new_v4(); + router.enqueue( + PromptRequest { + undecided: Some(rule), + ..req(3) + }, + PromptBinding::default(), + ); + router.enqueue(req(4), PromptBinding::default()); + assert_eq!(ui.recv().await.unwrap().undecided_rule_id, rule.to_string()); + assert_eq!(ui.recv().await.unwrap().undecided_rule_id, ""); + } + + #[tokio::test] + async fn with_no_answering_ui_an_undecided_prompt_gets_no_ui_action() { + let (tx, rx) = std::sync::mpsc::channel(); + let router = PromptRouter::new(shared(dp(3600)), Stats::new(), tx); + let _watcher = router.subscribe(1000, false); + router.enqueue( + PromptRequest { + undecided: Some(uuid::Uuid::new_v4()), + ..req(9) + }, + PromptBinding::default(), + ); + let pv = rx.try_recv().expect("no_ui_action applies immediately"); + assert_eq!(pv.prompt_id, 9); + assert_eq!(pv.verdict.source, VerdictSource::DefaultPolicy); + } + #[tokio::test] async fn user_answer_resolves_exactly_once() { let (tx, rx) = std::sync::mpsc::channel(); diff --git a/crates/cfc-daemon/tests/ipc_integration.rs b/crates/cfc-daemon/tests/ipc_integration.rs index 0e1bc3f..c8fa939 100644 --- a/crates/cfc-daemon/tests/ipc_integration.rs +++ b/crates/cfc-daemon/tests/ipc_integration.rs @@ -211,6 +211,7 @@ impl TestDaemon { prompt_id, connection: connection(443), process, + undecided: None, }) .await .expect("prompt channel closed"); @@ -416,6 +417,7 @@ fn observed(dst_port: u16, action: Action) -> ObservedConnection { action, source: VerdictSource::DefaultPolicy, }, + undecided: None, } } @@ -1286,11 +1288,7 @@ async fn observed_connections_reach_list_events_through_the_pipeline() { .expect("the pipeline is subscribed"); let blocked = observed(25, Action::Deny); d.store - .insert_events(&[cfc_daemon::convert::event_row_from_observed( - &blocked.connection, - &blocked.process, - &blocked.verdict, - )]) + .insert_events(&[cfc_daemon::convert::event_row_from_observed(&blocked)]) .expect("committing refusal before publication"); d.observed_tx .send(blocked) @@ -1410,6 +1408,7 @@ async fn stream_connections_maps_the_live_feed() { connection: conn, process: process(), verdict: Verdict::deny_from_rule(rule_id), + undecided: None, }) .expect("a subscriber exists"); @@ -1439,6 +1438,17 @@ async fn stream_connections_maps_the_live_feed() { let event = next_message(&mut stream).await; assert_eq!(event.verdict, pb::Action::Allow as i32); assert!(event.rule_id.is_empty()); + + // A prompted flow whose rule could not be decided names that rule. + let undecided = uuid::Uuid::new_v4(); + d.observed_tx + .send(ObservedConnection { + undecided: Some(undecided), + ..observed(53, Action::Allow) + }) + .expect("a subscriber exists"); + let event = next_message(&mut stream).await; + assert_eq!(event.rule_id, undecided.to_string()); } /// Copies /usr/bin/sleep into a tempdir and starts it: the ~/.local/bin shape diff --git a/crates/cfc-proto/proto/cfc.proto b/crates/cfc-proto/proto/cfc.proto index d373fde..ed0a170 100644 --- a/crates/cfc-proto/proto/cfc.proto +++ b/crates/cfc-proto/proto/cfc.proto @@ -66,6 +66,11 @@ message PromptEvent { // hash-bound deny is one file swap away from not applying, while the // path-bound one covers whatever bytes sit there next. bool binds_to_hash = 5; + // Set when a rule may apply but the process identity (exe, uid or digest) + // is incomplete, so the rule could not be decided: the UI shows the + // identity as unknown and names this rule. The answer applies to this + // connection. Empty when no rule is about the flow. + string undecided_rule_id = 6; } message VerdictRequest { @@ -99,6 +104,9 @@ message ConnectionEvent { ConnectionInfo connection = 1; ProcessInfo process = 2; Action verdict = 3; + // The rule that answered, or, when `verdict` did not come from a rule, the + // rule that could not be decided because the process identity was + // incomplete (see PromptEvent.undecided_rule_id). Empty otherwise. string rule_id = 4; } @@ -300,6 +308,9 @@ message Event { Action action = 11; // Verdict provenance: "rule" | "user" | "default" | "timeout". string source = 12; + // The rule that answered when `source` is "rule". With any other source, a + // non-empty id names the rule that could not be decided because the + // process identity was incomplete. string rule_id = 13; } diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 26bee6b..10f2a2e 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -2329,6 +2329,7 @@ mod tests { }), deadline_unix_ms: 1_700_000_015_000, binds_to_hash: false, + undecided_rule_id: String::new(), } } diff --git a/crates/cfc-ui/src/views/prompts.rs b/crates/cfc-ui/src/views/prompts.rs index 470135f..61277e9 100644 --- a/crates/cfc-ui/src/views/prompts.rs +++ b/crates/cfc-ui/src/views/prompts.rs @@ -636,6 +636,7 @@ mod tests { process: Some(process()), deadline_unix_ms: 1_700_000_015_000, binds_to_hash: false, + undecided_rule_id: String::new(), } } diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 50d9f98..f11cc9b 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -139,8 +139,9 @@ disconnects entirely, every outstanding prompt gets its fallback applied so no packet is stranded. **Pause is not a kill switch.** Rules are still evaluated while paused; -only the prompt is skipped, and only for flows that matched no rule. An -explicit Deny or Reject rule keeps blocking. Pause has a deadline: the +only the prompt is skipped, and only for flows that no rule is about. An +explicit Deny or Reject rule keeps blocking, and a flow that a rule may be +about but cannot be decided (see below) is still prompted. Pause has a deadline: the daemon clamps the requested duration (24h maximum), reports the real resume time, and auto-resumes. @@ -162,8 +163,13 @@ hundred microseconds before the packet's latency becomes visible. misses. UDP always reads all relevant tables first and requires one unique compatible inode: exact or wildcard local address, with exact or zero remote address. An unreadable table, an exhausted lookup budget or - several compatible inodes leave attribution unknown; an absent table - (`udp6` under `ipv6.disable=1`) counts as empty. The packet's socket UID, + several compatible inodes leave attribution unknown (with `--debug` the + journal says which: "table unreadable", "attribution budget expired", + "udp attribution ambiguous"); an absent table (`udp6` under + `ipv6.disable=1`) counts as empty. An unknown owner is judged like any + other incomplete identity: rules that name no program still answer, and + where a program rule may apply the flow is prompted with the identity + shown as unknown. The packet's socket UID, when present, filters candidates. All comparisons run on canonical form, so `::ffff:a.b.c.d` rows in the v6 tables match plain IPv4 flows - which is what dual-stack @@ -275,6 +281,23 @@ it. The in-kernel precompute (`Engine::process_wide_action` and `deny_still_possible_for`) walks the same way, so the connect hooks and the packet path agree. +**Incomplete identity is asked, not refused.** When the process's +executable, uid or digest is unknown (an unattributed socket, a binary over +the hashing cap, a process gone before `/proc` was read), a rule that tests +the missing field can be neither matched nor excluded. The scan walks past +such a rule and remembers whether it was an Allow or a refusal. If the rule +that then answers agrees with every rule passed that way (all allow, or all +refuse), it answers. Otherwise the flow is prompted, the prompt names the +first rule that could not be decided (`PromptEvent.undecided_rule_id`), and +the UIs show the identity as unknown. The user's answer applies to that +connection; an "always" answer for an unknown program is refused as before. +With no answering UI connected the flow takes `no_ui_action`, and the prompt +caps send overflow to the same fallback. Neither pause nor the loopback +allowance lifts such a flow. The recorded event and the live feed carry the +undecided rule's id in `rule_id` with a source other than `rule`. A legacy +hostname rule (`dst_host`) that cannot be decided is still refused, since no +answer can establish the name. + Disabled and expired rules are filtered at lookup, so a `Seconds(n)` rule stops matching the instant it expires rather than when the reaper next runs. A 30-second maintenance task flushes hit counts to disk and deletes expired diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 8f60d59..95c1ee1 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -197,8 +197,10 @@ usually past it. For such a program: - `cfc rules add --pin-hash` refuses the file. A digest supplied another way (`--sha256`, an import) is stored but can never be compared, so a rule - carrying one, Allow or Deny, refuses the program's flows wherever its other - fields match, and no rule below it can allow them. + carrying one, Allow or Deny, cannot be decided for the program. Wherever + its other fields match and the rules below it would answer differently, + the program's flows are prompted, naming that rule, and take + `no_ui_action` when no UI is connected. - On a root-sealed path (root-owned, with root-owned ancestors, as a package installs it) nothing else changes: "Allow always" saves a path-only rule. - On any other path (under a home directory, a user-writable `/opt` @@ -208,6 +210,18 @@ usually past it. For such a program: (`cfc rules add --exe `) works, but whoever can write that file inherits it. Installing the program root-owned is the better fix. +**Incomplete identity is asked, not refused.** A flow whose executable, +uid or digest is unknown (unattributed UDP, an expired attribution budget, +a process that exited first) cannot be checked against a program rule. When +such a rule and the rest of the rule set disagree, the flow is prompted +with the identity shown as unknown; with no UI connected it takes +`no_ui_action`. That is Deny on every shipped profile. If you set +`no_ui_action = "Allow"`, a `deny --exe` rule no longer holds on a machine +nobody is watching: a program can make its own UDP attribution ambiguous +(binding a port another of its user's sockets shares) or exit before +`/proc` is read, and its flow then gets the permissive fallback. Keep +`no_ui_action` at Deny where program denies matter. + ## What this firewall does *not* protect against Normal mode follows the desktop application firewall model of OpenSnitch and diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index f261d06..b0b469b 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -682,6 +682,30 @@ ranked either, so a rule that relied on `--dst-net 0.0.0.0/0` to outrank another may now lose the tie to a Deny. `cfc rules list` shows the rules involved; the hit counters show which one answers. +## A prompt says the program could not be identified + +Since 0.8.0 a flow whose program is only partly known (no socket owner +found, a binary too large to hash, a process that exited first) is asked +about when a rule naming a program may apply to it, instead of being +refused in silence. The prompt names that rule and says the program could +not be fully identified; your answer applies to that connection only. With +no app, tray or `sudo cfc prompts` connected the flow takes `no_ui_action`, +and the event log shows it with that rule's id and source `default`. + +These prompts appear even while paused or for loopback flows, because a +rule may be about them. If they are frequent, find out why attribution +fails. Run the daemon with `--debug` (`systemctl edit colony-firewalld`, +then repeat `ExecStart=` with `--debug` appended) and look for: + +```sh +journalctl -u colony-firewalld -g 'udp attribution ambiguous|attribution budget expired|table unreadable|identity is incomplete' +``` + +"udp attribution ambiguous" means several sockets could own the datagram +(typically `SO_REUSEPORT` or a wildcard-bound socket shared across +programs); scope a rule on the port instead of the program for that +traffic. + ## Where things live | Thing | Path | From 5858a07b9721e941622864a9e14c483c8fa44d0b Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:14:42 +0200 Subject: [PATCH 116/125] feat(clients): say when a prompt is about a program that could not be identified The GUI prompt card names the rule the daemon could not decide (from the loaded rule list, or by id) and says the answer applies to this connection. The tray notification body and the cfc prompts human output add the same line, and the JSON output carries undecided_rule_id. The changelog records the 0.8.0 behaviour change. --- CHANGELOG.md | 15 ++++++ crates/cfc-cli/src/prompts.rs | 30 +++++++++++ crates/cfc-tray/src/model.rs | 25 +++++++++ crates/cfc-ui/src/main.rs | 7 ++- crates/cfc-ui/src/views/prompts.rs | 81 ++++++++++++++++++++++++++---- 5 files changed, 146 insertions(+), 12 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 017cd6d..2f77348 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -19,6 +19,21 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). predicate when rules are ranked; it still limits a rule to one address family, and stored rules that carry only a `/0` keep loading. Some flows change verdict on upgrade: review `cfc rules list`. +- **Breaking: a flow whose process identity is incomplete is prompted + instead of silently refused** when a program rule may apply to it: an + unattributed socket (ambiguous UDP, an expired attribution budget), a + binary too large to hash, a process that exited first. The prompt names + the rule that could not be decided (`PromptEvent.undecided_rule_id`) and + the app, tray and `cfc prompts` show the program as unknown. With no UI + connected the flow takes `no_ui_action` (Deny on every shipped profile), + and prompt caps send overflow there too; pause and the loopback allowance + no longer let such flows through without asking. One "allow this program" + rule no longer blocks every unattributed flow that a generic rule allows. + The event log and live feed record the undecided rule in `rule_id` with a + source other than `rule`. A legacy hostname rule that cannot be decided is + still refused. With `no_ui_action = "Allow"`, a program can reach that + fallback by making its own attribution fail, so keep it at Deny where + program denies matter. - New loopback flows now go through the queue instead of being accepted outright (`oifname "lo" accept` is gone from the outbound table). While the daemon runs, explicit rules apply to them, so a loopback Deny that 0.7.0 diff --git a/crates/cfc-cli/src/prompts.rs b/crates/cfc-cli/src/prompts.rs index 9f1ddf2..9d6c571 100644 --- a/crates/cfc-cli/src/prompts.rs +++ b/crates/cfc-cli/src/prompts.rs @@ -126,6 +126,9 @@ struct PromptJson<'a> { dst_port: u32, dst_host: Option<&'a str>, binds_to_hash: bool, + /// A rule that may apply but could not be decided because the program + /// is only partly identified; null when no rule is about the flow. + undecided_rule_id: Option<&'a str>, #[serde(skip_serializing_if = "Option::is_none")] verdict: Option<&'a str>, #[serde(skip_serializing_if = "Option::is_none")] @@ -167,11 +170,24 @@ fn to_json<'a>( dst_port: conn.map(|c| c.dst_port).unwrap_or(0), dst_host: conn.and_then(|c| opt(&c.dst_host)), binds_to_hash: ev.binds_to_hash, + undecided_rule_id: opt(&ev.undecided_rule_id), verdict, accepted, } } +/// The human line for a prompt about a flow a rule may cover but could not +/// decide, because the program is only partly identified. +fn undecided_line(ev: &proto::PromptEvent) -> Option { + opt(&ev.undecided_rule_id).map(|id| { + format!( + " rule {} may apply, but the program could not be fully identified; \ + the answer applies to this connection", + output::terminal_safe(id) + ) + }) +} + /// The destination line: hostname when known, otherwise the IP, always /// with the port and protocol. pub fn describe_destination(conn: Option<&proto::ConnectionInfo>) -> String { @@ -724,6 +740,9 @@ fn print_prompt(ev: &proto::PromptEvent) { println!(" binding image hash unavailable; persistent allow cannot be saved"); } } + if let Some(line) = undecided_line(ev) { + println!("{line}"); + } } #[cfg(test)] @@ -951,6 +970,17 @@ mod tests { undecided_rule_id: String::new(), }; let v = serde_json::to_value(to_json(&ev, None, None)).unwrap(); + assert_eq!(v["undecided_rule_id"], serde_json::Value::Null); + assert_eq!(undecided_line(&ev), None); + let undecided = proto::PromptEvent { + undecided_rule_id: "r1".into(), + ..ev.clone() + }; + let u = serde_json::to_value(to_json(&undecided, None, None)).unwrap(); + assert_eq!(u["undecided_rule_id"], "r1"); + let line = undecided_line(&undecided).unwrap(); + assert!(line.contains("r1 may apply"), "{line}"); + assert!(line.contains("could not be fully identified"), "{line}"); assert_eq!(v["prompt_id"], "17"); assert_eq!(v["exe"], "/usr/bin/curl"); assert_eq!(v["uid"], 1000); diff --git a/crates/cfc-tray/src/model.rs b/crates/cfc-tray/src/model.rs index 99b0e88..c8028e7 100644 --- a/crates/cfc-tray/src/model.rs +++ b/crates/cfc-tray/src/model.rs @@ -314,6 +314,13 @@ pub fn prompt_notification(ev: &proto::PromptEvent, now_unix_ms: i64) -> PromptN }, ); } + // Why a flow a rule may cover is asked about at all. + if !ev.undecided_rule_id.is_empty() { + body.push_str( + "\nA rule may apply, but the program could not be fully identified; \ + your answer applies to this connection.", + ); + } let remaining = ev.deadline_unix_ms.saturating_sub(now_unix_ms); let timeout_ms = remaining.clamp(i64::from(MIN_PROMPT_TIMEOUT_MS), i64::from(u32::MAX)) as u32; PromptNotification { @@ -848,6 +855,24 @@ mod tests { assert!(n.body.starts_with("unknown:443 (tcp)"), "{}", n.body); } + #[test] + fn an_undecided_prompt_says_the_program_was_not_identified() { + let plain = prompt_notification(&prompt_event("", "", "1.1.1.1", 0), 0); + assert!(!plain.body.contains("could not be fully identified")); + let ev = proto::PromptEvent { + undecided_rule_id: "r1".into(), + ..prompt_event("", "", "1.1.1.1", 0) + }; + let n = prompt_notification(&ev, 0); + assert!( + n.body + .contains("A rule may apply, but the program could not be fully identified"), + "{}", + n.body + ); + assert!(n.body.contains("this connection")); + } + #[test] fn prompt_notification_summary_is_the_exe_basename() { let n = prompt_notification(&prompt_event("/usr/bin/curl", "example.com", "", 0), 0); diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 10f2a2e..906b398 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -1365,7 +1365,12 @@ impl App { let sidebar = self.sidebar(); let header = self.header_bar(paused); let body: Element<'_, Message> = match self.tab { - Tab::Prompts => views::prompts::view(&self.prompts, self.status.as_ref(), self.now_ms), + Tab::Prompts => views::prompts::view( + &self.prompts, + &self.rules, + self.status.as_ref(), + self.now_ms, + ), Tab::Rules => views::rules::view(views::rules::ListArgs { rules: &self.rules, filter: &self.rules_filter, diff --git a/crates/cfc-ui/src/views/prompts.rs b/crates/cfc-ui/src/views/prompts.rs index 61277e9..d65fe5c 100644 --- a/crates/cfc-ui/src/views/prompts.rs +++ b/crates/cfc-ui/src/views/prompts.rs @@ -31,6 +31,7 @@ const PROGRAM_LABEL: &str = "Program"; pub fn view<'a>( prompts: &'a [PromptCard], + rules: &'a [proto::RuleInfo], status: Option<&'a proto::StatusResponse>, now_ms: i64, ) -> Element<'a, Message> { @@ -38,7 +39,7 @@ pub fn view<'a>( return container( column![ text("No pending prompts").size(18), - text("Outbound flows without a matching rule will appear here for you to allow or block.").size(12), + text("Outbound flows that no rule decides will appear here for you to allow or block.").size(12), text("Keyboard, on the top prompt: A allow once / D block for now; Shift+A always allow this program, Shift+D always block it.").size(11), ] .spacing(8), @@ -60,7 +61,16 @@ pub fn view<'a>( let cards: Vec> = prompts .iter() .enumerate() - .map(|(i, c)| prompt_card(c, timeout_action, timeout_secs, now_ms, Some(i) == target)) + .map(|(i, c)| { + prompt_card( + c, + rules, + timeout_action, + timeout_secs, + now_ms, + Some(i) == target, + ) + }) .collect(); container(scrollable(column(cards).spacing(12).padding(8)).height(Length::Fill)) @@ -226,6 +236,24 @@ pub fn detail_rows(ev: &proto::PromptEvent) -> Vec { rows } +/// Why this flow was asked about although a rule may cover it: the daemon +/// could not fully identify the program, so it could not decide that rule. +/// Names the rule from the loaded list, or by id when it is not there. +pub fn undecided_warning(ev: &proto::PromptEvent, rules: &[proto::RuleInfo]) -> Option { + if ev.undecided_rule_id.is_empty() { + return None; + } + let rule = rules + .iter() + .find(|r| r.id == ev.undecided_rule_id) + .map(|r| convert::display_safe(&r.name)) + .unwrap_or_else(|| convert::display_safe(&ev.undecided_rule_id)); + Some(format!( + "Rule \"{rule}\" may apply, but the program could not be fully identified; \ + your answer applies to this connection." + )) +} + /// `"pid 4242 (parent pid 1310)"`. /// /// WFC names the parent program; the proto carries only `ppid`, so the @@ -371,13 +399,14 @@ pub fn action_consequence(choice: PromptAction, program: &str) -> String { type ButtonStyle = fn(&iced::Theme, iced::widget::button::Status) -> iced::widget::button::Style; -fn prompt_card( - card: &PromptCard, +fn prompt_card<'a>( + card: &'a PromptCard, + rules: &[proto::RuleInfo], timeout_action: i32, timeout_secs: u32, now_ms: i64, key_target: bool, -) -> Element<'_, Message> { +) -> Element<'a, Message> { let ev = &card.event; let program = program_label(ev); // Disabled for a moment after the card appears, so a click aimed at @@ -385,7 +414,7 @@ fn prompt_card( // verdict is on the way. let armed = card.armed(now_ms); - let marker: Element<'_, Message> = if key_target { + let marker: Element<'a, Message> = if key_target { text("A / D answer this prompt") .size(10) .color(crate::theme::PARCHMENT_MUTED) @@ -452,7 +481,7 @@ fn prompt_card( // Without an exe path the two program rows are dead: say why, once, // rather than leaving the user clicking a button that does nothing. - let unscopable: Element<'_, Message> = if verdict_for(PromptAction::AllowProgram, ev).is_some() + let unscopable: Element<'a, Message> = if verdict_for(PromptAction::AllowProgram, ev).is_some() { Space::new().into() } else { @@ -465,10 +494,20 @@ fn prompt_card( .into() }; - container(column![header, countdown, table, customize, actions, unscopable].spacing(9)) - .padding(12) - .style(crate::theme::panel) - .into() + let undecided: Element<'a, Message> = match undecided_warning(ev, rules) { + Some(warning) => text(warning) + .size(11) + .color(crate::theme::BURGUNDY_DARK) + .into(), + None => Space::new().into(), + }; + + container( + column![header, countdown, undecided, table, customize, actions, unscopable].spacing(9), + ) + .padding(12) + .style(crate::theme::panel) + .into() } fn detail_row<'a>(r: DetailRow) -> Element<'a, Message> { @@ -896,6 +935,26 @@ mod tests { ); } + #[test] + fn an_undecided_prompt_names_the_rule_it_could_not_decide() { + assert_eq!(undecided_warning(&event(), &[]), None); + let ev = proto::PromptEvent { + undecided_rule_id: "r1".into(), + ..event() + }; + let rules = [proto::RuleInfo { + id: "r1".into(), + name: "block telemetry".into(), + ..Default::default() + }]; + let line = undecided_warning(&ev, &rules).unwrap(); + assert!(line.contains("\"block telemetry\""), "{line}"); + assert!(line.contains("could not be fully identified"), "{line}"); + assert!(line.contains("this connection"), "{line}"); + // A rule not in the loaded list is still named, by its id. + assert!(undecided_warning(&ev, &[]).unwrap().contains("\"r1\"")); + } + #[test] fn program_label_falls_back_to_a_pronoun() { assert_eq!(program_label(&event()), "curl"); From ce49322e55c1bdcbebcde9fb1dc73badfd734e9a Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:27:44 +0200 Subject: [PATCH 117/125] fix(daemon): refuse an official client running under a seccomp filter A filter installed before exec survives it and can answer close_range and close with success, so the sealing prologue closed nothing and the installed tray kept a connection its parent had opened and still wrote on. The app and tray never install a filter; any non-zero Seccomp mode in /proc//status now makes the caller read-only. --- CHANGELOG.md | 6 ++++-- README.md | 6 +++--- crates/cfc-daemon/src/ipc.rs | 4 ++-- crates/cfc-daemon/src/official.rs | 35 +++++++++++++++++++++++++++++-- docs/ARCHITECTURE.md | 3 ++- docs/HARDENING.md | 5 ++++- 6 files changed, 48 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2f77348..fe98533 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -98,8 +98,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). its process runs the installed, root-sealed `/usr/bin/colony-firewall` or `/usr/bin/colony-firewall-tray` (by device and inode), sealed itself at startup (inherited descriptors closed, non-dumpable), holds the connection - itself, is not traced, runs in the host namespaces and mapped no executable - file from outside root-owned directories. Every other program of the + itself, is not traced, runs under no seccomp filter (one installed before + `exec` can fake the descriptor closing), runs in the host namespaces and + mapped no executable file from outside root-owned directories. Every other + program of the desktop user, a non-root `cfc` included, is read-only: its changes are refused with the reason (exit 1 in `cfc`), its prompt subscription does not count as a connected UI and it cannot answer prompts. A peer with the diff --git a/README.md b/README.md index 75fd5f2..a0a1f54 100644 --- a/README.md +++ b/README.md @@ -459,9 +459,9 @@ sudo cfc rules import-opensnitch /etc/opensnitchd/rules The daemon recognises the app and tray by the running image: it must be the installed, root-owned `/usr/bin/colony-firewall` or `colony-firewall-tray`, -started normally (not traced, no library preloaded from your files). After -an upgrade, restart both; until then they are read-only. Details and limits -are in [docs/HARDENING.md](docs/HARDENING.md). +started normally (not traced, no seccomp filter, no library preloaded from +your files). After an upgrade, restart both; until then they are read-only. +Details and limits are in [docs/HARDENING.md](docs/HARDENING.md). Executable rules require the canonical mapped target explicitly. An alias such as `/bin/tool` on a system where `/bin` links to `/usr/bin` is refused; diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 3dcefe9..78bb3a8 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -24,8 +24,8 @@ //! start-time reads) whose process passes [`crate::official::check`]: //! it runs one of the installed, root-sealed `[ipc] official_clients` //! binaries, ran their sealing prologue, holds this very connection, is -//! not traced and mapped no executable file from outside sealed -//! directories. `require_group = false` waives the group proof for +//! not traced, runs under no seccomp filter and mapped no executable +//! file from outside sealed directories. `require_group = false` waives the group proof for //! official clients only. //! //! Pause, resume and `ApplyRules` change the whole firewall at once, so diff --git a/crates/cfc-daemon/src/official.rs b/crates/cfc-daemon/src/official.rs index 1ad83e3..b101523 100644 --- a/crates/cfc-daemon/src/official.rs +++ b/crates/cfc-daemon/src/official.rs @@ -8,7 +8,9 @@ //! 1. its start time matches the one captured at accept (the "before" read); //! 2. it runs in the host mount and user namespaces, so no private mount or //! user namespace can show it a different `/usr/bin`; -//! 3. it is not traced and its effective uid is the connection's uid; +//! 3. it is not traced, runs under no seccomp filter (a filter can make +//! the prologue's `close_range` report success without closing +//! anything) and its effective uid is the connection's uid; //! 4. it is non-dumpable, which is the mark `seal_official_process` leaves: //! the kernel hands the files under `/proc/` to root exactly then; //! 5. its image is, by device and inode, one of the allowlisted binaries, @@ -116,7 +118,14 @@ fn inspect(pid: u32, uid: u32, sock_ino: u64, allowlist: &[PathBuf]) -> Result

Result<(), String> { let field = |name: &str| { status @@ -130,6 +139,14 @@ fn status_is_clean(status: &str, uid: u32) -> Result<(), String> { Some(_) => return Err("the caller is being traced".into()), None => return Err("the caller's status has no TracerPid".into()), } + if field("Seccomp:") + .and_then(|mut v| v.next()) + .is_some_and(|mode| mode != "0") + { + return Err( + "the caller runs under a seccomp filter, which the app and tray never install".into(), + ); + } let effective = field("Uid:").and_then(|mut v| v.nth(1)?.parse::().ok()); if effective != Some(uid) { return Err("the caller's effective uid is not the connection's uid".into()); @@ -383,6 +400,20 @@ mod tests { assert!(status_is_clean("Uid:\t1000\t1000\t1000\t1000\n", 1000).is_err()); } + #[test] + fn a_seccomp_filter_is_refused() { + let base = "TracerPid:\t0\nUid:\t1000\t1000\t1000\t1000\n"; + assert!(status_is_clean(&format!("{base}Seccomp:\t0\n"), 1000).is_ok()); + assert!( + status_is_clean(base, 1000).is_ok(), + "no seccomp support at all" + ); + for mode in ["1", "2"] { + let error = status_is_clean(&format!("{base}Seccomp:\t{mode}\n"), 1000).unwrap_err(); + assert!(error.contains("seccomp"), "{error}"); + } + } + #[test] fn uid_mismatch_in_status_is_refused() { // Real uid 1000, effective uid 1001: the effective one decides. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index f11cc9b..351fbaf 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -371,7 +371,8 @@ the entire attack surface. Two layers: `DeleteRule`, `ApplyRules`, `SetPaused`) are accepted from root (or the daemon's own uid) and from the installed app and tray only. From `SO_PEERCRED`'s pid, `official.rs` checks, between two start-time reads, - that the process runs in the host namespaces, is not traced, has sealed + that the process runs in the host namespaces, is not traced, runs under + no seccomp filter (one can fake the prologue's `close_range`), has sealed itself (`cfc_client::seal_official_process`: inherited descriptors closed, non-dumpable), runs one of the root-sealed `[ipc] official_clients` binaries by device and inode, holds this connection's client end itself diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 95c1ee1..57de61f 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -341,7 +341,10 @@ that wants to change something is checked against the calling process, all of it between two reads of that process's start time (so a reused pid fails): - it runs in the host's mount and user namespaces; -- it is not traced, and its effective uid is the connection's; +- it is not traced, runs under no seccomp filter (a filter installed before + `exec` survives it and can make the prologue's descriptor closing report + success without closing anything), and its effective uid is the + connection's; - it ran the app's sealing prologue at startup: every inherited descriptor closed and the process made non-dumpable, which the kernel shows by giving its `/proc` files to root; From 4121086c8cb03250a2fdf3d2f0af4bc76f119f9f Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:29:18 +0200 Subject: [PATCH 118/125] docs(daemon): name the pre-seal ptrace route and warn while Yama allows it Under the default kernel.yama.ptrace_scope = 1 a program can start the installed app or tray with PTRACE_TRACEME, patch it at the exec stop and detach; the patched process seals itself and passes every check, and no later read shows the past trace. HARDENING said non-dumpable stopped later ptrace, which read as covering this. It now lists the route with what closes it (ptrace_scope = 2, or the setgid design), and the daemon warns at startup while the scope is lower. --- CHANGELOG.md | 5 +++- TODO.md | 2 +- crates/cfc-client/src/lib.rs | 7 +++--- crates/cfc-daemon/src/ipc.rs | 6 +++++ crates/cfc-daemon/src/official.rs | 42 +++++++++++++++++++++++++++++-- docs/HARDENING.md | 26 ++++++++++++++++--- docs/TROUBLESHOOTING.md | 11 ++++++++ 7 files changed, 88 insertions(+), 11 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index fe98533..2b3d8fa 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -108,7 +108,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). daemon's own uid keeps full control (root in production). New `[ipc] official_clients` key; `require_group` now waives the group check for the app and tray only. See docs/HARDENING.md for what this still - trusts. + trusts. One of those routes is a program that starts the app or tray + under `ptrace` and patches it before it seals itself, which the default + Yama `ptrace_scope = 1` allows; the daemon warns at startup until + `kernel.yama.ptrace_scope` is 2. - **Breaking: pause, resume and rule import from the app or tray ask for an administrator password** through polkit (`org.projectcolony.firewall.pause`, `org.projectcolony.firewall.import-rules`, diff --git a/TODO.md b/TODO.md index 263b5b5..e8578ff 100644 --- a/TODO.md +++ b/TODO.md @@ -253,7 +253,7 @@ What defeats it completely: | **DNS tunnelling** | the resolver must be allowed for anything to work. CFC *observes* answers; it does not inspect or block queries. | | **Inherited or passed socket descriptors** | Existing connection authorization is not rechecked for each sending executable; socket attribution is ambiguous when ownership is shared. | | **CAP_NET_RAW packet sockets** | Packet-layer egress can bypass the IP OUTPUT hook. Layer-2 confinement is outside the shipped rules. | -| **Code inside the official app or tray** | the daemon accepts changes only from root and the installed, sealed app and tray, checking the running process (image, prologue, connection, tracer, file-backed executable mappings). Code already running inside them is not seen: a self-unmapping `LD_PRELOAD` payload living in anonymous memory, or synthetic X11/XWayland input clicking the GUI. Also a connection opened before exec'ing the app and handed back into it over D-Bus while a kept copy writes the request (a race, repeatable). Setgid binaries trusted by connect-time gid would close the first and the last; not done. | +| **Code inside the official app or tray** | the daemon accepts changes only from root and the installed, sealed app and tray, checking the running process (image, prologue, connection, tracer, file-backed executable mappings). Code already running inside them is not seen: a self-unmapping `LD_PRELOAD` payload living in anonymous memory, or synthetic X11/XWayland input clicking the GUI. Also a connection opened before exec'ing the app and handed back into it over D-Bus while a kept copy writes the request (a race, repeatable), and code patched into the app by a parent tracing it from exec before it seals itself (deterministic under the default Yama `ptrace_scope = 1`; scope 2 closes it, the daemon warns at startup). Setgid binaries trusted by connect-time gid would close the first and the last two; not done. | | **Prompt fatigue** | demonstrated on this machine: ten Firefox prompts in a row, all denied, browser lost. A malicious installer generating thirty prompts trains the user to click Allow. | And one tradeoff worth stating plainly: the ruleset is **fail-closed for diff --git a/crates/cfc-client/src/lib.rs b/crates/cfc-client/src/lib.rs index 1f13c5c..f962513 100644 --- a/crates/cfc-client/src/lib.rs +++ b/crates/cfc-client/src/lib.rs @@ -379,9 +379,10 @@ impl Client { /// route. /// 2. Marks the process non-dumpable. The kernel then gives the files under /// `/proc/` to root, which is the marker the daemon checks, and -/// same-user `ptrace`, -/// `/proc//mem` and `pidfd_getfd` are refused for the rest of the -/// process's life. +/// same-user `ptrace`, `/proc//mem` and `pidfd_getfd` are refused +/// for the rest of the process's life. Not before: a parent that traced +/// this process from `exec` had until now, which only Yama +/// `ptrace_scope >= 2` prevents. /// /// A failure is returned for the caller to log once logging is up. The app /// keeps working, read-only: the daemon refuses its changes with a reason. diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 78bb3a8..fc2e72d 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -1576,6 +1576,12 @@ pub async fn spawn( for warning in opts.ipc.official_client_warnings() { warn!("{warning}"); } + if !opts.ipc.official_clients.is_empty() { + let scope = std::fs::read_to_string("/proc/sys/kernel/yama/ptrace_scope").ok(); + if let Some(warning) = crate::official::ptrace_scope_warning(scope.as_deref()) { + warn!("{warning}"); + } + } let incoming = tokio_stream::wrappers::UnixListenerStream::new(uds) .map(|stream| stream.and_then(PeerStream::new)); diff --git a/crates/cfc-daemon/src/official.rs b/crates/cfc-daemon/src/official.rs index b101523..2cc59d5 100644 --- a/crates/cfc-daemon/src/official.rs +++ b/crates/cfc-daemon/src/official.rs @@ -38,8 +38,11 @@ //! //! What it cannot see: code already running inside the official image that //! moved itself into anonymous memory (anonymous executable mappings are -//! not judged, GPU drivers JIT into them), and synthetic input to the GUI -//! under X11. See docs/HARDENING.md. +//! not judged, GPU drivers JIT into them), code a tracer wrote into the +//! image before the prologue ran and then detached (Yama's default +//! `ptrace_scope = 1` lets a program trace its own child from `exec`; see +//! [`ptrace_scope_warning`]), and synthetic input to the GUI. See +//! docs/HARDENING.md. use crate::ipc::PeerId; use std::collections::HashMap; @@ -50,6 +53,26 @@ use std::path::{Path, PathBuf}; pub const DEFAULT_CLIENTS: [&str; 2] = ["/usr/bin/colony-firewall", "/usr/bin/colony-firewall-tray"]; +/// The startup warning for a Yama `ptrace_scope` (the contents of +/// `/proc/sys/kernel/yama/ptrace_scope`, `None` without Yama) that lets a +/// same-user program trace the app or tray from `exec`, patch it before it +/// seals itself and detach. No later check can see a past trace; only +/// scope 2 (tracing needs `CAP_SYS_PTRACE`, `PTRACE_TRACEME` included) or 3 +/// closes that route. +pub fn ptrace_scope_warning(scope: Option<&str>) -> Option { + let level = scope.and_then(|s| s.trim().parse::().ok()); + if level.is_some_and(|level| level >= 2) { + return None; + } + let shown = level.map_or_else(|| "unavailable".to_string(), |level| level.to_string()); + Some(format!( + "kernel.yama.ptrace_scope is {shown}: a program of the desktop user can start the \ + installed app or tray under ptrace, change its code before it seals itself and \ + make changes as it. Set kernel.yama.ptrace_scope = 2 to close that route (see \ + docs/HARDENING.md)" + )) +} + /// Checks `peer` against `allowlist`. Blocking (reads `/proc` and asks /// sock_diag): call it from `spawn_blocking`. `Ok` carries the matched /// allowlist entry; `Err` a reason fit to show the user. @@ -400,6 +423,21 @@ mod tests { assert!(status_is_clean("Uid:\t1000\t1000\t1000\t1000\n", 1000).is_err()); } + #[test] + fn only_ptrace_scope_two_or_more_is_quiet() { + for quiet in ["2\n", "3\n"] { + assert_eq!(ptrace_scope_warning(Some(quiet)), None); + } + for loud in [Some("0\n"), Some("1\n"), Some("garbage"), None] { + let warning = ptrace_scope_warning(loud).unwrap(); + assert!(warning.contains("ptrace_scope = 2"), "{warning}"); + } + assert!(ptrace_scope_warning(Some("1\n")) + .unwrap() + .contains(" is 1:")); + assert!(ptrace_scope_warning(None).unwrap().contains("unavailable")); + } + #[test] fn a_seccomp_filter_is_refused() { let base = "TracerPid:\t0\nUid:\t1000\t1000\t1000\t1000\n"; diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 57de61f..21da0fd 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -362,8 +362,9 @@ of it between two reads of that process's start time (so a reused pid fails): The prologue and the descriptor check are what stop the obvious trick: connect, write a whole request into the socket, then `exec` the installed app -with the socket inherited. Non-dumpable also stops later same-user `ptrace`, -`/proc//mem` and `pidfd_getfd` on the app. +with the socket inherited. Non-dumpable stops same-user `ptrace`, +`/proc//mem` and `pidfd_getfd` on the app from the moment the prologue +runs. It does nothing about a tracer that was there before (see below). `require_group = true` (the default) also requires the app or tray to be run by a proved group member. `require_group = false` waives that for the app @@ -385,6 +386,20 @@ runs inside the installed app is the app: - a preloaded payload that copies itself into anonymous executable memory and unmaps its file is not seen (anonymous executable mappings cannot be refused: GPU drivers JIT into them); +- code written into the app before it sealed itself. Under Yama's + `kernel.yama.ptrace_scope = 1` (the default on most distributions) a + program may trace its own child, so it can start the installed app or tray + with `PTRACE_TRACEME` or under a debugger, patch its code at the `exec` + stop and detach. The patched process then runs the real prologue and every + check passes: nothing the daemon reads later shows a past trace. Setting + `kernel.yama.ptrace_scope = 2` (only root may trace, `PTRACE_TRACEME` + included) closes this, at the cost of debugging your own programs without + root; the daemon warns at startup while it is lower: + + ```sh + echo 'kernel.yama.ptrace_scope = 2' | sudo tee /etc/sysctl.d/60-ptrace-scope.conf + sudo sysctl --system + ``` - synthetic input into the GUI under X11 or XWayland can click its buttons; - a socket handed back into the app: a same-user program that started the app itself (connect first, then exec the installed binary, keeping a copy @@ -400,8 +415,11 @@ The kernel-enforced next step would be making the two binaries setgid to a dedicated empty group and trusting the connect-time `SO_PEERCRED` gid: glibc then ignores `LD_PRELOAD`/`LD_AUDIT` (secure execution), the process is non-dumpable from `exec`, and a connection made before that `exec` carries -the wrong gid, which also closes the handed-back socket above. That costs -packaging work in every channel and is not done yet. +the wrong gid, which also closes the handed-back socket above. An +unprivileged tracer, or a seccomp filter (which needs `no_new_privs`), makes +`exec` drop the setgid, so it would close the pre-seal tracing route too, +whatever `ptrace_scope` says. That costs packaging work in every channel and +is not done yet. Side effects of the sealing prologue: the app and tray write no core dumps, attaching a debugger to them needs root, and a developer build run from diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index b0b469b..af8dea2 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -343,6 +343,17 @@ minutes). Root (`sudo cfc pause`) is never asked. What the refusals mean: - **"authorization timed out after 120 s"**: the dialog was left open; the daemon closed it. +## The daemon warns about `kernel.yama.ptrace_scope` + +`kernel.yama.ptrace_scope is 1: a program of the desktop user can start the +installed app or tray under ptrace ...` is logged once at startup. It is not +an error: with that setting any program of yours can start the app under a +debugger, change its code before it seals itself and make changes as it, +and the daemon cannot tell afterwards. Setting +`kernel.yama.ptrace_scope = 2` closes that route; the commands are in +[HARDENING.md](HARDENING.md#the-control-socket-and-who-can-talk-to-it). +Debugging your own programs then needs root. + ## Loopback and the local resolver The snippet's `output` hook matches loopback traffic too. On systems using From a5584662d7174d7403e98ffb3f7965e0373ca5dd Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:31:38 +0200 Subject: [PATCH 119/125] fix(tray): take prompt answers only from the notification server notify-rust's wait_for_action matched ActionInvoked with no sender, so any session-bus program could emit the signal for its own prompt's bubble and the tray, which the daemon trusts without a password, submitted Always allow for it. The tray now subscribes before showing the bubble, bound to the unique name that owns org.freedesktop.Notifications, and checks each message's sender again, since a signal addressed to the tray arrives whatever it subscribed to. A test on a private bus pins it. --- CHANGELOG.md | 4 + crates/cfc-tray/src/answers.rs | 215 +++++++++++++++++++++++++++++++++ crates/cfc-tray/src/main.rs | 122 ++++++++++++------- docs/HARDENING.md | 5 + 4 files changed, 300 insertions(+), 46 deletions(-) create mode 100644 crates/cfc-tray/src/answers.rs diff --git a/CHANGELOG.md b/CHANGELOG.md index 2b3d8fa..2d8c842 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -112,6 +112,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). under `ptrace` and patches it before it seals itself, which the default Yama `ptrace_scope = 1` allows; the daemon warns at startup until `kernel.yama.ptrace_scope` is 2. + The tray now takes a prompt notification's answer only from the + notification server's own bus connection: the button signal is one any + session-bus program can emit, and the tray, trusted by the daemon, used to + act on it, so a program could press "Always allow app" on its own prompt. - **Breaking: pause, resume and rule import from the app or tray ask for an administrator password** through polkit (`org.projectcolony.firewall.pause`, `org.projectcolony.firewall.import-rules`, diff --git a/crates/cfc-tray/src/answers.rs b/crates/cfc-tray/src/answers.rs new file mode 100644 index 0000000..6db3f40 --- /dev/null +++ b/crates/cfc-tray/src/answers.rs @@ -0,0 +1,215 @@ +//! Answers to the tray's prompt notifications, taken only from the +//! notification server. +//! +//! `ActionInvoked` is a signal, and any program on the session bus may emit +//! one with any notification id, broadcast or sent straight to the tray. +//! notify-rust's `wait_for_action` matched it with no sender, so a program +//! could press "Always allow app" on its own prompt through the tray, which +//! the daemon trusts. Here the server's unique name is looked up once (a +//! method reply, which no other client can forge), the bus is asked for +//! that sender's signals only, and every message's sender is checked again, +//! since a signal addressed to the tray reaches it whatever it subscribed to. +//! +//! The server itself is still trusted: a same-user program that replaces it +//! answers for the user (see docs/HARDENING.md). + +use crate::model::KEY_CLOSED; +use tokio_stream::StreamExt as _; +use zbus::message::Type; +use zbus::names::{OwnedUniqueName, UniqueName}; +use zbus::{Message, MessageStream}; + +const SERVER: &str = "org.freedesktop.Notifications"; +const PATH: &str = "/org/freedesktop/Notifications"; + +/// Signals of the notification server that owned its name at subscription. +pub struct Answers { + owner: OwnedUniqueName, + stream: MessageStream, +} + +impl Answers { + /// Subscribes on `conn`. Call it before showing the bubble, so a click + /// that comes before the wait starts is queued, not lost. + pub async fn subscribe(conn: &zbus::Connection) -> zbus::Result { + let owner = zbus::fdo::DBusProxy::new(conn) + .await? + .get_name_owner(SERVER.try_into()?) + .await?; + let rule = zbus::MatchRule::builder() + .msg_type(Type::Signal) + .sender(owner.as_ref())? + .path(PATH)? + .interface(SERVER)? + .build(); + let stream = MessageStream::for_match_rule(rule, conn, None).await?; + Ok(Self { owner, stream }) + } + + /// The action key picked on notification `id`, or [`KEY_CLOSED`] once + /// it closed. `None` when the bus connection ended first. + pub async fn next_for(&mut self, id: u32) -> Option { + while let Some(message) = self.stream.next().await { + if let Some(key) = message.ok().and_then(|m| answer(&m, &self.owner, id)) { + return Some(key); + } + } + None + } +} + +/// What `message` says about notification `id`, if it comes from `owner`. +fn answer(message: &Message, owner: &UniqueName<'_>, id: u32) -> Option { + let header = message.header(); + if header.message_type() != Type::Signal || header.sender() != Some(owner) { + return None; + } + match header.member()?.as_str() { + "ActionInvoked" => { + let (nid, key): (u32, String) = message.body().deserialize().ok()?; + (nid == id).then_some(key) + } + "NotificationClosed" => { + let (nid, _reason): (u32, u32) = message.body().deserialize().ok()?; + (nid == id).then(|| KEY_CLOSED.to_string()) + } + _ => None, + } +} + +#[cfg(test)] +mod tests { + use super::*; + use std::io::BufRead as _; + use std::process::{Child, Command, Stdio}; + use zbus::connection::Builder; + use zbus::names::BusName; + + /// A private session bus, killed on drop. + struct Bus { + child: Child, + address: String, + _dir: tempfile::TempDir, + } + + impl Drop for Bus { + fn drop(&mut self) { + let _ = self.child.kill(); + let _ = self.child.wait(); + } + } + + fn private_bus() -> Option { + let dir = tempfile::tempdir().ok()?; + let mut child = Command::new("dbus-daemon") + .arg("--session") + .arg("--nofork") + .arg("--print-address=1") + .arg(format!( + "--address=unix:path={}", + dir.path().join("bus").display() + )) + .stdout(Stdio::piped()) + .stderr(Stdio::null()) + .spawn() + .ok()?; + let mut address = String::new(); + std::io::BufReader::new(child.stdout.take()?) + .read_line(&mut address) + .ok()?; + let address = address.trim().to_string(); + let bus = Bus { + child, + address, + _dir: dir, + }; + (!bus.address.is_empty()).then_some(bus) + } + + async fn connect(bus: &Bus) -> zbus::Connection { + Builder::address(bus.address.as_str()) + .unwrap() + .build() + .await + .unwrap() + } + + async fn emit( + from: &zbus::Connection, + to: Option>, + member: &str, + body: &(u32, &str), + ) { + from.emit_signal(to, PATH, SERVER, member, body) + .await + .unwrap(); + } + + #[tokio::test] + async fn only_the_notification_server_answers() { + let Some(bus) = private_bus() else { + eprintln!("skipped: dbus-daemon is not available"); + return; + }; + let server = Builder::address(bus.address.as_str()) + .unwrap() + .name(SERVER) + .unwrap() + .build() + .await + .unwrap(); + let tray = connect(&bus).await; + let intruder = connect(&bus).await; + let mut answers = Answers::subscribe(&tray).await.unwrap(); + + // Forged, broadcast and addressed to the tray, with the right id. + let tray_name = BusName::from(tray.unique_name().unwrap().clone()); + emit(&intruder, None, "ActionInvoked", &(7, "allow")).await; + emit( + &intruder, + Some(tray_name.clone()), + "ActionInvoked", + &(7, "allow"), + ) + .await; + let closed: (u32, u32) = (7, 2); + intruder + .emit_signal(Some(tray_name), PATH, SERVER, "NotificationClosed", &closed) + .await + .unwrap(); + // A round trip on the intruder's connection: the bus has routed + // everything it sent before this reply, so the forgeries come first. + zbus::fdo::DBusProxy::new(&intruder) + .await + .unwrap() + .get_id() + .await + .unwrap(); + + // The server: another bubble first, then this one. + emit(&server, None, "ActionInvoked", &(8, "allow")).await; + emit(&server, None, "ActionInvoked", &(7, "deny")).await; + assert_eq!(answers.next_for(7).await.as_deref(), Some("deny")); + + server + .emit_signal( + None::>, + PATH, + SERVER, + "NotificationClosed", + &closed, + ) + .await + .unwrap(); + assert_eq!(answers.next_for(7).await.as_deref(), Some(KEY_CLOSED)); + } + + #[tokio::test] + async fn no_notification_server_is_an_error() { + let Some(bus) = private_bus() else { + eprintln!("skipped: dbus-daemon is not available"); + return; + }; + assert!(Answers::subscribe(&connect(&bus).await).await.is_err()); + } +} diff --git a/crates/cfc-tray/src/main.rs b/crates/cfc-tray/src/main.rs index 81313f4..617c61d 100644 --- a/crates/cfc-tray/src/main.rs +++ b/crates/cfc-tray/src/main.rs @@ -8,6 +8,7 @@ //! same control socket as the GUI and CLI - quitting the tray never //! touches the daemon. +mod answers; mod icon; mod model; mod theme; @@ -572,54 +573,46 @@ impl PromptNotifier { .map(|p| p.exe.clone()) .unwrap_or_default(); let tx = self.tx.clone(); - // One blocking task per shown notification: show() and - // wait_for_action() both block on D-Bus, and the wait lasts until - // the user acts or the bubble expires. - on_notification_thread(move || { - let mut notification = notify_rust::Notification::new(); - brand(&mut notification) - .summary(&n.summary) - .body(&n.body) - .timeout(notify_rust::Timeout::Milliseconds(n.timeout_ms)) - // Verdicts first and short: these are what the user came - // for, and long labels wrap the button row onto a second - // line. "Details" is the freedesktop `default` action, so - // clicking the bubble body opens the GUI too. - .action(model::KEY_ALLOW_ONCE, "Allow once"); - if n.offer_block { - notification.action(model::KEY_ALLOW, "Always allow app"); - } - notification.action(model::KEY_DENY, "Deny"); - if n.offer_block { - notification.action(model::KEY_BLOCK, "Block app"); - } - notification.action(model::KEY_DEFAULT, "Details"); - let shown_id = prompt_id.clone(); - let shown_tx = tx.clone(); - let done = move |key: &str| { - // Failing only means the main loop is gone; the process - // is on its way out. - let _ = tx.send(Cmd::PromptResult { - prompt_id, - exe, - key: key.to_string(), - }); - }; - match notification.show() { - Ok(handle) => { - let _ = shown_tx.send(Cmd::PromptShown { - prompt_id: shown_id, - id: handle.id(), - }); - handle.wait_for_action(done); - } + let mut notification = notify_rust::Notification::new(); + brand(&mut notification) + .summary(&n.summary) + .body(&n.body) + .timeout(notify_rust::Timeout::Milliseconds(n.timeout_ms)) + // Verdicts first and short: these are what the user came + // for, and long labels wrap the button row onto a second + // line. "Details" is the freedesktop `default` action, so + // clicking the bubble body opens the GUI too. + .action(model::KEY_ALLOW_ONCE, "Allow once"); + if n.offer_block { + notification.action(model::KEY_ALLOW, "Always allow app"); + } + notification.action(model::KEY_DENY, "Deny"); + if n.offer_block { + notification.action(model::KEY_BLOCK, "Block app"); + } + notification.action(model::KEY_DEFAULT, "Details"); + // Waited for past the prompt's deadline: by then the slot is + // reclaimed and the bubble closed, so this only ends a wait whose + // server vanished without saying so. + let wait = Duration::from_millis( + u64::try_from(ev.deadline_unix_ms.saturating_sub(now_unix_ms())).unwrap_or(0), + ) + ANSWER_GRACE; + tokio::spawn(async move { + let key = match await_answer(notification, &prompt_id, &tx, wait).await { + Ok(key) => key, Err(e) => { - warn!("showing prompt notification: {e}"); - // Free the slot; the daemon's timeout_action covers - // the prompt itself. - done(model::KEY_CLOSED); + // The daemon's timeout_action covers the prompt itself. + warn!("prompt notification: {e}"); + model::KEY_CLOSED.to_string() } - } + }; + // Failing only means the main loop is gone; the process is on + // its way out. + let _ = tx.send(Cmd::PromptResult { + prompt_id, + exe, + key, + }); }); } @@ -638,6 +631,8 @@ impl PromptNotifier { match shown { Ok(handle) => { let _ = tx.send(Cmd::OverflowShown { id: handle.id() }); + // notify-rust takes this answer from any sender + // (see `answers`); here that can only open the GUI. handle.wait_for_action(|key: &str| { let _ = tx.send(Cmd::OverflowResult { key: key.to_string(), @@ -704,6 +699,41 @@ impl PromptNotifier { } } +/// How long past a prompt's deadline its bubble's answer is waited for. +const ANSWER_GRACE: Duration = Duration::from_secs(10); + +/// Shows `notification` and returns the key the user picked, as told by the +/// notification server only (see [`answers`]); [`model::KEY_CLOSED`] when it +/// closed or `wait` ran out. +async fn await_answer( + notification: notify_rust::Notification, + prompt_id: &str, + tx: &mpsc::UnboundedSender, + wait: Duration, +) -> anyhow::Result { + let conn = zbus::Connection::session().await?; + let mut answers = answers::Answers::subscribe(&conn).await?; + // show() blocks on D-Bus through zbus's own runtime, which must not be + // entered from this one: a plain thread does it. + let (shown_tx, shown) = tokio::sync::oneshot::channel(); + on_notification_thread(move || { + let _ = shown_tx.send(notification.show().map(|handle| handle.id())); + }); + let id = shown + .await + .context("no thread to show the notification")? + .context("showing the notification")?; + let _ = tx.send(Cmd::PromptShown { + prompt_id: prompt_id.to_string(), + id, + }); + Ok(tokio::time::timeout(wait, answers.next_for(id)) + .await + .ok() + .flatten() + .unwrap_or_else(|| model::KEY_CLOSED.to_string())) +} + /// Closes notification bubbles by server id, off the main loop. /// /// The bubble's handle is consumed by the thread waiting on it, so this is diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 21da0fd..4049d54 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -400,6 +400,11 @@ runs inside the installed app is the app: echo 'kernel.yama.ptrace_scope = 2' | sudo tee /etc/sysctl.d/60-ptrace-scope.conf sudo sysctl --system ``` +- the session's notification server, which the tray's prompt buttons go + through. The tray takes an answer only from the connection that owns + `org.freedesktop.Notifications` (a button signal any other program emits + is ignored), but a same-user program that stops the notification daemon + and takes that name answers for you, "Always allow app" included; - synthetic input into the GUI under X11 or XWayland can click its buttons; - a socket handed back into the app: a same-user program that started the app itself (connect first, then exec the installed binary, keeping a copy From b5f17cb3a338eeaaf4a6a1f84f097db4ec531a9e Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:32:45 +0200 Subject: [PATCH 120/125] fix(packaging): ask for the password on every pause and resume The tray's menu is a dbusmenu object any program of the user can click, and polkit kept a pause authorization for the tray process for five minutes, so within that window a forged click paused filtering with no dialog. org.projectcolony.firewall.pause is now auth_admin; rule import keeps auth_admin_keep. HARDENING names the dbusmenu route. --- CHANGELOG.md | 6 +++-- README.md | 2 +- crates/cfc-daemon/src/polkit.rs | 33 ++++++++++++++++++++++++--- docs/HARDENING.md | 11 ++++++--- docs/TROUBLESHOOTING.md | 4 ++-- pkg/README.md | 2 +- pkg/org.projectcolony.firewall.policy | 10 +++++--- 7 files changed, 53 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2d8c842..9a55862 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -118,8 +118,10 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). act on it, so a program could press "Always allow app" on its own prompt. - **Breaking: pause, resume and rule import from the app or tray ask for an administrator password** through polkit - (`org.projectcolony.firewall.pause`, `org.projectcolony.firewall.import-rules`, - `auth_admin_keep`). The policy file is installed by every package; polkit + (`org.projectcolony.firewall.pause`, asked every time, and + `org.projectcolony.firewall.import-rules`, kept a few minutes). Pause is + not kept because any program of the user can click the tray's menu over + D-Bus. The policy file is installed by every package; polkit is an optional dependency. Root is never asked, and answering a prompt or editing a rule never asks. The tray no longer blocks its prompt notifications while the dialog is open. diff --git a/README.md b/README.md index a0a1f54..78f956e 100644 --- a/README.md +++ b/README.md @@ -454,7 +454,7 @@ sudo cfc rules import-opensnitch /etc/opensnitchd/rules | who | read status, rules, logs, live view, prompts | answer prompts, add/edit/delete rules | pause, resume, import rules | |---|---|---|---| | root (`sudo cfc`) | yes | yes | yes | -| the installed Colony Firewall app and tray, run by a `colony-firewall` group member | yes | yes | after an administrator password (polkit, kept a few minutes) | +| the installed Colony Firewall app and tray, run by a `colony-firewall` group member | yes | yes | after an administrator password (polkit; asked for every pause or resume, kept a few minutes for imports) | | any other program of a group member, including `cfc` without sudo | yes | no | no | The daemon recognises the app and tray by the running image: it must be the diff --git a/crates/cfc-daemon/src/polkit.rs b/crates/cfc-daemon/src/polkit.rs index 08e5b63..0fa7ef6 100644 --- a/crates/cfc-daemon/src/polkit.rs +++ b/crates/cfc-daemon/src/polkit.rs @@ -5,9 +5,11 @@ //! the daemon asks polkit's `CheckAuthorization` for the calling process, //! with user interaction allowed, and the user's polkit agent shows its //! password dialog. The shipped policy -//! (`pkg/org.projectcolony.firewall.policy`) asks for `auth_admin_keep`, so -//! one password covers a few minutes. Root never gets here, and neither do -//! prompt answers or single-rule edits. +//! (`pkg/org.projectcolony.firewall.policy`) asks every time for pause and +//! resume (`auth_admin`: any same-user program can click the tray's menu +//! over D-Bus, so a kept authorization would let it pause unseen) and keeps +//! an import authorization a few minutes (`auth_admin_keep`). Root never +//! gets here, and neither do prompt answers or single-rule edits. //! //! One fresh system-bus connection per call: these calls are rare, and a //! connection kept open would be one more thing to babysit across D-Bus @@ -145,6 +147,31 @@ fn call_error(e: &zbus::Error) -> String { mod tests { use super::*; + /// The `` the shipped policy file gives `action`. + fn shipped_defaults(action: &str) -> Vec<&'static str> { + const POLICY: &str = include_str!("../../../pkg/org.projectcolony.firewall.policy"); + let block = POLICY + .split("")) + .nth(1) + .and_then(|rest| rest.split('<').next()) + .unwrap_or_else(|| panic!("{action} has no {key}")) + }) + .collect() + } + + #[test] + fn pause_asks_every_time_and_an_import_is_kept() { + assert_eq!(shipped_defaults(PAUSE), ["auth_admin"; 3]); + assert_eq!(shipped_defaults(IMPORT_RULES), ["auth_admin_keep"; 3]); + } + #[test] fn outcome_mapping() { let none = HashMap::new(); diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 4049d54..b6fee59 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -373,9 +373,14 @@ and tray only; it never makes anything else writable. **polkit for whole-firewall changes.** Pause, resume and rule import change everything at once, so even the app and tray need an administrator password for them: the daemon asks polkit (`org.projectcolony.firewall.pause`, -`org.projectcolony.firewall.import-rules`, both `auth_admin_keep`, so one -password covers a few minutes) and your session's polkit agent shows the -dialog. Without an agent (start one, e.g. `hyprpolkitagent` or +`auth_admin`, a password every time; `org.projectcolony.firewall.import-rules`, +`auth_admin_keep`, one password covers a few minutes) and your session's +polkit agent shows the dialog. Pause is never kept because the tray's menu +is a D-Bus object any program of yours can click (`com.canonical.dbusmenu` +`Event`): polkit, not the tray, is what stands between such a click and a +pause, and a kept authorization would have let it through for five minutes +after your last pause or resume. Such a click still raises a genuine +password dialog you did not ask for; cancel it. Without an agent (start one, e.g. `hyprpolkitagent` or `polkit-gnome`) or without polkit, the request is refused with that reason and `sudo cfc pause` still works. The daemon waits 120 s for an answer, then cancels the dialog. Answering a prompt and editing one rule never ask. diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index af8dea2..4cacf6e 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -325,8 +325,8 @@ journalctl -u colony-firewalld -g 'refusing a firewall change' ## Pause, resume or import asks for a password, or fails Pause, resume and rule import change the whole firewall at once, so the app -and tray need an administrator password for them (polkit, kept for a few -minutes). Root (`sudo cfc pause`) is never asked. What the refusals mean: +and tray need an administrator password for them (polkit: every time for +pause and resume, kept a few minutes for an import). Root (`sudo cfc pause`) is never asked. What the refusals mean: - **"authorization dialog dismissed"**: you cancelled it. - **"no polkit authentication agent answered in your session"**: nothing in diff --git a/pkg/README.md b/pkg/README.md index ae039d4..9bb445d 100644 --- a/pkg/README.md +++ b/pkg/README.md @@ -45,7 +45,7 @@ Key design points: programs or from `sudo cfc`. - **`org.projectcolony.firewall.policy`** declares the polkit actions the daemon asks about when the app or tray pauses, resumes or imports rules - (`auth_admin_keep`). Installed to `/usr/share/polkit-1/actions/`; polkit + (`auth_admin` for pause and resume, `auth_admin_keep` for imports). Installed to `/usr/share/polkit-1/actions/`; polkit is an optional dependency, and without it only `sudo cfc` can do those. - **XDG autostart** launches the GUI in every desktop session so prompts actually reach the user. Per-user opt-out: copy the file to diff --git a/pkg/org.projectcolony.firewall.policy b/pkg/org.projectcolony.firewall.policy index 8d351bc..a63b391 100644 --- a/pkg/org.projectcolony.firewall.policy +++ b/pkg/org.projectcolony.firewall.policy @@ -6,6 +6,10 @@ Colony Firewall Control: actions the daemon asks polkit about when the installed app or tray requests them. Root (sudo cfc) is never asked. Installed to /usr/share/polkit-1/actions/; polkitd picks it up by itself. + + Pause asks every time, without the few minutes a kept authorization would + give: the tray's menu can be clicked over D-Bus by any program of the + user, and a kept authorization would let such a click pause silently. --> @@ -17,9 +21,9 @@ Pause or resume Colony Firewall filtering Authentication is required to pause or resume the firewall - auth_admin_keep - auth_admin_keep - auth_admin_keep + auth_admin + auth_admin + auth_admin From e674f64dfc7238afe5393af3179c15e586ae76cd Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:35:18 +0200 Subject: [PATCH 121/125] feat(daemon)!: require polkit for an Allow rule that names no program An enabled Allow with no exe predicate lets every program through wherever it matches, so two of them (protocol tcp, protocol udp) did what a pause does, without the pause password, from the app's rule editor. Storing one from the app or tray, through UpsertRule or a customized prompt answer, now needs org.projectcolony.firewall.allow-every-program (auth_admin_keep). A refused prompt answer still applies once and saves no rule. Program rules, Deny rules and root never ask. The app waits for the dialog on rule saves and customized answers. BREAKING CHANGE: saving an Allow rule that names no program from the app asks for an administrator password. --- CHANGELOG.md | 11 ++- README.md | 4 +- crates/cfc-client/src/lib.rs | 3 +- crates/cfc-daemon/src/ipc.rs | 130 +++++++++++++++++++++++++- crates/cfc-daemon/src/polkit.rs | 18 ++-- crates/cfc-ui/src/main.rs | 18 +++- docs/ARCHITECTURE.md | 9 +- docs/HARDENING.md | 45 ++++++--- docs/TROUBLESHOOTING.md | 15 +-- pkg/README.md | 8 +- pkg/org.projectcolony.firewall.policy | 10 ++ 11 files changed, 224 insertions(+), 47 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9a55862..b2964b8 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -121,9 +121,14 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). (`org.projectcolony.firewall.pause`, asked every time, and `org.projectcolony.firewall.import-rules`, kept a few minutes). Pause is not kept because any program of the user can click the tray's menu over - D-Bus. The policy file is installed by every package; polkit - is an optional dependency. Root is never asked, and answering a prompt or - editing a rule never asks. The tray no longer blocks its prompt + D-Bus. Storing an enabled Allow rule that names no program + (`allow --protocol tcp`), from the rule editor or a customized prompt + answer, asks too (`org.projectcolony.firewall.allow-every-program`, kept a + few minutes): it lets every program through, which is what a pause does, + and it was the way around the pause password. The policy file is + installed by every package; polkit is an optional dependency. Root is + never asked, and answering a prompt or editing a rule that names a + program, or a Deny, never asks. The tray no longer blocks its prompt notifications while the dialog is open. Upgrading from 0.7.0: diff --git a/README.md b/README.md index 78f956e..a7cf9ce 100644 --- a/README.md +++ b/README.md @@ -451,10 +451,10 @@ sudo cfc rules import-opensnitch /etc/opensnitchd/rules ### Who can change what -| who | read status, rules, logs, live view, prompts | answer prompts, add/edit/delete rules | pause, resume, import rules | +| who | read status, rules, logs, live view, prompts | answer prompts, add/edit/delete rules | pause, resume, import rules, Allow rules that name no program | |---|---|---|---| | root (`sudo cfc`) | yes | yes | yes | -| the installed Colony Firewall app and tray, run by a `colony-firewall` group member | yes | yes | after an administrator password (polkit; asked for every pause or resume, kept a few minutes for imports) | +| the installed Colony Firewall app and tray, run by a `colony-firewall` group member | yes | yes | after an administrator password (polkit; asked for every pause or resume, kept a few minutes for the others) | | any other program of a group member, including `cfc` without sudo | yes | no | no | The daemon recognises the app and tray by the running image: it must be the diff --git a/crates/cfc-client/src/lib.rs b/crates/cfc-client/src/lib.rs index f962513..37950e1 100644 --- a/crates/cfc-client/src/lib.rs +++ b/crates/cfc-client/src/lib.rs @@ -190,7 +190,8 @@ impl Client { } /// Connects for a request that may wait on an administrator password - /// (pause, resume): the daemon asks polkit and polkit asks the user. + /// (pause, resume, import, an Allow rule for every program): the daemon + /// asks polkit and polkit asks the user. pub async fn connect_interactive(socket_path: impl AsRef) -> Result { Self::connect_with_timeout(socket_path, INTERACTIVE_TIMEOUT).await } diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index fc2e72d..3299994 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -28,10 +28,11 @@ //! file from outside sealed directories. `require_group = false` waives the group proof for //! official clients only. //! -//! Pause, resume and `ApplyRules` change the whole firewall at once, so -//! an official client also needs a polkit authorization for them -//! ([`crate::polkit`]); root does not. Answering a prompt and editing one -//! rule never ask for a password. +//! Pause, resume and `ApplyRules` change the whole firewall at once, and +//! an Allow rule that names no program opens it for every program, so an +//! official client also needs a polkit authorization for them +//! ([`crate::polkit`]); root does not. Answering a prompt and editing a +//! rule that names a program (or denies) never ask for a password. //! //! Every other peer is read-only: its change RPCs get PERMISSION_DENIED //! with the reason, its prompt subscription does not count as a UI (so @@ -158,6 +159,13 @@ fn log_refusal(rpc: &'static str, peer: PeerId, status: &Status) { ); } +/// An enabled Allow that names no program: wherever it matches, every +/// program gets through without a prompt, which is a pause for that much of +/// the traffic. One `allow --protocol tcp` is most of one. +fn opens_for_every_program(rule: &cfc_core::Rule) -> bool { + rule.enabled && rule.action == cfc_core::Action::Allow && !rule.scope.names_program() +} + fn bind_prompt_allow( rule: &mut cfc_core::Rule, binding: &crate::prompts::PromptBinding, @@ -604,6 +612,9 @@ impl FirewallService { .ok_or_else(|| Status::invalid_argument("rule required"))?; let mut rule = convert::rule_from_pb(&proto).map_err(Status::invalid_argument)?; convert::reject_unpersistable_duration(rule.duration).map_err(Status::invalid_argument)?; + self.confirm_generic_allow(peer, &rule) + .await + .map_err(Status::permission_denied)?; // Every caller must select the canonical mapped target explicitly. // Missing targets with unchanged ancestry remain valid for preinstallation. if !keeps_stored_exe(&self.engine.snapshot(), &rule) { @@ -713,6 +724,24 @@ impl FirewallService { }) } + /// From the app or tray, an Allow that applies to every program needs + /// polkit's [`crate::polkit::GENERIC_ALLOW`]; root never does. `Err` is + /// the reason to show. + async fn confirm_generic_allow( + &self, + peer: PeerId, + rule: &cfc_core::Rule, + ) -> Result<(), String> { + if !opens_for_every_program(rule) || gate(peer.uid, self.own_uid, true) == Gate::Privileged + { + return Ok(()); + } + let action = crate::polkit::GENERIC_ALLOW; + (self.polkit)(peer, action) + .await + .map_err(|reason| format!("{reason} (polkit action {action})")) + } + fn policy(&self) -> crate::config::DefaultPolicy { *self .policy @@ -873,6 +902,18 @@ impl Firewall for FirewallService { Ok(rule) }) { Ok(mut rule) => { + // "Customize" in the app can turn the answer into a rule + // for every program; that needs what UpsertRule needs. + if let Err(error) = self.confirm_generic_allow(peer, &rule).await { + persist_error = format!("the verdict was applied, but the standing rule could not be saved: {error}"); + return Ok(Response::new(VerdictResponse { + accepted, + persisted_rule_id: String::new(), + persist_error, + persist_note, + error: String::new(), + })); + } // Persist only an explicit mapped target. The one-time // verdict is already applied, so report a rejected standing // policy through persist_error rather than retrying it. @@ -2074,6 +2115,87 @@ mod tests { ) } + /// `allow dst_port 443` for every program, or for `exe` when given. + fn allow_pb(exe: Option<&str>, enabled: bool) -> RuleInfo { + let mut rule = cfc_core::Rule::new( + "https", + cfc_core::Action::Allow, + cfc_core::RuleScope { + dst_port: Some(443), + exe_sha256: exe.map(str::to_string), + ..cfc_core::RuleScope::any() + }, + ); + rule.enabled = enabled; + convert::rule_to_pb(&rule) + } + + async fn upsert(svc: &FirewallService, rule: RuleInfo, who: PeerId) -> Result<(), Status> { + svc.upsert_rule(request(UpsertRuleRequest { rule: Some(rule) }, who)) + .await + .map(drop) + } + + #[test] + fn only_an_enabled_allow_naming_no_program_opens_for_every_program() { + let rule = |pb: RuleInfo| convert::rule_from_pb(&pb).unwrap(); + assert!(opens_for_every_program(&rule(allow_pb(None, true)))); + assert!(!opens_for_every_program(&rule(allow_pb(None, false)))); + let hash = "a".repeat(64); + assert!(!opens_for_every_program(&rule(allow_pb(Some(&hash), true)))); + assert!(!opens_for_every_program(&rule(rule_pb())), "a Deny"); + } + + #[tokio::test] + async fn an_allow_for_every_program_needs_polkit_from_the_app() { + let official = peer(1000, GROUP_GID, OFFICIAL_PID); + let svc = service_with(true, polkit_denies); + let status = upsert(&svc, allow_pb(None, true), official) + .await + .unwrap_err(); + assert_eq!(status.code(), tonic::Code::PermissionDenied); + assert!( + status.message().contains(crate::polkit::GENERIC_ALLOW), + "{}", + status.message() + ); + assert_eq!(svc.engine.rule_count(), 0, "nothing was stored"); + + // Granted, it is stored; a disabled one, a program's Allow and a + // Deny never ask, and neither does root. + let svc = service_with(true, polkit_allows); + upsert(&svc, allow_pb(None, true), official).await.unwrap(); + let svc = service_with(true, polkit_must_not_be_asked); + let hash = "a".repeat(64); + upsert(&svc, allow_pb(None, false), official).await.unwrap(); + upsert(&svc, allow_pb(Some(&hash), true), official) + .await + .unwrap(); + upsert(&svc, rule_pb(), official).await.unwrap(); + upsert(&svc, allow_pb(None, true), peer(0, 0, 1)) + .await + .unwrap(); + assert_eq!(svc.engine.rule_count(), 4); + } + + #[tokio::test] + async fn a_prompt_answer_customized_to_every_program_needs_polkit() { + let svc = service_with(true, polkit_denies); + let _ui = pending_prompt(&svc).await; + let mut req = answer(peer(1000, GROUP_GID, OFFICIAL_PID)); + req.get_mut().duration = cfc_proto::v1::Duration::Always as i32; + req.get_mut().persist_scope = allow_pb(None, true).scope; + let reply = svc.submit_verdict(req).await.unwrap().into_inner(); + assert!(reply.accepted, "the answer itself still applies"); + assert!(reply.persisted_rule_id.is_empty()); + assert!( + reply.persist_error.contains(crate::polkit::GENERIC_ALLOW), + "{}", + reply.persist_error + ); + assert_eq!(svc.engine.rule_count(), 0); + } + #[tokio::test] async fn read_only_peer_cannot_answer_even_its_own_prompt() { let svc = service(true); diff --git a/crates/cfc-daemon/src/polkit.rs b/crates/cfc-daemon/src/polkit.rs index 0fa7ef6..94484ed 100644 --- a/crates/cfc-daemon/src/polkit.rs +++ b/crates/cfc-daemon/src/polkit.rs @@ -1,15 +1,17 @@ //! Administrator authorization through polkit. //! -//! Pause, resume and rule import change the whole firewall at once, so even -//! the official app and tray must have them confirmed by an administrator: +//! Pause, resume and rule import change the whole firewall at once, and an +//! Allow rule that names no program opens it for every program, so even the +//! official app and tray must have them confirmed by an administrator: //! the daemon asks polkit's `CheckAuthorization` for the calling process, //! with user interaction allowed, and the user's polkit agent shows its //! password dialog. The shipped policy //! (`pkg/org.projectcolony.firewall.policy`) asks every time for pause and //! resume (`auth_admin`: any same-user program can click the tray's menu //! over D-Bus, so a kept authorization would let it pause unseen) and keeps -//! an import authorization a few minutes (`auth_admin_keep`). Root never -//! gets here, and neither do prompt answers or single-rule edits. +//! an import or allow-every-program authorization a few minutes +//! (`auth_admin_keep`). Root never gets here, and neither do prompt answers +//! or edits of rules that name a program or deny. //! //! One fresh system-bus connection per call: these calls are rare, and a //! connection kept open would be one more thing to babysit across D-Bus @@ -25,6 +27,9 @@ use zbus::zvariant::Value; pub const PAUSE: &str = "org.projectcolony.firewall.pause"; /// Import, replace or bundle-install rules (`ApplyRules`). pub const IMPORT_RULES: &str = "org.projectcolony.firewall.import-rules"; +/// Store an enabled Allow rule that names no program (`UpsertRule`, or a +/// prompt answer customized into one), which lets every program through. +pub const GENERIC_ALLOW: &str = "org.projectcolony.firewall.allow-every-program"; /// How long the daemon waits for the user to answer the dialog. Clients wait /// longer (`cfc_client::INTERACTIVE_TIMEOUT`), so this answer reaches them. pub const TIMEOUT: Duration = Duration::from_secs(120); @@ -64,7 +69,7 @@ pub async fn check(peer: PeerId, action: &'static str) -> Result<(), String> { let unreachable = |e: zbus::Error| { format!( "this needs administrator authorization, but the system D-Bus is unreachable \ - ({e}); use sudo cfc pause, resume or rules import" + ({e}); use sudo cfc instead" ) }; let bus = zbus::connection::Builder::system() @@ -137,7 +142,7 @@ fn call_error(e: &zbus::Error) -> String { _ => false, }; if unknown { - "polkit is not installed or not running; use sudo cfc pause, resume or rules import".into() + "polkit is not installed or not running; use sudo cfc instead".into() } else { format!("polkit check failed: {e}") } @@ -170,6 +175,7 @@ mod tests { fn pause_asks_every_time_and_an_import_is_kept() { assert_eq!(shipped_defaults(PAUSE), ["auth_admin"; 3]); assert_eq!(shipped_defaults(IMPORT_RULES), ["auth_admin_keep"; 3]); + assert_eq!(shipped_defaults(GENERIC_ALLOW), ["auth_admin_keep"; 3]); } #[test] diff --git a/crates/cfc-ui/src/main.rs b/crates/cfc-ui/src/main.rs index 906b398..6e4f39b 100644 --- a/crates/cfc-ui/src/main.rs +++ b/crates/cfc-ui/src/main.rs @@ -1746,9 +1746,14 @@ async fn fetch_rules(path: PathBuf) -> Result, String> { /// or changed, and the enable toggle sends the stored path back unchanged, /// which the daemon accepts as is. Checking it again refused to toggle a rule /// whose target had since become an alias. +/// +/// An Allow that names no program waits on an administrator password: the +/// daemon asks polkit first. async fn upsert_rule(path: PathBuf, rule: proto::RuleInfo) -> Result { let line = saved_rule_line(&rule); - let mut client = Client::connect(&path).await.map_err(|e| e.to_string())?; + let mut client = Client::connect_interactive(&path) + .await + .map_err(|e| e.to_string())?; client.upsert_rule(rule).await.map_err(|e| e.to_string())?; Ok(line) } @@ -1912,9 +1917,14 @@ async fn submit_verdict( applied, message, }; - let mut client = Client::connect(&path) - .await - .map_err(|e| fail(false, e.to_string()))?; + // A customized rule may be an Allow for every program, which waits on an + // administrator password before it is stored. + let connected = if require_confirmed_rule { + Client::connect_interactive(&path).await + } else { + Client::connect(&path).await + }; + let mut client = connected.map_err(|e| fail(false, e.to_string()))?; let outcome = client .submit_verdict(&prompt_id, action, duration, scope) .await diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 351fbaf..8333c66 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -378,9 +378,12 @@ the entire attack surface. Two layers: binaries by device and inode, holds this connection's client end itself (`UNIX_DIAG` names it) and mapped no executable file from outside sealed directories. The check runs on the blocking pool. -3. **polkit.** `SetPaused` (pause and resume) and `ApplyRules` from the app - or tray also need `CheckAuthorization` for - `org.projectcolony.firewall.pause` or `org.projectcolony.firewall.import-rules` +3. **polkit.** `SetPaused` (pause and resume), `ApplyRules`, and an + `UpsertRule` (or a customized prompt answer) storing an enabled Allow that + names no program, from the app or tray, also need `CheckAuthorization` + for `org.projectcolony.firewall.pause`, + `org.projectcolony.firewall.import-rules` or + `org.projectcolony.firewall.allow-every-program` (`polkit.rs`, one system-bus connection per call, 120 s timeout, the dialog cancelled on expiry). Root is never asked. diff --git a/docs/HARDENING.md b/docs/HARDENING.md index b6fee59..1130ce2 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -322,7 +322,7 @@ and the daemon checks the caller per RPC: |-----|-------------------|-----------------------|-------------------| | `ListRules`, `GetStatus`, `ListEvents`, `StreamConnections` | yes | yes | yes | | `StreamPrompts` | yes, counts as a UI | yes, counts as a UI | sees its prompts, does **not** count as a UI | -| `SubmitVerdict`, `UpsertRule`, `DeleteRule` | yes | yes, no password | refused | +| `SubmitVerdict`, `UpsertRule`, `DeleteRule` | yes | yes, no password (an Allow that names no program: after polkit authorization) | refused | | `SetPaused` (pause **and** resume), `ApplyRules` (import, replace, bundles) | yes, no password | after polkit authorization | refused | There is no RPC that changes `[default_policy]`: that is root editing @@ -371,19 +371,30 @@ by a proved group member. `require_group = false` waives that for the app and tray only; it never makes anything else writable. **polkit for whole-firewall changes.** Pause, resume and rule import change -everything at once, so even the app and tray need an administrator password -for them: the daemon asks polkit (`org.projectcolony.firewall.pause`, -`auth_admin`, a password every time; `org.projectcolony.firewall.import-rules`, -`auth_admin_keep`, one password covers a few minutes) and your session's -polkit agent shows the dialog. Pause is never kept because the tray's menu -is a D-Bus object any program of yours can click (`com.canonical.dbusmenu` -`Event`): polkit, not the tray, is what stands between such a click and a -pause, and a kept authorization would have let it through for five minutes -after your last pause or resume. Such a click still raises a genuine -password dialog you did not ask for; cancel it. Without an agent (start one, e.g. `hyprpolkitagent` or -`polkit-gnome`) or without polkit, the request is refused with that reason -and `sudo cfc pause` still works. The daemon waits 120 s for an answer, then -cancels the dialog. Answering a prompt and editing one rule never ask. +everything at once, and an Allow rule that names no program (no `exe_path` +or `exe_sha256`: `allow --protocol tcp`, `allow --dst-port 443`) lets every +program through wherever it matches, which is a pause for that traffic. So +even the app and tray need an administrator password for them. The daemon +asks polkit and your session's polkit agent shows the dialog: + +| action | for | default | +|--------|-----|---------| +| `org.projectcolony.firewall.pause` | `SetPaused`, pause and resume | `auth_admin`, a password every time | +| `org.projectcolony.firewall.import-rules` | `ApplyRules` (import, replace, bundles) | `auth_admin_keep`, one password covers a few minutes | +| `org.projectcolony.firewall.allow-every-program` | `UpsertRule` storing an enabled Allow that names no program, or a prompt answer customized into one | `auth_admin_keep` | + +Pause is never kept because the tray's menu is a D-Bus object any program of +yours can click (`com.canonical.dbusmenu` `Event`): polkit, not the tray, +stands between such a click and a pause, and a kept authorization would let +it through for five minutes after your last pause or resume. Such a click +still raises a genuine password dialog you did not ask for; cancel it. + +Without an agent (start one, e.g. `hyprpolkitagent` or `polkit-gnome`) or +without polkit, the request is refused with that reason and `sudo cfc` still +works. The daemon waits 120 s for an answer, then cancels the dialog. +Answering a prompt and editing a rule that names a program, or a Deny, +never ask. A customized prompt answer that polkit refuses still applies +once; only its standing rule is not saved. **What this still trusts.** The check is about the *process*, so code that runs inside the installed app is the app: @@ -410,7 +421,11 @@ runs inside the installed app is the app: `org.freedesktop.Notifications` (a button signal any other program emits is ignored), but a same-user program that stops the notification daemon and takes that name answers for you, "Always allow app" included; -- synthetic input into the GUI under X11 or XWayland can click its buttons; +- synthetic input into the GUI can click its buttons: under X11 or + XWayland, and on Wayland compositors that offer the virtual-keyboard + protocol to every client (Hyprland and sway do; `wtype` uses it). That + reaches every rule a program-scoped edit can make, not the polkit-guarded + ones above; - a socket handed back into the app: a same-user program that started the app itself (connect first, then exec the installed binary, keeping a copy of the connection in a child) can pass that copy into the sealed app as a diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 4cacf6e..dfc0029 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -322,11 +322,14 @@ The journal names the caller and the reason for every refusal: journalctl -u colony-firewalld -g 'refusing a firewall change' ``` -## Pause, resume or import asks for a password, or fails +## Pause, resume, import or an Allow rule asks for a password, or fails -Pause, resume and rule import change the whole firewall at once, so the app -and tray need an administrator password for them (polkit: every time for -pause and resume, kept a few minutes for an import). Root (`sudo cfc pause`) is never asked. What the refusals mean: +Pause, resume and rule import change the whole firewall at once, and an +Allow rule that names no program lets every program through, so the app and +tray need an administrator password for them (polkit: every time for pause +and resume, kept a few minutes for the others). Root (`sudo cfc`) is never +asked. Rules that name a program, and Deny rules, never ask. What the +refusals mean: - **"authorization dialog dismissed"**: you cancelled it. - **"no polkit authentication agent answered in your session"**: nothing in @@ -336,8 +339,8 @@ pause and resume, kept a few minutes for an import). Root (`sudo cfc pause`) is - **"polkit is not installed or not running"** or **"the system D-Bus is unreachable"**: install polkit, or use `sudo cfc`. - **"not authorized by polkit policy"**: a local polkit rule denies - `org.projectcolony.firewall.pause` or `org.projectcolony.firewall.import-rules` - for you. `pkaction --verbose --action-id org.projectcolony.firewall.pause` + `org.projectcolony.firewall.pause`, `org.projectcolony.firewall.import-rules` + or `org.projectcolony.firewall.allow-every-program` for you. `pkaction --verbose --action-id org.projectcolony.firewall.pause` shows the defaults; a missing action means the policy file is not installed in `/usr/share/polkit-1/actions/`. - **"authorization timed out after 120 s"**: the dialog was left open; the diff --git a/pkg/README.md b/pkg/README.md index 9bb445d..f0ea050 100644 --- a/pkg/README.md +++ b/pkg/README.md @@ -44,9 +44,11 @@ Key design points: lets the installed app and tray connect; changes come from those two programs or from `sudo cfc`. - **`org.projectcolony.firewall.policy`** declares the polkit actions the - daemon asks about when the app or tray pauses, resumes or imports rules - (`auth_admin` for pause and resume, `auth_admin_keep` for imports). Installed to `/usr/share/polkit-1/actions/`; polkit - is an optional dependency, and without it only `sudo cfc` can do those. + daemon asks about when the app or tray pauses, resumes, imports rules or + stores an Allow rule that names no program (`auth_admin` for pause and + resume, `auth_admin_keep` for the others). Installed to + `/usr/share/polkit-1/actions/`; polkit is an optional dependency, and + without it only `sudo cfc` can do those. - **XDG autostart** launches the GUI in every desktop session so prompts actually reach the user. Per-user opt-out: copy the file to `~/.config/autostart/` and set `Hidden=true`. diff --git a/pkg/org.projectcolony.firewall.policy b/pkg/org.projectcolony.firewall.policy index a63b391..1c2ff5b 100644 --- a/pkg/org.projectcolony.firewall.policy +++ b/pkg/org.projectcolony.firewall.policy @@ -36,4 +36,14 @@ auth_admin_keep + + + Allow Colony Firewall traffic for every program + Authentication is required to let every program through the firewall + + auth_admin_keep + auth_admin_keep + auth_admin_keep + + From 2617939f283447c15a37fe5a896933a18fb2203f Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:38:45 +0200 Subject: [PATCH 122/125] fix: say when an upgrade leaves the app and tray read-only After an upgrade the daemon refuses a running app or tray (its image is no longer the installed binary), but the tray only logged refused answers, its old prompt stream kept counting as a UI so every prompt waited out the timeout, and the pacman note was shown only when crossing 0.8.0. The daemon now checks an answering prompt subscription again before each prompt and ends it with the reason when it fails, which drops it from the census. The tray notifies once when its binary was replaced and shows the reason for a refused answer. The pacman and RPM scriptlets ask for a restart on every upgrade. --- CHANGELOG.md | 10 +- crates/cfc-daemon/src/ipc.rs | 187 +++++++++++++++------ crates/cfc-daemon/src/prompts.rs | 7 + crates/cfc-tray/src/main.rs | 13 ++ crates/cfc-tray/src/model.rs | 43 +++++ docs/ARCHITECTURE.md | 5 +- docs/HARDENING.md | 5 +- docs/TROUBLESHOOTING.md | 6 +- packaging/rpm/colony-firewall-control.spec | 2 + pkg/colony-firewall-control.install | 10 +- pkg/colony.json | 2 +- 11 files changed, 229 insertions(+), 61 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index b2964b8..f57c403 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -135,9 +135,13 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). 1. Scripts that ran `cfc` as a regular user to change rules, pause or answer prompts must use `sudo cfc`. Reading (`status`, `rules list`, `log`, `live`) is unchanged. - 2. Restart Colony Firewall and its tray after the upgrade. The 0.7 - processes, and any process still running a replaced binary, are - read-only until relaunched. + 2. Restart Colony Firewall and its tray after the upgrade, and after every + later one. The 0.7 processes, and any process still running a replaced + binary, are read-only until relaunched: the tray says so in a + notification, a refused prompt answer shows the daemon's reason, and + such a process stops counting as a UI at the next prompt, so later + prompts are not held for the full timeout. The pacman and RPM scriptlets print + this on every upgrade. 3. Pausing from the app or tray needs a polkit agent in the session (most desktops run one; on Hyprland, `hyprpolkitagent`). Without one the request is refused with that reason and `sudo cfc pause` works. diff --git a/crates/cfc-daemon/src/ipc.rs b/crates/cfc-daemon/src/ipc.rs index 3299994..0fdc468 100644 --- a/crates/cfc-daemon/src/ipc.rs +++ b/crates/cfc-daemon/src/ipc.rs @@ -497,6 +497,49 @@ impl PromptAudience { // Service // --------------------------------------------------------------------------- +/// Who may change the firewall: what [`Control::standing`] needs, apart from +/// the service so a prompt stream can ask again while it runs. +#[derive(Clone)] +struct Control { + auth: SocketAuth, + /// The daemon's effective uid; see [`Gate::Privileged`]. + own_uid: u32, + /// `[ipc] official_clients`, bound at startup. + official_clients: Arc<[PathBuf]>, + official: OfficialCheck, +} + +impl Control { + /// May `peer` change the firewall? `Ok(None)` for a privileged peer, + /// `Ok(Some(exe))` for the official app or tray, `Err` (with the reason + /// the client shows) for everyone else, who is read-only. + async fn standing(&self, peer: PeerId) -> Result, Status> { + let group_ok = !self.auth.require_group || peer_is_group_member(peer, self.auth.group_gid); + match gate(peer.uid, self.own_uid, group_ok) { + Gate::Privileged => Ok(None), + Gate::DenyGroup => Err(Status::permission_denied(format!( + "firewall changes require root, or the installed Colony Firewall app or \ + tray run by a member of group '{}'", + self.auth.group + ))), + Gate::NeedOfficial => { + let (check, list) = (self.official, self.official_clients.clone()); + tokio::task::spawn_blocking(move || check(&peer, &list)) + .await + .unwrap_or_else(|error| Err(format!("the identity check failed ({error})"))) + .map(Some) + .map_err(|reason| { + Status::permission_denied(format!( + "read-only access: {reason}. Firewall changes are accepted only \ + from the installed Colony Firewall app and tray, or from root \ + (sudo cfc ...)." + )) + }) + } + } + } +} + struct FirewallService { engine: Engine, store: RuleStore, @@ -505,12 +548,7 @@ struct FirewallService { stats: Stats, /// Live default policy; SIGHUP swaps it, so status reflects reloads. policy: SharedPolicy, - auth: SocketAuth, - /// The daemon's effective uid; see [`Gate::Privileged`]. - own_uid: u32, - /// `[ipc] official_clients`, bound at startup. - official_clients: Arc<[PathBuf]>, - official: OfficialCheck, + control: Control, polkit: PolkitCheck, audience: Arc, /// Wall-clock deadline of the current pause, 0 when not paused. Held @@ -532,7 +570,7 @@ impl FirewallService { if level == Access::ReadOnly { return Ok(peer); } - let outcome = match self.standing(peer).await { + let outcome = match self.control.standing(peer).await { // Only an official client is asked; root never is. Ok(Some(exe)) => match level { Access::Elevated(action) => (self.polkit)(peer, action) @@ -573,35 +611,6 @@ impl FirewallService { } } - /// May `peer` change the firewall? `Ok(None)` for a privileged peer, - /// `Ok(Some(exe))` for the official app or tray, `Err` (with the reason - /// the client shows) for everyone else, who is read-only. - async fn standing(&self, peer: PeerId) -> Result, Status> { - let group_ok = !self.auth.require_group || peer_is_group_member(peer, self.auth.group_gid); - match gate(peer.uid, self.own_uid, group_ok) { - Gate::Privileged => Ok(None), - Gate::DenyGroup => Err(Status::permission_denied(format!( - "firewall changes require root, or the installed Colony Firewall app or \ - tray run by a member of group '{}'", - self.auth.group - ))), - Gate::NeedOfficial => { - let (check, list) = (self.official, self.official_clients.clone()); - tokio::task::spawn_blocking(move || check(&peer, &list)) - .await - .unwrap_or_else(|error| Err(format!("the identity check failed ({error})"))) - .map(Some) - .map_err(|reason| { - Status::permission_denied(format!( - "read-only access: {reason}. Firewall changes are accepted only \ - from the installed Colony Firewall app and tray, or from root \ - (sudo cfc ...)." - )) - }) - } - } - } - async fn upsert_rule_checked( &self, peer: PeerId, @@ -732,7 +741,8 @@ impl FirewallService { peer: PeerId, rule: &cfc_core::Rule, ) -> Result<(), String> { - if !opens_for_every_program(rule) || gate(peer.uid, self.own_uid, true) == Gate::Privileged + if !opens_for_every_program(rule) + || gate(peer.uid, self.control.own_uid, true) == Gate::Privileged { return Ok(()); } @@ -763,7 +773,7 @@ impl Firewall for FirewallService { .await?; // Every peer is shown the prompts addressed to it; only one that may // answer them is counted as a UI and enters their audience. - let answering = match self.standing(peer).await { + let answering = match self.control.standing(peer).await { Ok(_) => true, Err(status) => { tracing::debug!( @@ -779,6 +789,10 @@ impl Firewall for FirewallService { let mut sub = self.router.subscribe(peer.uid, answering); let audience = self.audience.clone(); let uid = peer.uid; + // Asked again before each prompt: an upgrade replaces the binary + // under a running app or tray, whose answers are then refused. Left + // counted as a UI, it would hold every prompt for the full timeout. + let recheck = answering.then(|| self.control.clone()); tokio::spawn(async move { loop { match sub.recv().await { @@ -797,6 +811,15 @@ impl Firewall for FirewallService { } continue; } + if let Some(control) = &recheck { + if let Err(status) = control.standing(peer).await { + // Ending the stream drops this census entry; + // the client resubscribes read-only and is + // told why. + let _ = tx.send(Err(status)).await; + break; + } + } // Record before handing the event over: this // subscriber is about to learn the prompt id, so it // must be entitled to answer it by the time it can. @@ -1633,10 +1656,12 @@ pub async fn spawn( router, stats, policy, - auth, - own_uid: nix::unistd::geteuid().as_raw(), - official_clients: opts.ipc.official_clients.clone().into(), - official: crate::official::check, + control: Control { + auth, + own_uid: nix::unistd::geteuid().as_raw(), + official_clients: opts.ipc.official_clients.clone().into(), + official: crate::official::check, + }, polkit: |peer, action| Box::pin(crate::polkit::check(peer, action)), audience: Arc::new(PromptAudience::default()), resume_at_ms: Arc::new(AtomicI64::new(0)), @@ -1727,7 +1752,19 @@ mod tests { /// The pid the stub official check accepts. const OFFICIAL_PID: i32 = 77; + /// An official client whose binary is replaced after its first check. + const UPGRADED_PID: i32 = 4243; + fn stub_official(peer: &PeerId, _: &[PathBuf]) -> Result { + static UPGRADED_CHECKS: std::sync::atomic::AtomicUsize = + std::sync::atomic::AtomicUsize::new(0); + if peer.pid == Some(UPGRADED_PID) { + return if UPGRADED_CHECKS.fetch_add(1, Ordering::Relaxed) == 0 { + Ok(PathBuf::from("/usr/bin/colony-firewall-tray")) + } else { + Err("this colony-firewall-tray is not the installed one (restart it after an upgrade)".into()) + }; + } if peer.pid == Some(OFFICIAL_PID) { Ok(PathBuf::from("/usr/bin/colony-firewall")) } else { @@ -1780,15 +1817,17 @@ mod tests { observed_tx: broadcast::channel(16).0, stats, policy, - auth: SocketAuth { - group: "cfc-test".into(), - group_gid: Some(GROUP_GID), - group_gated: true, - require_group, + control: Control { + auth: SocketAuth { + group: "cfc-test".into(), + group_gid: Some(GROUP_GID), + group_gated: true, + require_group, + }, + own_uid: OWN_UID, + official_clients: Arc::from(Vec::new()), + official: stub_official, }, - own_uid: OWN_UID, - official_clients: Arc::from(Vec::new()), - official: stub_official, polkit, audience: Arc::new(PromptAudience::default()), resume_at_ms: Arc::new(AtomicI64::new(0)), @@ -2196,6 +2235,54 @@ mod tests { assert_eq!(svc.engine.rule_count(), 0); } + #[tokio::test] + async fn an_upgraded_client_stops_counting_as_a_ui_at_its_next_prompt() { + use std::net::{IpAddr, Ipv4Addr}; + let svc = service(true); + let mut stream = svc + .stream_prompts(request( + SubscribeRequest::default(), + peer(1000, GROUP_GID, UPGRADED_PID), + )) + .await + .unwrap() + .into_inner(); + assert!(svc.router.has_answering_ui(1000), "official at subscribe"); + + let (tx, rx) = mpsc::channel(1); + tokio::spawn(crate::prompts::run_router_task(rx, svc.router.clone())); + let mut process = cfc_core::Process::unknown(4321); + process.uid = Some(1000); + tx.send(PromptRequest { + prompt_id: 9, + connection: cfc_core::Connection::new( + cfc_core::Protocol::Tcp, + cfc_core::Direction::Outbound, + IpAddr::V4(Ipv4Addr::LOCALHOST), + 40000, + IpAddr::V4(Ipv4Addr::new(1, 1, 1, 1)), + 443, + ), + process, + undecided: None, + }) + .await + .unwrap(); + + let status = stream.next().await.unwrap().unwrap_err(); + assert_eq!(status.code(), tonic::Code::PermissionDenied); + assert!(status.message().contains("restart"), "{}", status.message()); + assert!(stream.next().await.is_none(), "the stream ended"); + for _ in 0..1000 { + if !svc.router.has_answering_ui(1000) { + break; + } + tokio::task::yield_now().await; + } + assert!(!svc.router.has_answering_ui(1000), "no longer a UI"); + assert!(!svc.audience.allows(9, 1000), "and never told the id"); + } + #[tokio::test] async fn read_only_peer_cannot_answer_even_its_own_prompt() { let svc = service(true); diff --git a/crates/cfc-daemon/src/prompts.rs b/crates/cfc-daemon/src/prompts.rs index 1127b0c..7cca9f7 100644 --- a/crates/cfc-daemon/src/prompts.rs +++ b/crates/cfc-daemon/src/prompts.rs @@ -200,6 +200,13 @@ impl PromptRouter { } } + /// Whether an answering subscriber may see a prompt about `uid`'s + /// process. + #[cfg(test)] + pub(crate) fn has_answering_ui(&self, uid: u32) -> bool { + self.inner.has_audience(Some(uid)) + } + /// Resolves a pending prompt with the user's verdict. Returns what the /// prompt remembered about its process, or `None` if the id is unknown or /// the prompt already resolved another way (e.g. it timed out first), in diff --git a/crates/cfc-tray/src/main.rs b/crates/cfc-tray/src/main.rs index 617c61d..fe68a6d 100644 --- a/crates/cfc-tray/src/main.rs +++ b/crates/cfc-tray/src/main.rs @@ -884,6 +884,9 @@ async fn submit_prompt_verdict( } Err(e) => { warn!("submitting verdict: {e}"); + if let Some(body) = model::verdict_refused_body(&e) { + notify_brief(body); + } *client = None; } } @@ -1059,6 +1062,8 @@ async fn run(sealed: std::io::Result<()>) -> anyhow::Result<()> { let mut notifier = PromptNotifier::new(handle_tx); let mut was_reachable: Option = None; let generic = !actions_supported; + // Set once the tray has said its binary was replaced under it. + let mut told_replaced = false; let mut ticker = tokio::time::interval(POLL_INTERVAL); ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Delay); @@ -1066,6 +1071,14 @@ async fn run(sealed: std::io::Result<()>) -> anyhow::Result<()> { tokio::select! { _ = ticker.tick() => { notifier.reclaim_expired(); + // After an upgrade the daemon refuses this process: say so + // instead of failing every answer and pause quietly. + if !told_replaced + && std::fs::read_link("/proc/self/exe").is_ok_and(|exe| model::replaced_on_disk(&exe)) + { + told_replaced = true; + notify_brief(model::REPLACED_BODY.into()); + } if !refresh(&handle, &mut client, &socket, &mut gate, &mut was_reachable, generic).await { break; } diff --git a/crates/cfc-tray/src/model.rs b/crates/cfc-tray/src/model.rs index c8028e7..ad8aedb 100644 --- a/crates/cfc-tray/src/model.rs +++ b/crates/cfc-tray/src/model.rs @@ -113,6 +113,27 @@ pub fn pause_failed_body(verb: &str, err: &ClientError) -> String { } } +/// Notification body for a prompt answer the daemon refused, `None` when the +/// failure was not a refusal (those are only logged). The daemon's reason +/// already says what to do, e.g. restart the tray after an upgrade. +pub fn verdict_refused_body(err: &ClientError) -> Option { + match err { + ClientError::Denied(reason) => Some(format!("Your answer was not accepted: {reason}")), + _ => None, + } +} + +/// Shown once when this tray's binary was replaced on disk while it ran. +pub const REPLACED_BODY: &str = "Colony Firewall was updated. Quit the tray from its menu and \ + start it again, and restart the app: until then the firewall refuses their answers and \ + changes."; + +/// Whether `/proc/self/exe` (as read) names a file that was replaced or +/// removed since this process started, which an upgrade does. +pub fn replaced_on_disk(exe: &std::path::Path) -> bool { + exe.as_os_str().as_encoded_bytes().ends_with(b" (deleted)") +} + /// "2h 05m" / "5m 00s" / "42s". Negative input clamps to "0s". pub fn format_countdown(secs: i64) -> String { let s = secs.max(0); @@ -709,6 +730,28 @@ mod tests { ); } + #[test] + fn a_refused_answer_says_why_and_other_failures_stay_quiet() { + let body = verdict_refused_body(&ClientError::Denied( + "read-only access: restart it after an upgrade".into(), + )) + .unwrap(); + assert!(body.contains("not accepted"), "{body}"); + assert!(body.contains("restart it after an upgrade"), "{body}"); + assert_eq!(verdict_refused_body(&ClientError::StreamClosed), None); + } + + #[test] + fn a_replaced_binary_is_noticed() { + use std::path::Path; + assert!(replaced_on_disk(Path::new( + "/usr/bin/colony-firewall-tray (deleted)" + ))); + assert!(!replaced_on_disk(Path::new( + "/usr/bin/colony-firewall-tray" + ))); + } + // --- menu model --------------------------------------------------------- #[test] diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 8333c66..a1c37b0 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -389,7 +389,10 @@ the entire attack surface. Two layers: Every other peer is read-only. Its prompt subscription is not counted in the router's census, so `no_ui_action` still applies when only such peers -listen, and it never enters a prompt's audience. Prompt ownership comes on +listen, and it never enters a prompt's audience. An answering subscription +is checked again before each prompt it is handed; one that fails (an app or +tray whose binary an upgrade replaced) ends with the reason, which drops it +from the census. Prompt ownership comes on top: the daemon records which answering subscriber uids actually received each prompt and refuses a verdict from anyone else, so one desktop session cannot answer another's. Root is exempt. diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 1130ce2..971a4c7 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -353,7 +353,10 @@ of it between two reads of that process's start time (so a reused pid fails): and that file is root-owned, unwritable by group and other, in root-owned directories nobody else can write. Re-checked every time, so after an upgrade a process still running the replaced binary is read-only until it - is restarted; + is restarted. Its prompt subscription is checked again before each prompt + and ends at the first one after the upgrade (which still waits out the + timeout), so it stops counting as a UI and later prompts take + `no_ui_action`; the tray says it must be restarted; - it holds the client end of this very connection itself, on a descriptor above stderr (found through sock_diag's `UNIX_DIAG`); - every executable file it mapped comes from such sealed directories, so an diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index dfc0029..86737e6 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -300,7 +300,11 @@ installed Colony Firewall app and tray, or from root (sudo cfc ...). (restart it after an upgrade)"** or **"the caller did not seal itself at startup"**: the app or tray was upgraded under you, or is a 0.7 build. Quit and start it again (the tray from your session's autostart or by - hand, `colony-firewall-tray &`). + hand, `colony-firewall-tray &`). The tray tells you once in a + notification when its binary was replaced, and shows the reason when an + answer is refused. Until it restarts, the first prompt after the upgrade + waits out `prompt_timeout_secs` and later ones take `no_ui_action`, unless + the app is open and current. - **"the caller loaded /home/…/something.so"**: a library from outside the root-owned system directories is mapped into the app, usually a global `LD_PRELOAD` (MangoHud, gamemode) or a user-installed Vulkan layer or diff --git a/packaging/rpm/colony-firewall-control.spec b/packaging/rpm/colony-firewall-control.spec index cde4bd3..fd8b390 100644 --- a/packaging/rpm/colony-firewall-control.spec +++ b/packaging/rpm/colony-firewall-control.spec @@ -189,6 +189,8 @@ if [ $1 -gt 1 ]; then echo "Firewall rules could not be refreshed; reload colony-firewall-nft and inspect the journal before relying on filtering." >&2 exit 1 } + # The daemon trusts only the binaries now installed. + echo "Restart Colony Firewall and its tray now: until then the daemon refuses their answers and changes." fi %preun diff --git a/pkg/colony-firewall-control.install b/pkg/colony-firewall-control.install index eb0421d..fc47e28 100644 --- a/pkg/colony-firewall-control.install +++ b/pkg/colony-firewall-control.install @@ -51,13 +51,15 @@ post_upgrade() { } if [ "$(vercmp "$2" 0.8.0)" -lt 0 ]; then cat <<'EOF' -==> 0.8.0: restart Colony Firewall and its tray now; until then they are - read-only (the daemon trusts only the installed binaries). Pause/resume - and rule import ask for an administrator password through polkit. - cfc without sudo is read-only: use sudo cfc for changes. +==> 0.8.0: pause/resume, rule import and Allow rules that name no program + ask for an administrator password through polkit. cfc without sudo is + read-only: use sudo cfc for changes. EOF fi cat <<'EOF' +==> Restart Colony Firewall and its tray now (quit the tray from its menu, + then start both again). The daemon trusts only the binaries now + installed, so until then it refuses their answers and changes. ==> colony-firewall-control upgraded. Changes: /usr/share/doc/colony-firewall-control/README.md and https://github.com/Project-Colony/Colony-Firewall-Control/blob/main/CHANGELOG.md diff --git a/pkg/colony.json b/pkg/colony.json index ed505f4..1dafbda 100644 --- a/pkg/colony.json +++ b/pkg/colony.json @@ -63,6 +63,6 @@ "optionalDepends": { "libnotify": "desktop notifications", "libx11": "run the GUI in X11/XWayland sessions", - "polkit": "administrator prompt when the app or tray pauses filtering or imports rules" + "polkit": "administrator prompt when the app or tray pauses filtering, imports rules or allows every program" } } From ecc3c80d0f02fa891e2023d9affe82fdeba03fc8 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:41:45 +0200 Subject: [PATCH 123/125] fix(rules): credit the program Allow that carves out a program Deny Under a held generic Allow, a matching program Allow returned the generic Allow, so the hit counter, live view and log named the wrong rule and the documented carve-out (`allow --exe X --dst-port 443` next to `deny --exe X`) stayed at zero hits, inviting its deletion. The program Allow now answers itself; the action is unchanged. Document and pin two intended consequences of the program-Deny override: a flow from an unknown program no longer passes a generic Allow while any program Deny could match it, and a --sha256-only Deny holds every unhashed image (over 64 MiB) open under every generic Allow. Both are prompted per D3; the docs name the remedies. --- CHANGELOG.md | 8 +++ crates/cfc-core/src/rule.rs | 101 +++++++++++++++++++++++++++++++----- docs/ARCHITECTURE.md | 4 +- docs/HARDENING.md | 5 +- docs/TROUBLESHOOTING.md | 11 ++++ 5 files changed, 114 insertions(+), 15 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index f57c403..0a551b1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -15,6 +15,8 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). refuses the agent there too, and the in-kernel connect hooks refuse it outright. Rules that name a program keep their specificity order among themselves, so `allow --exe X --dst-port 443` still beats `deny --exe X`. + Under a generic Allow, a matching program Allow is the rule that answers + and is credited with the hit, so the carve-out keeps a hit count. A `/0` network (`--dst-net 0.0.0.0/0`, `::/0`) no longer counts as a predicate when rules are ranked; it still limits a rule to one address family, and stored rules that carry only a `/0` keep loading. Some flows @@ -29,6 +31,12 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). and prompt caps send overflow there too; pause and the loopback allowance no longer let such flows through without asking. One "allow this program" rule no longer blocks every unattributed flow that a generic rule allows. + The reverse holds: once any program Deny or Reject exists, a flow whose + program is unknown no longer passes a generic Allow it matches (that Deny + may be about it), and no Allow rule can settle it; scope the Deny to + destinations or make attribution succeed. A `--sha256` Deny without + `--exe` does this to every image over 64 MiB (Chromium, Electron, VS + Code); add `--exe` to it. The event log and live feed record the undecided rule in `rule_id` with a source other than `rule`. A legacy hostname rule that cannot be decided is still refused. With `no_ui_action = "Allow"`, a program can reach that diff --git a/crates/cfc-core/src/rule.rs b/crates/cfc-core/src/rule.rs index ba7536c..fbdf681 100644 --- a/crates/cfc-core/src/rule.rs +++ b/crates/cfc-core/src/rule.rs @@ -683,15 +683,17 @@ impl RuleSet { } continue; } - winner = Some(match generic_allow { - // A lower program Allow cannot turn the answer into a refusal. - Some(held) if rule.action == Action::Allow => held, - None if rule.action == Action::Allow && !rule.scope.names_program() => { - generic_allow = Some(rule); - continue; - } - _ => rule, - }); + if generic_allow.is_none() + && rule.action == Action::Allow + && !rule.scope.names_program() + { + generic_allow = Some(rule); + continue; + } + // Under a held generic Allow this is a program rule. A program + // Deny overrides the Allow; a program Allow keeps the action but + // answers itself, since it is what stops a lower program Deny. + winner = Some(rule); break; } let Some(winner) = winner.or(generic_allow) else { @@ -1383,7 +1385,9 @@ mod tests { } #[test] - fn a_program_allow_below_a_generic_allow_leaves_it_the_answer() { + fn a_program_allow_below_a_generic_allow_answers_with_the_same_action() { + // The program Allow is credited (hit counter, live view, log): it is + // the rule that keeps a lower program Deny from winning. let set = sorted(vec![ scoped( "allow-x", @@ -1405,8 +1409,34 @@ mod tests { ]); assert_eq!( winner(&set, &mk_conn(), &mk_proc("/usr/bin/x")), - Some("allow-https") + Some("allow-x") ); + // The documented carve-out (docs/TROUBLESHOOTING.md): the program + // Allow under the generic Allow is what keeps `deny --exe` from + // winning on 443, so it is the rule credited there. + let carve_out = sorted(vec![ + set.rules + .iter() + .find(|r| r.name == "allow-https") + .unwrap() + .clone(), + scoped( + "allow-x-https", + Action::Allow, + RuleScope { + dst_port: Some(443), + ..exe("/usr/bin/x") + }, + ), + scoped("deny-x", Action::Deny, exe("/usr/bin/x")), + ]); + let http = Connection { + dst_port: 80, + ..mk_conn() + }; + let x = mk_proc("/usr/bin/x"); + assert_eq!(winner(&carve_out, &mk_conn(), &x), Some("allow-x-https")); + assert_eq!(winner(&carve_out, &http, &x), Some("deny-x")); } #[test] @@ -1569,7 +1599,7 @@ mod tests { )); let mut hashed = mk_proc("/usr/bin/y"); hashed.sha256 = Some("aa".repeat(32)); - assert_eq!(winner(&set, &mk_conn(), &hashed), Some("allow-net")); + assert_eq!(winner(&set, &mk_conn(), &hashed), Some("allow-h")); hashed.sha256 = Some("cc".repeat(32)); assert_eq!(winner(&set, &mk_conn(), &hashed), Some("deny-y")); } @@ -1608,6 +1638,51 @@ mod tests { ); } + #[test] + fn a_digest_only_deny_leaves_every_unhashed_image_open_under_a_generic_allow() { + // Documented in docs/HARDENING.md: an image over the hashing cap + // (Chromium, Electron) could be the denied one, so a `--sha256`-only + // Deny keeps every such flow open under every generic Allow. Adding + // `--exe` settles every other program. + let https = RuleScope { + protocol: Some(Protocol::Tcp), + dst_port: Some(443), + ..RuleScope::any() + }; + let digest = || Some("bb".repeat(32)); + let chromium = mk_proc("/usr/lib/chromium/chromium"); + let digest_only = sorted(vec![ + scoped("allow-https", Action::Allow, https.clone()), + scoped( + "deny-tool", + Action::Deny, + RuleScope { + exe_sha256: digest(), + ..RuleScope::any() + }, + ), + ]); + assert!(matches!( + digest_only.lookup(&mk_conn(), &chromium, now()), + Match::Undecidable(r) if r.name == "deny-tool" + )); + let with_exe = sorted(vec![ + scoped("allow-https", Action::Allow, https), + scoped( + "deny-tool", + Action::Deny, + RuleScope { + exe_sha256: digest(), + ..exe("/usr/bin/tool") + }, + ), + ]); + assert_eq!( + winner(&with_exe, &mk_conn(), &chromium), + Some("allow-https") + ); + } + #[test] fn lookup_is_independent_of_insertion_order_with_the_override() { // Four rules whose pairwise precedence is a cycle: program allow (3) @@ -1676,7 +1751,7 @@ mod tests { } assert_eq!( answers.into_iter().collect::>(), - [Some("generic-allow".to_owned())] + [Some("program-allow".to_owned())] ); // Without the program allow, the program deny overrides the generic diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index a1c37b0..46b328a 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -274,7 +274,9 @@ name a program keep their specificity order among themselves, so `allow --exe X --dst-port 443` still beats `deny --exe X`. Everything else keeps the order above. The scan holds the first matching generic Allow instead of returning it, and only a lower program rule can still change the -answer: a program Deny or Reject replaces it, a program Allow leaves it. The +answer: a program Deny or Reject replaces it, and a program Allow answers in +its place with the same action (it is credited with the hit, since it is what +keeps a lower program Deny from winning). The relation is not a total order (program Allow 3 > program Deny 2 > generic Allow 5 > generic Deny 4 > program Allow 3), so no sort key could express it. The in-kernel precompute (`Engine::process_wide_action` and diff --git a/docs/HARDENING.md b/docs/HARDENING.md index 971a4c7..2532967 100644 --- a/docs/HARDENING.md +++ b/docs/HARDENING.md @@ -200,7 +200,10 @@ usually past it. For such a program: carrying one, Allow or Deny, cannot be decided for the program. Wherever its other fields match and the rules below it would answer differently, the program's flows are prompted, naming that rule, and take - `no_ui_action` when no UI is connected. + `no_ui_action` when no UI is connected. A `--sha256` Deny without + `--exe` therefore holds every such image open under every generic Allow + (any of them could be the denied one); give it `--exe` too, so other + programs are decided by path. - On a root-sealed path (root-owned, with root-owned ancestors, as a package installs it) nothing else changes: "Allow always" saves a path-only rule. - On any other path (under a home directory, a user-writable `/opt` diff --git a/docs/TROUBLESHOOTING.md b/docs/TROUBLESHOOTING.md index 86737e6..508b08c 100644 --- a/docs/TROUBLESHOOTING.md +++ b/docs/TROUBLESHOOTING.md @@ -710,6 +710,17 @@ not be fully identified; your answer applies to that connection only. With no app, tray or `sudo cfc prompts` connected the flow takes `no_ui_action`, and the event log shows it with that rule's id and source `default`. +The usual rule named is a program Deny or Reject (`deny --exe X`, often one +an earlier "Deny always" answer created): since 0.8.0 it beats every +generic Allow, so a flow from an unknown program no longer passes a generic +Allow it matches while that Deny might be about it. Unattributed UDP under +`allow --uid 1000 --protocol udp --dst-port 53` is the common case. No Allow +rule can settle this. Either scope the named Deny to destinations +(`--dst-port`, `--dst-net`) so it cannot apply to these flows, or make +attribution succeed (below). A `--sha256` Deny without `--exe` does the same +to every image over 64 MiB (Chromium, Electron, VS Code), since any of them +could be the denied one: re-create it with `--exe` as well. + These prompts appear even while paused or for loopback flows, because a rule may be about them. If they are frequent, find out why attribution fails. Run the daemon with `--debug` (`systemctl edit colony-firewalld`, From ffb6a4f72440a27d146c3915cace71f24331ff64 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 12:45:36 +0200 Subject: [PATCH 124/125] fix(ebpf): clear the kernel refusal when the rules abstain for a process The resync sweeps kept a standing in-kernel DENY whenever an undecidable Deny was still reachable. Since incomplete identity is prompted on the packet path, that kernel entry refused connect() with EPERM for a flow the packet path would have asked about, so no prompt ever came (for example a deny --sha256 rule left after a deny --exe for the same program is deleted). Every kernel writer now asks Engine::denies_process_wide, which is true only when process_wide_action answers Deny or Reject; an abstention clears and the packet path decides. deny_still_possible_for is removed. --- CHANGELOG.md | 10 +- crates/cfc-daemon/src/decision.rs | 205 ++++++++------------------ crates/cfc-daemon/src/ebpf/enforce.rs | 93 +++--------- docs/ARCHITECTURE.md | 9 +- 4 files changed, 102 insertions(+), 215 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 0a551b1..1e4784f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -29,8 +29,14 @@ and [Semantic Versioning](https://semver.org/spec/v2.0.0.html). the app, tray and `cfc prompts` show the program as unknown. With no UI connected the flow takes `no_ui_action` (Deny on every shipped profile), and prompt caps send overflow there too; pause and the loopback allowance - no longer let such flows through without asking. One "allow this program" - rule no longer blocks every unattributed flow that a generic rule allows. + no longer let such flows through without asking. The in-kernel connect + hooks follow: a running program whose only remaining refusal depends on a + digest the kernel side does not read (a `deny --sha256` rule, once a + `deny --exe` for that program is deleted or expires) no longer keeps a + kernel refusal that failed its `connect()` with no prompt; the packet + path decides it. One "allow + this program" rule no longer blocks every unattributed flow that a + generic rule allows. The reverse holds: once any program Deny or Reject exists, a flow whose program is unknown no longer passes a generic Allow it matches (that Deny may be about it), and no Allow rule can settle it; scope the Deny to diff --git a/crates/cfc-daemon/src/decision.rs b/crates/cfc-daemon/src/decision.rs index ff57289..d611a36 100644 --- a/crates/cfc-daemon/src/decision.rs +++ b/crates/cfc-daemon/src/decision.rs @@ -270,73 +270,22 @@ impl Engine { None } - /// Whether some resolution of the rules this caller cannot decide would - /// still deny this process outright - the question that separates the two - /// meanings of `process_wide_action`'s `None`. + /// Whether the kernel should refuse every `connect()` of this process: + /// exactly when [`Self::process_wide_action`] answers Deny or Reject. /// - /// The in-kernel sweeps turn on this distinction. `None` covers two - /// opposite situations: an abstention (a hash-scoped rule the caller - /// cannot decide) and a rule that was simply *deleted* (nobody replaces a - /// deny with an explicit allow, they delete it - and keeping the entry - /// then means the kernel goes on refusing a program no rule denies). - /// - /// An earlier version of this answered "is any rule undecidable?", which - /// conflates a third case: when the only undecidable rule is an *allow*, - /// both resolutions of the ambiguity end without a deny (the hash - /// matches and the process is allowed, or it does not and no rule - /// speaks), yet the old answer kept a standing kernel DENY on the - /// strength of a rule that could never justify one. So this walks the - /// rules in precedence order, the same filters as `process_wide_action` - /// (the inbound skip included, so the two cannot disagree about which - /// rules are in play), and answers whether a deny is still *reachable*: - /// - /// * an undecidable deny that constrains no destination: reachable - the - /// matching resolution denies process-wide. Answer yes. - /// * an undecidable allow, or an undecidable rule the packet path would - /// own anyway (destination-scoped): the deny-reachable resolution is - /// the one where it does not match. Walk on. - /// * a decidable match ends the walk exactly as `process_wide_action` - /// does: its action (or the packet path, for a destination-scoped - /// rule) is the whole answer, and nothing below it can matter. - /// * a decidable Allow that names no program does not end it: a program - /// Deny below it still wins in `lookup`, so from there on only rules - /// that name a program are looked at. - pub fn deny_still_possible_for(&self, proc: &Process) -> bool { - let now_unix_ms = chrono::Utc::now().timestamp_millis(); - let rules = self.inner.rules.read(); - let mut saw_generic_allow = false; - for rule in rules - .rules - .iter() - .filter(|r| r.enabled && !r.is_expired(now_unix_ms)) - .filter(|r| r.scope.direction != Some(cfc_core::Direction::Inbound)) - { - if saw_generic_allow && !rule.scope.names_program() { - continue; - } - if rule.scope.undecidable_for(proc) { - if matches!( - rule.action, - cfc_core::Action::Deny | cfc_core::Action::Reject - ) && !rule.scope.constrains_destination() - { - return true; - } - continue; - } - if !rule.scope.matches_process(proc) { - continue; - } - if is_generic_allow(rule) { - saw_generic_allow = true; - continue; - } - return matches!( - rule.action, - cfc_core::Action::Deny | cfc_core::Action::Reject - ) && !rule.scope.constrains_destination(); - } - false + /// The one question every in-kernel writer asks (exec, both resync + /// sweeps, the exe table), so they cannot disagree. An abstention is not + /// a refusal: the packet path decides that flow, with the digest in hand + /// when it can hash the image, and prompts when the identity stays + /// incomplete. Keeping a kernel entry there instead (what the sweeps did + /// until 0.8.0) refused in silence, with EPERM at `connect()`, a flow the + /// packet path would have asked about, so the prompt never came. A rule + /// that was deleted reads the same way and clears too. + pub fn denies_process_wide(&self, proc: &Process) -> bool { + matches!( + self.process_wide_action(proc), + Some(cfc_core::Action::Deny | cfc_core::Action::Reject) + ) } /// How many rules are loaded, without copying any of them. @@ -635,74 +584,68 @@ mod tests { // never get to revise? #[test] - fn a_deleted_rule_is_distinguishable_from_an_abstention() { - // The two meanings of `None`, which the orphan sweep must not conflate. + fn the_kernel_refuses_only_what_process_wide_action_denies() { + // A hash-scoped deny the caller cannot decide abstains, and an + // abstention installs no kernel entry: the packet path decides it. let mut hashed = RuleScope::any(); hashed.exe_path = Some(PathBuf::from("/usr/bin/curl")); hashed.exe_sha256 = Some("aa".repeat(32)); let engine = engine_with(vec![Rule::new("h".to_string(), Action::Deny, hashed)]); - - let no_hash = Process { - exe: PathBuf::from("/usr/bin/curl"), - ..Process::unknown(1) - }; - // Abstention on a deny: the matching resolution refuses, so a - // standing kernel entry must survive. Keep. + let no_hash = proc("/usr/bin/curl"); assert_eq!(engine.process_wide_action(&no_hash), None); - assert!(engine.deny_still_possible_for(&no_hash)); + assert!(!engine.denies_process_wide(&no_hash)); - // The same rule pinned to a DIFFERENT binary is decidable for this - // process, so it must not block the sweep from clearing. - let other = Process { - exe: PathBuf::from("/usr/bin/wget"), - ..Process::unknown(1) - }; - assert_eq!(engine.process_wide_action(&other), None); - assert!(!engine.deny_still_possible_for(&other)); + // A decidable program deny does reach the kernel, Reject included. + for action in [Action::Deny, Action::Reject] { + let engine = engine_with(vec![exe_rule("d", "/usr/bin/curl", action)]); + assert!(engine.denies_process_wide(&no_hash), "{action:?}"); + } + // And an empty rule set (the deleted-rule case) refuses nothing. + assert!(!engine_with(vec![]).denies_process_wide(&no_hash)); + } - // And a rule set with nothing in it - the deleted-rule case - is the - // one the sweep exists for: None, no deny reachable, clear. - let empty = engine_with(vec![]); - assert_eq!(empty.process_wide_action(&no_hash), None); - assert!(!empty.deny_still_possible_for(&no_hash)); + #[test] + fn deleting_a_program_deny_lifts_the_kernel_refusal_under_a_digest_deny() { + // `deny --exe slack` plus a digest-only blocklist rule. The sweep + // reads no digest, so the blocklist rule is undecidable for slack. + let slack = "/usr/lib/slack/slack"; + let mut digest = RuleScope::any(); + digest.exe_sha256 = Some("bb".repeat(32)); + let deny_slack = exe_rule("deny-slack", slack, Action::Deny); + let mut blocklist = Rule::new("blocklist".to_string(), Action::Deny, digest); + // Imported later, so the program deny ranks first and answers. + blocklist.created_at = deny_slack.created_at + chrono::Duration::seconds(1); + let engine = engine_with(vec![deny_slack.clone(), blocklist.clone()]); + let p = proc(slack); + assert!(engine.denies_process_wide(&p)); + + // Once the program deny is gone, the kernel must stop refusing: + // the packet path asks about this flow instead, naming the rule it + // could not decide, and a standing kernel DENY would refuse the + // connect() before the queue ever saw it. + engine.remove_rule(deny_slack.id); + assert!(!engine.denies_process_wide(&p)); + assert!(matches!( + engine.peek(&conn(443), &p), + Decision::NeedsPrompt { undecided: Some(id), .. } if id == blocklist.id + )); } #[test] - fn an_undecidable_allow_cannot_prop_up_a_kernel_deny() { - // The case the old any-undecidable answer got backwards: the only - // rule naming this exe is a hash-scoped ALLOW. Whichever way the - // unknown hash resolves - it matches and the process is allowed, or - // it does not and no rule speaks - no deny is reachable, so a - // standing kernel DENY (from a deny rule since deleted) must clear. + fn an_undecidable_allow_over_a_deny_installs_no_kernel_refusal() { + // Whether the deny below fires depends on a digest the sweep does + // not read, so the packet path owns the flow (it hashes the image, + // or prompts when it cannot). let mut hashed_allow = RuleScope::any(); hashed_allow.exe_path = Some(PathBuf::from("/usr/bin/curl")); hashed_allow.exe_sha256 = Some("aa".repeat(32)); - let engine = engine_with(vec![Rule::new( - "pin".to_string(), - Action::Allow, - hashed_allow.clone(), - )]); - - let no_hash = Process { - exe: PathBuf::from("/usr/bin/curl"), - ..Process::unknown(1) - }; - assert_eq!(engine.process_wide_action(&no_hash), None); - assert!( - !engine.deny_still_possible_for(&no_hash), - "an allow-only abstention pinned a stale deny" - ); - - // But the same allow layered over a plain deny is the textbook - // reason to keep: if the hash does not match, the deny below fires. - let mut plain_deny = RuleScope::any(); - plain_deny.exe_path = Some(PathBuf::from("/usr/bin/curl")); let engine = engine_with(vec![ Rule::new("pin".to_string(), Action::Allow, hashed_allow), - Rule::new("deny".to_string(), Action::Deny, plain_deny), + exe_rule("deny", "/usr/bin/curl", Action::Deny), ]); + let no_hash = proc("/usr/bin/curl"); assert_eq!(engine.process_wide_action(&no_hash), None); - assert!(engine.deny_still_possible_for(&no_hash)); + assert!(!engine.denies_process_wide(&no_hash)); } #[test] @@ -959,7 +902,7 @@ mod tests { ]); let x = proc("/usr/bin/x"); assert_eq!(engine.process_wide_action(&x), Some(Action::Deny)); - assert!(engine.deny_still_possible_for(&x)); + assert!(engine.denies_process_wide(&x)); assert!(matches!( engine.peek(&conn(443), &x), Decision::Resolved(v) if v.action == Action::Deny @@ -967,27 +910,7 @@ mod tests { let y = proc("/usr/bin/y"); assert_eq!(engine.process_wide_action(&y), None); - assert!(!engine.deny_still_possible_for(&y)); - } - - #[test] - fn deny_still_possible_sees_a_program_deny_below_a_generic_allow() { - // The program Deny is pinned to a digest this process lacks: if it - // matches, it overrides the generic Allow ranked above it. - let mut hashed = RuleScope::any(); - hashed.exe_path = Some(PathBuf::from("/usr/bin/x")); - hashed.exe_sha256 = Some("aa".repeat(32)); - let mut generic = RuleScope::any(); - generic.uid = Some(1000); - generic.protocol = Some(Protocol::Tcp); - generic.dst_port = Some(443); - let engine = engine_with(vec![ - Rule::new("deny-x-pinned", Action::Deny, hashed), - Rule::new("allow-user-https", Action::Allow, generic), - ]); - let no_hash = proc("/usr/bin/x"); - assert_eq!(engine.process_wide_action(&no_hash), None); - assert!(engine.deny_still_possible_for(&no_hash)); + assert!(!engine.denies_process_wide(&y)); } #[test] @@ -1003,7 +926,7 @@ mod tests { ]); let p = proc("/usr/bin/curl"); assert_eq!(engine.process_wide_action(&p), None); - assert!(!engine.deny_still_possible_for(&p)); + assert!(!engine.denies_process_wide(&p)); } #[test] @@ -1014,7 +937,7 @@ mod tests { ]); let x = proc("/usr/bin/x"); assert_eq!(engine.process_wide_action(&x), None); - assert!(!engine.deny_still_possible_for(&x)); + assert!(!engine.denies_process_wide(&x)); } #[test] diff --git a/crates/cfc-daemon/src/ebpf/enforce.rs b/crates/cfc-daemon/src/ebpf/enforce.rs index 2341db6..026b0f3 100644 --- a/crates/cfc-daemon/src/ebpf/enforce.rs +++ b/crates/cfc-daemon/src/ebpf/enforce.rs @@ -62,7 +62,7 @@ use aya::maps::{HashMap as BpfHashMap, MapData, PerCpuArray}; use aya::programs::links::FdLink; use aya::programs::{CgroupAttachMode, CgroupSockAddr}; use aya::Ebpf; -use cfc_core::{Action, Process}; +use cfc_core::Process; use cfc_ebpf_common::{enforce_stat, ABI_VERSION}; use parking_lot::Mutex; use tracing::{debug, warn}; @@ -315,12 +315,7 @@ impl VerdictSink { // // Both the read and the decision happen out here, so the lock covers // map operations and nothing else. - enum DenyOp { - Deny, - Clear, - Keep, - } - let deny_work: Vec<(u32, DenyOp)> = views + let deny_work: Vec<(u32, bool)> = views .iter() .filter(|(pid, _, judged_at)| { // A pid with no start time now, or a different one, is not the @@ -328,34 +323,16 @@ impl VerdictSink { // process was already gone when it was read. judged_at.is_some() && proc_starttime(*pid) == *judged_at }) - .map(|(pid, as_process, _)| { - // The same three-way answer as the orphan branch below, - // because these are the same question at different ages. This - // loop used to collapse it to deny-or-clear, so a hash-scoped - // rule that made the engine abstain *cleared* a standing - // kernel deny for a recently-exec'd process while the orphan - // branch kept it for an old one - identical binary, identical - // rules, opposite enforcement, selected by exec age. - let op = match self.engine.process_wide_action(as_process) { - Some(Action::Deny | Action::Reject) => DenyOp::Deny, - Some(_) => DenyOp::Clear, - None if self.engine.deny_still_possible_for(as_process) => DenyOp::Keep, - None => DenyOp::Clear, - }; - (*pid, op) - }) + .map(|(pid, as_process, _)| (*pid, self.engine.denies_process_wide(as_process))) .collect(); let mut map = self.map.lock(); - for (pid, op) in &deny_work { - let pid = *pid; - let r = match op { - DenyOp::Deny => { - denied += 1; - map.insert(pid, cfc_ebpf_common::verdict::DENY, 0) - } - DenyOp::Clear => clear(&mut map, pid), - DenyOp::Keep => Ok(()), + for &(pid, deny) in &deny_work { + let r = if deny { + denied += 1; + map.insert(pid, cfc_ebpf_common::verdict::DENY, 0) + } else { + clear(&mut map, pid) }; if let Err(e) = r { warn!(pid, "verdict resync failed: {e}"); @@ -406,24 +383,11 @@ impl VerdictSink { doomed.push((pid, None)); continue; }; - // `None` is two opposite answers and they must not be conflated. - // An abstention that could still resolve to a refusal - a - // hash-scoped deny the sweep cannot decide - keeps the entry: - // clearing would lift a refusal nobody replaced. But "no rule - // matched at all" is the *deleted rule*, and that is the case - // this sweep exists for - nobody replaces a deny with an explicit - // allow, they delete it. Reading `None` as "keep" made the sweep - // fail at its one job whenever a rule was removed; reading every - // abstention as "keep" then pinned stale denies on the strength - // of allow rules that could never justify one. - match self.engine.process_wide_action(&proc) { - Some(Action::Deny | Action::Reject) => {} - Some(_) => doomed.push((pid, judged_at)), - None => { - if !self.engine.deny_still_possible_for(&proc) { - doomed.push((pid, judged_at)); - } - } + // Anything short of a process-wide refusal clears, an abstention + // included: the packet path decides that flow (and prompts when + // the identity stays incomplete), as it does after `on_exec`. + if !self.engine.denies_process_wide(&proc) { + doomed.push((pid, judged_at)); } } @@ -509,18 +473,11 @@ impl VerdictSink { _ => None, }; let as_process = corrected.as_ref().unwrap_or(proc); - let deny = matches!( - self.engine.process_wide_action(as_process), - Some(Action::Deny | Action::Reject) - ); + let deny = self.engine.denies_process_wide(as_process); let mut map = self.map.lock(); - // Two-way, not three-way like resync: an abstention here still - // clears. The difference is principled, not an oversight - any - // existing entry for this pid was written for the binary it just - // exec'd AWAY from, so there is no standing refusal for the current - // binary to preserve; keeping it would enforce the predecessor's - // verdict on its successor. The packet path decides the ambiguous - // case with the real hash in hand. + // An abstention clears, like in resync: any existing entry for this + // pid was written for the binary it just exec'd away from, and the + // packet path decides the ambiguous case with the real hash in hand. let r = if deny { map.insert(pid, cfc_ebpf_common::verdict::DENY, 0) } else { @@ -578,10 +535,7 @@ impl VerdictSink { exe: exe.clone(), ..Process::unknown(0) }; - if matches!( - self.engine.process_wide_action(&proc), - Some(Action::Deny | Action::Reject) - ) { + if self.engine.denies_process_wide(&proc) { let key = cfc_ebpf_common::hash_exe_path(exe.as_os_str().as_encoded_bytes()); wanted.insert(key, cfc_ebpf_common::verdict::DENY); } @@ -664,8 +618,8 @@ fn proc_exe(pid: u32) -> Option { /// This is *the* decider for both loops in `resync`. `matches_process` looks at /// three things - the executable path, its hash, and the uid - so a view built /// from the resolved path and the live uid is the whole decision surface; the -/// hash stays `None` here on purpose, which is what makes a hash-scoped rule -/// abstain and keeps the tri-state the sweep depends on. +/// hash stays `None` here on purpose: a hash-scoped rule then abstains, the +/// sweep clears, and the packet path, which reads the hash, decides. /// /// `None` means the process is gone or its /proc is unreadable, which callers /// must treat as "no answer" rather than falling back to a guess. @@ -1191,6 +1145,7 @@ pub(super) fn stats(map: &PerCpuArray<&MapData, u64>) -> anyhow::Result program Deny 2 > generic Allow 5 > generic Deny 4 > program Allow 3), so no sort key could express -it. The in-kernel precompute (`Engine::process_wide_action` and -`deny_still_possible_for`) walks the same way, so the connect hooks and the -packet path agree. +it. The in-kernel precompute (`Engine::process_wide_action`) walks the same +way, and every kernel writer (exec, both resync sweeps, the exe table) asks +`Engine::denies_process_wide`, so the connect hooks refuse a process only when +the packet path would refuse every flow of it. When the walk abstains (a rule +the hash-blind kernel side cannot decide), the kernel entry is cleared and +the packet path decides, prompting if the identity stays incomplete. **Incomplete identity is asked, not refused.** When the process's executable, uid or digest is unknown (an unattributed socket, a binary over From 1913d9f4d7b63dd59147ba049ca15e73f6732556 Mon Sep 17 00:00:00 2001 From: MotherSphere Date: Thu, 8 Oct 2026 13:06:01 +0200 Subject: [PATCH 125/125] test(daemon): wait for the child's exec to finish before reading its fds A vfork parent resumes inside execve, before the child's close-on-exec descriptors are closed, so the stdio test could still see our socket on a high descriptor of the child and fail on a fast runner. --- crates/cfc-daemon/src/official.rs | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/crates/cfc-daemon/src/official.rs b/crates/cfc-daemon/src/official.rs index 2cc59d5..0832424 100644 --- a/crates/cfc-daemon/src/official.rs +++ b/crates/cfc-daemon/src/official.rs @@ -531,6 +531,15 @@ mod tests { .unwrap(), ); let proc = PathBuf::from(format!("/proc/{}", child.0.id())); + // A vfork parent wakes inside execve before the child's close-on-exec + // descriptors are closed, so for a moment the child still holds our + // copy above stderr. comm is renamed after that close: wait for it. + for _ in 0..500 { + if std::fs::read_to_string(proc.join("comm")).is_ok_and(|c| c.trim() == "sleep") { + break; + } + std::thread::sleep(std::time::Duration::from_millis(10)); + } // Its fds are inspected by the test as the same user; the socket is // still ours, so ask about our end. let error = holds_connection(&proc, inode(ours.as_raw_fd())).unwrap_err();