diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index cfbe7af70..647e8d0be 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -1561,7 +1561,7 @@ checksum = "7fcf8323ef1faaee30a44a340193b1ac6814fd9b7b4e88e9d4519a3e4abe1cfd" [[package]] name = "smoltcp" version = "0.12.0" -source = "git+https://github.com/DragonOS-Community/smoltcp?rev=9805dae47858c7247303d9d6d19ca504632c1e4a#9805dae47858c7247303d9d6d19ca504632c1e4a" +source = "git+https://github.com/DragonOS-Community/smoltcp?rev=afde7359455a5b9feacd2b8684994232acbf8903#afde7359455a5b9feacd2b8684994232acbf8903" dependencies = [ "bitflags 1.3.2", "byteorder", diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index 19848e32d..21bf0ea11 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -62,7 +62,7 @@ linkme = "=0.3.27" num = { version = "=0.4.0", default-features = false } num-derive = "=0.3" num-traits = { git = "https://git.mirrors.dragonos.org.cn/DragonOS-Community/num-traits.git", rev = "1597c1c", default-features = false } -smoltcp = { version = "=0.12.0", git = "https://github.com/DragonOS-Community/smoltcp", rev = "9805dae47858c7247303d9d6d19ca504632c1e4a", default-features = false, features = [ +smoltcp = { version = "=0.12.0", git = "https://github.com/DragonOS-Community/smoltcp", rev = "afde7359455a5b9feacd2b8684994232acbf8903", default-features = false, features = [ "alloc", "medium-ethernet", "socket-raw", diff --git a/kernel/src/driver/net/deferred_index.rs b/kernel/src/driver/net/deferred_index.rs index 86d589b72..3f150e388 100644 --- a/kernel/src/driver/net/deferred_index.rs +++ b/kernel/src/driver/net/deferred_index.rs @@ -2,15 +2,16 @@ use alloc::vec::Vec; use system_error::SystemError; pub(super) type NodeId = usize; +pub(super) type RouteIndexKey = [u8; 21]; #[derive(Clone, Copy, Debug)] enum Node { Free { next: Option }, - Leaf { key: u64, slot: usize }, + Leaf { key: RouteIndexKey, slot: usize }, Branch { bit: u8, children: [NodeId; 2] }, } -/// A fallibly allocated Patricia index for attacker-controlled 64-bit keys. +/// A fallibly allocated Patricia index for complete (family, ifindex, address) keys. /// /// Branch bits strictly increase from root to leaf, bounding every lookup by /// the key width without depending on secret hash entropy. Node IDs remain @@ -29,7 +30,7 @@ impl DeferredRouteIndex { self.leaves } - pub(super) fn get(&self, key: u64) -> Option<(NodeId, usize)> { + pub(super) fn get(&self, key: RouteIndexKey) -> Option<(NodeId, usize)> { let leaf = self.find_leaf(key)?; match self.nodes[leaf] { Node::Leaf { @@ -48,7 +49,7 @@ impl DeferredRouteIndex { } /// Inserts a key after `try_reserve_insert`; this method cannot allocate. - pub(super) fn insert_prepared(&mut self, key: u64, slot: usize) -> NodeId { + pub(super) fn insert_prepared(&mut self, key: RouteIndexKey, slot: usize) -> NodeId { debug_assert!(self.get(key).is_none()); let Some(root) = self.root else { let leaf = self.alloc_prepared(Node::Leaf { key, slot }); @@ -64,8 +65,14 @@ impl DeferredRouteIndex { Node::Leaf { key, .. } => key, _ => unreachable!("Patricia traversal ends at a leaf"), }; - let differing_bit = (key ^ existing_key).leading_zeros() as u8; - debug_assert!(differing_bit < 64); + let differing_byte = key + .iter() + .zip(existing_key) + .position(|(a, b)| *a != b) + .expect("inserted Patricia keys are distinct"); + let differing_bit = (differing_byte * 8 + + (key[differing_byte] ^ existing_key[differing_byte]).leading_zeros() as usize) + as u8; let mut parent = None; let mut current = root; @@ -95,7 +102,7 @@ impl DeferredRouteIndex { leaf } - pub(super) fn set_slot(&mut self, leaf: NodeId, key: u64, slot: usize) { + pub(super) fn set_slot(&mut self, leaf: NodeId, key: RouteIndexKey, slot: usize) { match &mut self.nodes[leaf] { Node::Leaf { key: leaf_key, @@ -108,7 +115,7 @@ impl DeferredRouteIndex { } } - pub(super) fn remove(&mut self, key: u64) -> Option { + pub(super) fn remove(&mut self, key: RouteIndexKey) -> Option { let root = self.root?; if let Node::Leaf { key: leaf_key, @@ -153,7 +160,7 @@ impl DeferredRouteIndex { Some(slot) } - fn find_leaf(&self, key: u64) -> Option { + fn find_leaf(&self, key: RouteIndexKey) -> Option { let mut current = self.root?; loop { match self.nodes[current] { @@ -166,8 +173,8 @@ impl DeferredRouteIndex { } } - fn direction(key: u64, bit: u8) -> usize { - ((key >> (63 - bit)) & 1) as usize + fn direction(key: RouteIndexKey, bit: u8) -> usize { + ((key[bit as usize / 8] >> (7 - bit % 8)) & 1) as usize } fn branch_children(&self, node: NodeId) -> [NodeId; 2] { @@ -209,3 +216,46 @@ impl DeferredRouteIndex { self.free_count += 1; } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_key_bit_survives_insert_remove_and_slot_updates() { + let mut index = DeferredRouteIndex::default(); + // Include the all-zero key and every differing bit, especially the + // high IPv6 bytes which the old 64-bit key could not represent. + let mut keys = alloc::vec![[0; 21]]; + for bit in 0..168 { + let mut key = [0; 21]; + key[bit / 8] = 1 << (7 - bit % 8); + keys.push(key); + } + for (slot, key) in keys.iter().copied().enumerate() { + index.try_reserve_insert().unwrap(); + index.insert_prepared(key, slot); + } + assert_eq!(index.len(), keys.len()); + for (slot, key) in keys.iter().copied().enumerate() { + let (leaf, actual) = index.get(key).unwrap(); + assert_eq!(actual, slot); + index.set_slot(leaf, key, slot + 1000); + } + // Alternating removals exercise root replacement and stable leaf IDs. + for parity in 0..2 { + for slot in (parity..keys.len()).step_by(2) { + assert_eq!(index.remove(keys[slot]), Some(slot + 1000)); + assert!(index.get(keys[slot]).is_none()); + } + } + assert_eq!(index.len(), 0); + for (slot, key) in keys.iter().copied().enumerate().rev() { + index.try_reserve_insert().unwrap(); + index.insert_prepared(key, slot); + } + for (slot, key) in keys.iter().copied().enumerate() { + assert_eq!(index.get(key).unwrap().1, slot); + } + } +} diff --git a/kernel/src/driver/net/deferred_queue.rs b/kernel/src/driver/net/deferred_queue.rs index f84dd4eba..1ecc7074d 100644 --- a/kernel/src/driver/net/deferred_queue.rs +++ b/kernel/src/driver/net/deferred_queue.rs @@ -1,5 +1,5 @@ use super::{ - deferred_index::{DeferredRouteIndex, NodeId}, + deferred_index::{DeferredRouteIndex, NodeId, RouteIndexKey}, local_queue::{LocalOutputDisposition, LocalOutputPacket}, }; use alloc::{collections::VecDeque, vec::Vec}; @@ -7,12 +7,58 @@ use alloc::{collections::VecDeque, vec::Vec}; #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub(super) struct DeferredRouteKey { pub(super) oif: u32, - pub(super) next_hop: smoltcp::wire::Ipv4Address, + pub(super) next_hop: smoltcp::wire::IpAddress, } impl DeferredRouteKey { - fn packed(self) -> u64 { - ((self.oif as u64) << 32) | u32::from_be_bytes(self.next_hop.octets()) as u64 + fn packed(self) -> RouteIndexKey { + let mut key = [0; 21]; + key[1..5].copy_from_slice(&self.oif.to_be_bytes()); + match self.next_hop { + smoltcp::wire::IpAddress::Ipv4(address) => { + key[0] = 4; + key[17..].copy_from_slice(&address.octets()); + } + smoltcp::wire::IpAddress::Ipv6(address) => { + key[0] = 6; + key[5..].copy_from_slice(&address.octets()); + } + } + key + } +} + +#[cfg(test)] +mod key_tests { + use super::*; + use smoltcp::wire::{IpAddress, Ipv4Address, Ipv6Address}; + + #[test] + fn family_device_and_all_address_bytes_identify_a_neighbor() { + let v4 = DeferredRouteKey { + oif: 2, + next_hop: IpAddress::Ipv4(Ipv4Address::new(192, 0, 2, 1)), + }; + let mut bytes = [0; 16]; + bytes[12..].copy_from_slice(&[192, 0, 2, 1]); + let v6 = DeferredRouteKey { + oif: 2, + next_hop: IpAddress::Ipv6(Ipv6Address::from(bytes)), + }; + assert_ne!(v4.packed(), v6.packed()); + assert_ne!(v6.packed(), DeferredRouteKey { oif: 3, ..v6 }.packed()); + for byte in 0..16 { + let mut different = bytes; + different[byte] ^= 0x80; + assert_ne!( + v6.packed(), + DeferredRouteKey { + next_hop: IpAddress::Ipv6(Ipv6Address::from(different)), + ..v6 + } + .packed() + ); + } } } @@ -43,7 +89,7 @@ pub(super) struct DeferredRouteLimits { /// Per-interface neighbor-resolution backlog. /// -/// The hash index gives direct access by `(oif, next_hop)`, while `heap` is an +/// The Patricia index gives bounded access by `(oif, next_hop)`, while `heap` is an /// indexed min-heap whose root is the next schedulable neighbor. In-flight /// buckets sort after every schedulable bucket and therefore never publish a /// stale retry deadline. Both structures are protected by the containing @@ -300,7 +346,7 @@ impl DeferredRouteQueue { /// compaction and heap rebuilding are each performed only once. pub(super) fn release_resolved( &mut self, - mut is_resolved: impl FnMut(smoltcp::wire::Ipv4Address) -> bool, + mut is_resolved: impl FnMut(smoltcp::wire::IpAddress) -> bool, ready: &mut VecDeque, ) { let mut removed_any = false; diff --git a/kernel/src/driver/net/iface.rs b/kernel/src/driver/net/iface.rs index 4b6fa3532..6539ccf7b 100644 --- a/kernel/src/driver/net/iface.rs +++ b/kernel/src/driver/net/iface.rs @@ -181,7 +181,7 @@ pub trait Iface: crate::driver::base::device::Device { self.raw_transmit(&frame) } - /// Sends an already-routed IPv4 packet through this interface. + /// Sends an already-routed IP packet through this interface. /// /// Ethernet devices share this implementation so FIB eligibility cannot /// drift from driver-specific forwarding hooks. smoltcp resolves the @@ -193,10 +193,9 @@ pub trait Iface: crate::driver::base::device::Device { next_hop: &smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result<(), RouteSendError> { - let smoltcp::wire::IpAddress::Ipv4(next_hop) = *next_hop else { - return Err(SystemError::EAFNOSUPPORT.into()); - }; - let frame_capacity = (14usize + ip_packet.len()).max(42); + // A short original packet may instead emit an ARP request or an + // IPv6 Neighbor Solicitation (Ethernet + IPv6 + NS with SLLAO). + let frame_capacity = (14usize + ip_packet.len()).max(86); let mut frame = Vec::new(); frame .try_reserve_exact(frame_capacity) @@ -204,24 +203,18 @@ pub trait Iface: crate::driver::base::device::Device { let frame_prepared = Cell::new(false); let permanent_neighbor = self .net_namespace() - .and_then(|netns| { - crate::net::neighbor::lookup( - &netns, - self.nic_id() as u32, - smoltcp::wire::IpAddress::Ipv4(next_hop), - ) - }) + .and_then(|netns| crate::net::neighbor::lookup(&netns, self.nic_id() as u32, *next_hop)) .map(smoltcp::wire::HardwareAddress::Ethernet); let dispatch = { let mut interface = self.smol_iface().lock(); - interface.dispatch_ipv4_packet( + interface.dispatch_ip_packet( crate::time::Instant::now().into(), PreparedFrameTxToken { frame: &mut frame, prepared: &frame_prepared, }, - next_hop, + *next_hop, permanent_neighbor, ip_packet, ) @@ -234,20 +227,20 @@ pub trait Iface: crate::driver::base::device::Device { } match dispatch { Ok(()) => Ok(()), - Err(smoltcp::iface::Ipv4PacketDispatchError::NeighborPending { retry_at }) => { + Err(smoltcp::iface::IpPacketDispatchError::NeighborPending { retry_at }) => { Err(RouteSendError::RetryAt { retry_at, probe_sent, }) } - Err(smoltcp::iface::Ipv4PacketDispatchError::NoRoute) => { + Err(smoltcp::iface::IpPacketDispatchError::NoRoute) => { Err(SystemError::ENETUNREACH.into()) } - Err(smoltcp::iface::Ipv4PacketDispatchError::Exhausted) => { + Err(smoltcp::iface::IpPacketDispatchError::Exhausted) => { Err(SystemError::EAGAIN_OR_EWOULDBLOCK.into()) } - Err(smoltcp::iface::Ipv4PacketDispatchError::Malformed) - | Err(smoltcp::iface::Ipv4PacketDispatchError::InvalidHardwareAddress) => { + Err(smoltcp::iface::IpPacketDispatchError::Malformed) + | Err(smoltcp::iface::IpPacketDispatchError::InvalidHardwareAddress) => { Err(SystemError::EINVAL.into()) } } @@ -261,9 +254,7 @@ pub trait Iface: crate::driver::base::device::Device { next_hop: &smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result<(), SystemError> { - let smoltcp::wire::IpAddress::Ipv4(next_hop) = *next_hop else { - return Err(SystemError::EAFNOSUPPORT); - }; + let next_hop = *next_hop; if ip_packet.len() > self.mtu() { return Err(SystemError::EMSGSIZE); } @@ -283,35 +274,34 @@ pub trait Iface: crate::driver::base::device::Device { return Ok(()); } let tx_generation = self.common().tx_completion_generation(); - let (retry_at, probe_sent) = - match self.route_and_send(&smoltcp::wire::IpAddress::Ipv4(next_hop), ip_packet) { - Ok(()) => return Ok(()), - Err(RouteSendError::RetryAt { - retry_at, - probe_sent, - }) => (retry_at, probe_sent), - Err(RouteSendError::Failed(SystemError::ENOBUFS)) - | Err(RouteSendError::Failed(SystemError::EAGAIN_OR_EWOULDBLOCK)) => { - let now: smoltcp::time::Instant = crate::time::Instant::now().into(); - let delay_us = self.common().next_local_output_tx_backoff_us(); - let retry_at = now + smoltcp::time::Duration::from_micros(delay_us); - let (packet, reservation) = self.common().prepare_routed_output( - self.nic_id() as u32, - next_hop, - ip_packet, - )?; - reservation.requeue_backpressured(packet, retry_at); - let retry_at = if self.common().release_tx_backpressure_after(tx_generation) { - now - } else { - retry_at - }; - self.common() - .schedule_local_output(retry_at, napi, scheduler_netns); - return Ok(()); - } - Err(RouteSendError::Failed(error)) => return Err(error), - }; + let (retry_at, probe_sent) = match self.route_and_send(&next_hop, ip_packet) { + Ok(()) => return Ok(()), + Err(RouteSendError::RetryAt { + retry_at, + probe_sent, + }) => (retry_at, probe_sent), + Err(RouteSendError::Failed(SystemError::ENOBUFS)) + | Err(RouteSendError::Failed(SystemError::EAGAIN_OR_EWOULDBLOCK)) => { + let now: smoltcp::time::Instant = crate::time::Instant::now().into(); + let delay_us = self.common().next_local_output_tx_backoff_us(); + let retry_at = now + smoltcp::time::Duration::from_micros(delay_us); + let (packet, reservation) = self.common().prepare_routed_output( + self.nic_id() as u32, + next_hop, + ip_packet, + )?; + reservation.requeue_backpressured(packet, retry_at); + let retry_at = if self.common().release_tx_backpressure_after(tx_generation) { + now + } else { + retry_at + }; + self.common() + .schedule_local_output(retry_at, napi, scheduler_netns); + return Ok(()); + } + Err(RouteSendError::Failed(error)) => return Err(error), + }; self.common().enqueue_routed_output( self.nic_id() as u32, @@ -337,11 +327,11 @@ pub trait Iface: crate::driver::base::device::Device { Ok(()) } - /// Hands a namespace-local IPv4 packet to this interface's protocol stack + /// Hands a namespace-local IP packet to this interface's protocol stack /// without emitting it on the link. The shared queue in `IfaceCommon` /// makes local delivery a protocol-stack capability rather than an /// optional device-driver feature. - fn inject_local_ipv4_packet( + fn inject_local_ip_packet( &self, ingress_ifindex: u32, source_mac: smoltcp::wire::EthernetAddress, diff --git a/kernel/src/driver/net/iface_common.rs b/kernel/src/driver/net/iface_common.rs index 71d52f0f9..a60469e28 100644 --- a/kernel/src/driver/net/iface_common.rs +++ b/kernel/src/driver/net/iface_common.rs @@ -194,7 +194,7 @@ impl IfaceCommon { pub(super) fn enqueue_routed_output( &self, oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_packet: &[u8], retry_at: smoltcp::time::Instant, probe_sent: bool, @@ -211,7 +211,7 @@ impl IfaceCommon { pub(super) fn prepare_routed_output( &self, oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result<(LocalOutputPacket, LocalOutputReservation<'_>), SystemError> { if ip_packet.len() > self.mtu.load(Ordering::Acquire) { @@ -253,7 +253,7 @@ impl IfaceCommon { pub(super) fn enqueue_existing_routed_output( &self, oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result, SystemError> { let pending = self @@ -381,12 +381,8 @@ impl IfaceCommon { } let ifindex = self.iface_id as u32; self.local_input_queue.release_resolved_outputs(|next_hop| { - configured.is_some_and(|neighbors| { - neighbors - .lookup(ifindex, smoltcp::wire::IpAddress::Ipv4(next_hop)) - .is_some() - }) || interface - .is_neighbor_resolved(timestamp, smoltcp::wire::IpAddress::Ipv4(next_hop)) + configured.is_some_and(|neighbors| neighbors.lookup(ifindex, next_hop).is_some()) + || interface.is_neighbor_resolved(timestamp, next_hop) }); } @@ -419,16 +415,15 @@ impl IfaceCommon { &self, netns: Option<&Arc>, needs_routed_poll: bool, - authoritative_ipv4_output: bool, - configured_ipv4_output: bool, + authoritative_output: bool, + configured_output: bool, ) -> PollModeRecheck { - if !authoritative_ipv4_output - && netns.is_some_and(|netns| netns.router().requires_authoritative_ipv4_output()) + if !authoritative_output + && netns.is_some_and(|netns| netns.router().requires_authoritative_output()) { PollModeRecheck::Authoritative } else if (!needs_routed_poll && self.needs_namespace_routing()) - || (!configured_ipv4_output - && netns.is_some_and(crate::net::neighbor::has_ipv4_entries)) + || (!configured_output && netns.is_some_and(crate::net::neighbor::has_ethernet_entries)) { PollModeRecheck::Routed } else { @@ -514,17 +509,17 @@ impl IfaceCommon { } let netns = self.net_namespace(); - let authoritative_ipv4_output = force_authoritative + let authoritative_output = force_authoritative || netns .as_ref() - .is_some_and(|netns| netns.router().requires_authoritative_ipv4_output()); - let configured_ipv4_output = netns + .is_some_and(|netns| netns.router().requires_authoritative_output()); + let configured_output = netns .as_ref() - .is_some_and(crate::net::neighbor::has_ipv4_entries); + .is_some_and(crate::net::neighbor::has_ethernet_entries); let needs_routed_poll = self.needs_namespace_routing() || scope == IfacePollScope::LocalOnly - || authoritative_ipv4_output - || configured_ipv4_output; + || authoritative_output + || configured_output; let router = if needs_routed_poll { netns.as_ref().map(|netns| netns.router()) } else { @@ -549,7 +544,7 @@ impl IfaceCommon { continue; } let configured_neighbors = netns.as_ref().and_then(|netns| { - crate::net::neighbor::has_ipv4_entries(netns) + crate::net::neighbor::has_ethernet_entries(netns) .then(|| crate::net::neighbor::read(netns)) }); @@ -559,8 +554,8 @@ impl IfaceCommon { let restart = self.recheck_poll_mode( netns.as_ref(), needs_routed_poll, - authoritative_ipv4_output, - configured_ipv4_output, + authoritative_output, + configured_output, ); if restart != PollModeRecheck::Current { drop(configured_neighbors); @@ -576,7 +571,7 @@ impl IfaceCommon { configured_neighbors: configured_neighbors.as_ref(), owner_ifindex: self.iface_id as u32, owner_is_up, - authoritative_ipv4_output, + authoritative_output, }); let (has_events, poll_again, deadline_rearm) = { @@ -701,17 +696,17 @@ impl IfaceCommon { } let netns = self.net_namespace(); - let authoritative_ipv4_output = force_authoritative + let authoritative_output = force_authoritative || netns .as_ref() - .is_some_and(|netns| netns.router().requires_authoritative_ipv4_output()); - let configured_ipv4_output = netns + .is_some_and(|netns| netns.router().requires_authoritative_output()); + let configured_output = netns .as_ref() - .is_some_and(crate::net::neighbor::has_ipv4_entries); + .is_some_and(crate::net::neighbor::has_ethernet_entries); let needs_routed_poll = self.needs_namespace_routing() || scope == IfacePollScope::LocalOnly - || authoritative_ipv4_output - || configured_ipv4_output; + || authoritative_output + || configured_output; let router = if needs_routed_poll { netns.as_ref().map(|netns| netns.router()) } else { @@ -736,15 +731,15 @@ impl IfaceCommon { continue; } let configured_neighbors = netns.as_ref().and_then(|netns| { - crate::net::neighbor::has_ipv4_entries(netns) + crate::net::neighbor::has_ethernet_entries(netns) .then(|| crate::net::neighbor::read(netns)) }); let restart = self.recheck_poll_mode( netns.as_ref(), needs_routed_poll, - authoritative_ipv4_output, - configured_ipv4_output, + authoritative_output, + configured_output, ); if restart != PollModeRecheck::Current { drop(configured_neighbors); @@ -760,7 +755,7 @@ impl IfaceCommon { configured_neighbors: configured_neighbors.as_ref(), owner_ifindex: self.iface_id as u32, owner_is_up, - authoritative_ipv4_output, + authoritative_output, }); let mut processed = 0usize; @@ -1417,7 +1412,7 @@ impl IfaceCommon { pub(crate) fn release_configured_neighbor( &self, ifindex: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ) -> bool { self.local_input_queue.release_neighbor(ifindex, next_hop) } @@ -1428,7 +1423,7 @@ impl IfaceCommon { pub(crate) fn configured_neighbor_committed( &self, ifindex: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ) { if self.release_configured_neighbor(ifindex, next_hop) { self.schedule_registered_local_output(crate::time::Instant::now().into()); diff --git a/kernel/src/driver/net/local_output.rs b/kernel/src/driver/net/local_output.rs index 8e87df476..c78171a70 100644 --- a/kernel/src/driver/net/local_output.rs +++ b/kernel/src/driver/net/local_output.rs @@ -64,9 +64,8 @@ pub(super) fn defer_native_output_after_tx_backpressure( /// A namespace-local view over the target interface's transport stack. /// Ingress retains the physical ifindex. Output is staged until the smoltcp -/// locks are released: IPv4 may then select another device through the -/// namespace FIB, while native same-interface and non-IPv4 traffic keeps the -/// underlying device path. +/// locks are released: unicast IP may then select another device through the +/// namespace FIB, while native link-local control traffic keeps the device path. pub(super) struct LocalInputDevice<'a, D: SmolDevice + ?Sized> { pub(super) device: &'a mut D, pub(super) common: &'a IfaceCommon, @@ -74,7 +73,7 @@ pub(super) struct LocalInputDevice<'a, D: SmolDevice + ?Sized> { } /// Delegates receive to the physical device while routing every response and -/// standalone IPv4 transmission through the same deferred output FIFO as +/// standalone routed IP transmission through the same deferred output FIFO as /// namespace-local input. pub(super) struct RoutedTxDevice<'a, D: SmolDevice + ?Sized> { pub(super) device: &'a mut D, @@ -125,17 +124,31 @@ pub(super) struct OutputBackendPolicy<'a> { pub(super) configured_neighbors: Option<&'a crate::net::neighbor::NeighborReadGuard<'a>>, pub(super) owner_ifindex: u32, pub(super) owner_is_up: bool, - pub(super) authoritative_ipv4_output: bool, + pub(super) authoritative_output: bool, } impl OutputBackendPolicy<'_> { + fn outbound_ip_mtu( + self, + destination: smoltcp::wire::IpAddress, + meta: PacketMeta, + native_mtu: usize, + ) -> usize { + match self.classify(destination.version(), destination, meta) { + OutputBackendDecision::Deferred(Some(route)) => route.ip_mtu.min(u16::MAX as usize), + // A missing route must not prevent TCP from advancing its timers. + // Actual packet dispatch still rejects the missing route. + _ => native_mtu, + } + } + pub(super) fn classify( self, version: smoltcp::wire::IpVersion, destination: smoltcp::wire::IpAddress, meta: PacketMeta, ) -> OutputBackendDecision { - if version != smoltcp::wire::IpVersion::Ipv4 { + if version == smoltcp::wire::IpVersion::Ipv6 && destination.is_multicast() { return OutputBackendDecision::NativeOwner; } let constrained_oif = (meta.id != 0).then_some(meta.id); @@ -149,7 +162,7 @@ impl OutputBackendPolicy<'_> { && !self.configured_neighbors.is_some_and(|neighbors| { neighbors.lookup(route.oif, route.next_hop).is_some() }) - && (!self.authoritative_ipv4_output + && (!self.authoritative_output || route.table != crate::net::route::RT_TABLE_DEFAULT) => { OutputBackendDecision::NativeOwner @@ -344,15 +357,7 @@ impl LocalInputTxToken<'_> { context: LocalOutputDisposition::local_context(route.oif), })); } - let smoltcp::wire::IpAddress::Ipv4(next_hop) = route.next_hop else { - self.medium = smoltcp::phy::Medium::Ip; - self.disposition = LocalOutputDisposition::Drop; - return Ok(Some(smoltcp::phy::TxEgressOverride { - medium: smoltcp::phy::Medium::Ip, - ip_mtu: self.owner_ip_mtu, - context: LocalOutputDisposition::DROP_CONTEXT, - })); - }; + let next_hop = route.next_hop; self.medium = smoltcp::phy::Medium::Ip; // The address owner and selected egress may have different MTUs. // Grow only the token that actually needs the larger route MTU; @@ -506,6 +511,11 @@ pub(super) fn local_tx_token<'a>( } impl SmolDevice for LocalInputDevice<'_, D> { + fn outbound_ip_mtu(&self, destination: smoltcp::wire::IpAddress, meta: PacketMeta) -> usize { + self.backend_policy + .outbound_ip_mtu(destination, meta, self.device.capabilities().ip_mtu()) + } + type RxToken<'a> = LocalInputRxToken where @@ -541,6 +551,11 @@ impl SmolDevice for LocalInputDevice<'_, D> { } impl SmolDevice for RoutedTxDevice<'_, D> { + fn outbound_ip_mtu(&self, destination: smoltcp::wire::IpAddress, meta: PacketMeta) -> usize { + self.backend_policy + .outbound_ip_mtu(destination, meta, self.device.capabilities().ip_mtu()) + } + type RxToken<'a> = RoutedRxToken> where @@ -623,7 +638,7 @@ pub(super) fn transmit_routed_stack_output( }; if packet.medium != smoltcp::phy::Medium::Ip || packet.frame.len() > ip_mtu - || packet.frame.first().map(|byte| byte >> 4) != Some(4) + || smoltcp::wire::IpVersion::of_packet(&packet.frame).ok() != Some(next_hop.version()) { return LocalOutputTransmitResult::Drop(packet, SystemError::EINVAL); } @@ -635,7 +650,7 @@ pub(super) fn transmit_routed_stack_output( }; return LocalOutputTransmitResult::Drop(packet, error); } - match iface.route_and_send(&smoltcp::wire::IpAddress::Ipv4(next_hop), &packet.frame) { + match iface.route_and_send(&next_hop, &packet.frame) { Ok(()) => LocalOutputTransmitResult::Sent(packet), Err(RouteSendError::RetryAt { retry_at, @@ -738,7 +753,7 @@ where LocalOutputDisposition::Local { oif, ip_mtu } => { if packet.medium != smoltcp::phy::Medium::Ip || packet.frame.len() > ip_mtu - || packet.frame.first().map(|byte| byte >> 4) != Some(4) + || smoltcp::wire::IpVersion::of_packet(&packet.frame).is_err() { return LocalOutputTransmitResult::Drop(packet, SystemError::EINVAL); } @@ -748,7 +763,7 @@ where if packet.frame.len() > iface.mtu() { return LocalOutputTransmitResult::Drop(packet, SystemError::EMSGSIZE); } - match iface.inject_local_ipv4_packet(oif, iface.mac(), &packet.frame, false) { + match iface.inject_local_ip_packet(oif, iface.mac(), &packet.frame, false) { Ok(()) => LocalOutputTransmitResult::Sent(packet), // This is receive-backlog congestion, not physical TX // backpressure. Linux may drop locally delivered packets when diff --git a/kernel/src/driver/net/local_queue.rs b/kernel/src/driver/net/local_queue.rs index 100039592..f90d6b8e1 100644 --- a/kernel/src/driver/net/local_queue.rs +++ b/kernel/src/driver/net/local_queue.rs @@ -34,6 +34,12 @@ impl LocalInputPacket { } pub(super) fn into_frame(self, medium: smoltcp::phy::Medium) -> Result, SystemError> { + let ethertype = match smoltcp::wire::IpVersion::of_packet(&self.ip_packet) + .map_err(|_| SystemError::EINVAL)? + { + smoltcp::wire::IpVersion::Ipv4 => [0x08, 0x00], + smoltcp::wire::IpVersion::Ipv6 => [0x86, 0xdd], + }; if medium == smoltcp::phy::Medium::Ip { return Ok(self.ip_packet); } @@ -46,7 +52,7 @@ impl LocalInputPacket { .map_err(|_| SystemError::ENOMEM)?; frame.extend_from_slice(&self.destination_mac.0); frame.extend_from_slice(&self.source_mac.0); - frame.extend_from_slice(&[0x08, 0x00]); + frame.extend_from_slice(ðertype); frame.extend_from_slice(&self.ip_packet); Ok(frame) } @@ -83,55 +89,127 @@ pub(super) enum LocalOutputDisposition { }, Routed { oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_mtu: usize, }, Drop, } impl LocalOutputDisposition { - pub(super) const DROP_CONTEXT: u64 = 0; - const LOCAL_CONTEXT: u64 = 1 << 63; - // Reserved opaque smoltcp fragment context. Valid ifindices use the - // positive i32 range, so neither routed nor local encodings can collide - // with the all-ones value. - pub(super) const NATIVE_CONTEXT: u64 = u64::MAX; - - pub(super) fn routed_context(oif: u32, next_hop: smoltcp::wire::Ipv4Address) -> u64 { + pub(super) const DROP_CONTEXT: [u64; 3] = [0; 3]; + pub(super) const NATIVE_CONTEXT: [u64; 3] = [1, 0, 0]; + const LOCAL_TAG: u64 = 2; + const IPV4_TAG: u64 = 4; + const IPV6_TAG: u64 = 6; + + pub(super) fn routed_context(oif: u32, next_hop: smoltcp::wire::IpAddress) -> [u64; 3] { debug_assert_ne!(oif, 0); - debug_assert_eq!(oif & (1 << 31), 0); - ((oif as u64) << 32) | u32::from_be_bytes(next_hop.octets()) as u64 + match next_hop { + smoltcp::wire::IpAddress::Ipv4(address) => [ + ((oif as u64) << 32) | Self::IPV4_TAG, + 0, + u32::from_be_bytes(address.octets()) as u64, + ], + smoltcp::wire::IpAddress::Ipv6(address) => { + let bytes = address.octets(); + [ + ((oif as u64) << 32) | Self::IPV6_TAG, + u64::from_be_bytes(bytes[..8].try_into().unwrap()), + u64::from_be_bytes(bytes[8..].try_into().unwrap()), + ] + } + } } - pub(super) fn local_context(oif: u32) -> u64 { + pub(super) fn local_context(oif: u32) -> [u64; 3] { debug_assert_ne!(oif, 0); - debug_assert_eq!(oif & (1 << 31), 0); - Self::LOCAL_CONTEXT | ((oif as u64) << 32) + [((oif as u64) << 32) | Self::LOCAL_TAG, 0, 0] } - pub(super) fn from_context(context: u64, ip_mtu: usize) -> Self { + pub(super) fn from_context(context: [u64; 3], ip_mtu: usize) -> Self { if context == Self::NATIVE_CONTEXT { return Self::NativeOwner; } - if context & Self::LOCAL_CONTEXT != 0 { - return Self::Local { - oif: ((context >> 32) as u32) & !(1 << 31), - ip_mtu, - }; - } - let oif = (context >> 32) as u32; + let oif = (context[0] >> 32) as u32; if oif == 0 { return Self::Drop; } - let octets = (context as u32).to_be_bytes(); + let next_hop = match context[0] & u32::MAX as u64 { + Self::LOCAL_TAG if context[1..] == [0, 0] => return Self::Local { oif, ip_mtu }, + Self::IPV4_TAG if context[1] == 0 && context[2] <= u32::MAX as u64 => { + smoltcp::wire::IpAddress::Ipv4(smoltcp::wire::Ipv4Address::from( + (context[2] as u32).to_be_bytes(), + )) + } + Self::IPV6_TAG => { + let mut bytes = [0; 16]; + bytes[..8].copy_from_slice(&context[1].to_be_bytes()); + bytes[8..].copy_from_slice(&context[2].to_be_bytes()); + smoltcp::wire::IpAddress::Ipv6(smoltcp::wire::Ipv6Address::from(bytes)) + } + _ => return Self::Drop, + }; Self::Routed { oif, - next_hop: smoltcp::wire::Ipv4Address::new(octets[0], octets[1], octets[2], octets[3]), + next_hop, ip_mtu, } } } +#[cfg(test)] +mod context_tests { + use super::*; + use smoltcp::wire::{IpAddress, Ipv4Address, Ipv6Address}; + + #[test] + fn routed_context_preserves_both_families_and_full_ipv6_address() { + let addresses = [ + IpAddress::Ipv4(Ipv4Address::new(192, 0, 2, 1)), + IpAddress::Ipv6(Ipv6Address::from([ + 0xfd, 0x12, 0x34, 0x56, 0x78, 0x9a, 0xbc, 0xde, 0xf0, 0x11, 0x22, 0x33, 0xc0, 0, 2, + 1, + ])), + ]; + for address in addresses { + for ifindex in [1, 42, i32::MAX as u32] { + let context = LocalOutputDisposition::routed_context(ifindex, address); + let LocalOutputDisposition::Routed { + oif, + next_hop, + ip_mtu, + } = LocalOutputDisposition::from_context(context, 1280) + else { + panic!("routed context changed disposition") + }; + assert_eq!((oif, next_hop, ip_mtu), (ifindex, address, 1280)); + } + } + assert!(matches!( + LocalOutputDisposition::from_context(LocalOutputDisposition::local_context(42), 1500), + LocalOutputDisposition::Local { + oif: 42, + ip_mtu: 1500 + } + )); + assert!(matches!( + LocalOutputDisposition::from_context(LocalOutputDisposition::NATIVE_CONTEXT, 1500), + LocalOutputDisposition::NativeOwner + )); + assert!(matches!( + LocalOutputDisposition::from_context(LocalOutputDisposition::DROP_CONTEXT, 1500), + LocalOutputDisposition::Drop + )); + // Unknown tags and malformed IPv4 payloads fail closed. + for context in [[(42 << 32) | 3, 0, 0], [(42 << 32) | 4, 1, 0], [6, 1, 2]] { + assert!(matches!( + LocalOutputDisposition::from_context(context, 1500), + LocalOutputDisposition::Drop + )); + } + } +} + #[derive(Debug, Default)] pub(super) struct LocalOutputQueueState { pub(super) packets: VecDeque, @@ -604,7 +682,7 @@ impl LocalInputQueue { pub(super) fn release_resolved_outputs( &self, - mut is_resolved: impl FnMut(smoltcp::wire::Ipv4Address) -> bool, + mut is_resolved: impl FnMut(smoltcp::wire::IpAddress) -> bool, ) { let mut output = self.output.lock(); let LocalOutputQueueState { @@ -615,7 +693,7 @@ impl LocalInputQueue { deferred_routes.release_resolved(&mut is_resolved, packets); } - pub(super) fn release_neighbor(&self, oif: u32, next_hop: smoltcp::wire::Ipv4Address) -> bool { + pub(super) fn release_neighbor(&self, oif: u32, next_hop: smoltcp::wire::IpAddress) -> bool { let mut output = self.output.lock(); let LocalOutputQueueState { packets, diff --git a/kernel/src/driver/net/loopback.rs b/kernel/src/driver/net/loopback.rs index aac4a1507..9d7633939 100644 --- a/kernel/src/driver/net/loopback.rs +++ b/kernel/src/driver/net/loopback.rs @@ -602,7 +602,7 @@ impl Iface for LoopbackInterface { if ip_packet.len() > self.mtu() { return Err(SystemError::EMSGSIZE.into()); } - self.inject_local_ipv4_packet(self.nic_id() as u32, self.mac(), ip_packet, false) + self.inject_local_ip_packet(self.nic_id() as u32, self.mac(), ip_packet, false) .map_err(Into::into) } diff --git a/kernel/src/net/neighbor/mod.rs b/kernel/src/net/neighbor/mod.rs index 9436f35f2..a8cd70116 100644 --- a/kernel/src/net/neighbor/mod.rs +++ b/kernel/src/net/neighbor/mod.rs @@ -114,8 +114,8 @@ pub(crate) fn read(netns: &Arc) -> NeighborReadGuard<'_> { netns.neighbor_table().read() } -pub(crate) fn has_ipv4_entries(netns: &Arc) -> bool { - netns.neighbor_table().has_ipv4_entries() +pub(crate) fn has_ethernet_entries(netns: &Arc) -> bool { + netns.neighbor_table().has_ethernet_entries() } /// Allocation-complete removal of all configured neighbors owned by one @@ -185,9 +185,7 @@ fn release_deferred_neighbor(iface: &Arc, entry: NeighborEntry) { if !entry.ethernet_output { return; } - let IpAddress::Ipv4(next_hop) = entry.destination else { - return; - }; + let next_hop = entry.destination; iface .common() .configured_neighbor_committed(entry.ifindex, next_hop); @@ -199,9 +197,9 @@ pub(crate) fn release_deferred_after_enqueue( netns: &Arc, common: &IfaceCommon, ifindex: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: IpAddress, ) -> bool { - if lookup(netns, ifindex, IpAddress::Ipv4(next_hop)).is_none() { + if lookup(netns, ifindex, next_hop).is_none() { return false; } common.release_configured_neighbor(ifindex, next_hop) diff --git a/kernel/src/net/neighbor/table.rs b/kernel/src/net/neighbor/table.rs index 31c5a8938..0e9dfd87d 100644 --- a/kernel/src/net/neighbor/table.rs +++ b/kernel/src/net/neighbor/table.rs @@ -26,9 +26,9 @@ use super::{ #[derive(Debug)] pub(crate) struct NeighborTable { entries: RwSem>, - /// Lock-free gate for the IPv4 egress path. The exact mapping remains in + /// Lock-free gate for configured Ethernet egress. The exact mapping remains in /// `entries`; this count only decides whether a poll must consult it. - ipv4_entries: AtomicUsize, + ethernet_entries: AtomicUsize, } /// Allocation-free view for packet paths that perform multiple lookups while @@ -92,7 +92,7 @@ impl NeighborTable { pub(crate) fn new() -> Self { Self { entries: RwSem::new(Vec::new()), - ipv4_entries: AtomicUsize::new(0), + ethernet_entries: AtomicUsize::new(0), } } @@ -157,11 +157,11 @@ impl NeighborTable { flags: update.flags, kind: RTN_UNICAST, }; - if entry.ethernet_output && matches!(entry.destination, IpAddress::Ipv4(_)) { + if entry.ethernet_output { // Publish the conservative slow-path gate before the // non-fallible insert, so a packet cannot miss a committed // configured mapping. - self.ipv4_entries.fetch_add(1, AtomicOrdering::Release); + self.ethernet_entries.fetch_add(1, AtomicOrdering::Release); } entries.insert(index, entry); Ok(NeighborMutationOutcome::Added(entry)) @@ -178,8 +178,8 @@ impl NeighborTable { let mut entries = self.entries.write(); let index = find(&entries, ifindex, destination).map_err(|_| SystemError::ENOENT)?; let removed = entries.remove(index); - if removed.ethernet_output && matches!(removed.destination, IpAddress::Ipv4(_)) { - self.ipv4_entries.fetch_sub(1, AtomicOrdering::Release); + if removed.ethernet_output { + self.ethernet_entries.fetch_sub(1, AtomicOrdering::Release); } Ok(removed) } @@ -208,8 +208,8 @@ impl NeighborTable { Ok(NeighborSnapshot { entries: snapshot }) } - pub(super) fn has_ipv4_entries(&self) -> bool { - self.ipv4_entries.load(AtomicOrdering::Acquire) != 0 + pub(super) fn has_ethernet_entries(&self) -> bool { + self.ethernet_entries.load(AtomicOrdering::Acquire) != 0 } pub(super) fn prepare_iface_purge( @@ -258,19 +258,17 @@ impl NeighborTable { entries: &mut Vec, range: core::ops::Range, ) { - let removed_ipv4 = entries[range.clone()] + let removed_ethernet = entries[range.clone()] .iter() - .filter(|entry| { - entry.ethernet_output && matches!(entry.destination, IpAddress::Ipv4(_)) - }) + .filter(|entry| entry.ethernet_output) .count(); entries.drain(range); - if removed_ipv4 != 0 { + if removed_ethernet != 0 { // Keep a stale-positive gate until the mappings are gone. A packet // may take the slow path unnecessarily, but can never miss a // configured mapping that is still published. - self.ipv4_entries - .fetch_sub(removed_ipv4, AtomicOrdering::Release); + self.ethernet_entries + .fetch_sub(removed_ethernet, AtomicOrdering::Release); } } } diff --git a/kernel/src/net/route/fib.rs b/kernel/src/net/route/fib.rs index 59b1e318b..08382d36b 100644 --- a/kernel/src/net/route/fib.rs +++ b/kernel/src/net/route/fib.rs @@ -8,7 +8,7 @@ use super::{ fib_index::{projection_key, AliasPlacement, BroadcastLookup, FibIndex, ProjectionKey}, is_ipv4, RouteDeleteSelector, RouteEntry, RouteLookupResult, RouteMutationOutcome, RouteNewFlags, RouteNotifications, RouteSourcePolicy, RT_SCOPE_LINK, RT_TABLE_DEFAULT, - RT_TABLE_LOCAL, + RT_TABLE_LOCAL, RT_TABLE_MAIN, }; #[derive(Debug, Default, PartialEq, Eq)] @@ -621,9 +621,11 @@ impl FibTable { /// A flat smoltcp LPM projection cannot preserve RPDB table priority when /// a more-specific default-table route overlaps a main-table route. Keep - /// such namespaces on the authoritative IPv4 egress path instead. - pub(in crate::net) fn requires_authoritative_ipv4_output(&self) -> bool { - self.index.has_ipv4_routes(RT_TABLE_DEFAULT) + /// such namespaces on the authoritative egress path instead. IPv6 also + /// needs that path when main-table routing is configured: a socketless + /// ICMPv6 reply may leave through a different interface from its owner. + pub(in crate::net) fn requires_authoritative_output(&self) -> bool { + self.index.has_ipv4_routes(RT_TABLE_DEFAULT) || self.index.has_ipv6_routes(RT_TABLE_MAIN) } pub(super) fn resolve_gateway( diff --git a/kernel/src/net/route/fib_index.rs b/kernel/src/net/route/fib_index.rs index b72c4d06d..d190ccbbe 100644 --- a/kernel/src/net/route/fib_index.rs +++ b/kernel/src/net/route/fib_index.rs @@ -325,6 +325,13 @@ impl FibIndex { }) } + pub(super) fn has_ipv6_routes(&self, table: u32) -> bool { + self.prefix_counts.contains_key(&PrefixDomain { + table, + family: AddressFamily::Ipv6, + }) + } + /// Reserve every bucket needed by a later infallible commit. pub(super) fn prepare_insert(&mut self, route: RouteEntry) -> Result<(), SystemError> { let result = (|| { diff --git a/kernel/src/net/route/mod.rs b/kernel/src/net/route/mod.rs index 25a0348df..9ce0183d3 100644 --- a/kernel/src/net/route/mod.rs +++ b/kernel/src/net/route/mod.rs @@ -29,6 +29,7 @@ pub(crate) use lifecycle::{ commit_addresses, prepare_address_link_change, prepare_link_state_change, register_iface, unregister_iface, AddressLinkChange, PreparedAddressRouteCommit, PreparedLinkStateChange, }; +pub(crate) use source::resolve_ipv6_output_route; pub(crate) use source::{resolve_ipv4_output_flow, resolve_ipv4_route, Ipv4OutputFlow}; use transaction::{ prepare_with_devices, projection_for_iface, transact_single, transact_with_devices, diff --git a/kernel/src/net/route/source.rs b/kernel/src/net/route/source.rs index 6ecbce9ae..4bc4d95af 100644 --- a/kernel/src/net/route/source.rs +++ b/kernel/src/net/route/source.rs @@ -1,4 +1,4 @@ -//! IPv4 output-route and source-address resolution. +//! Output-route and source-address validation for socket and routing callers. //! //! Socket and rtnetlink callers consume this module's result instead of //! independently interpreting FIB source policy. @@ -132,3 +132,45 @@ pub(crate) fn resolve_ipv4_output_flow( source: resolved.source, }) } + +/// Validate an already selected native IPv6 source against the output route. +/// Source selection itself remains in the existing socket source policy. +/// In particular, IPv6 never uses IPv4's explicit-device on-link fallback. +pub(crate) fn resolve_ipv6_output_route( + netns: &Arc, + destination: IpAddress, + required_oif: Option, + fixed_source: Option, +) -> Result { + if !matches!(destination, IpAddress::Ipv6(_)) { + return Err(SystemError::EAFNOSUPPORT); + } + let devices = netns.device_list(); + let router = netns.router(); + let routes = super::lock_output_routes(&router, devices); + let route = routes + .lookup(destination, required_oif) + .ok_or(SystemError::ENETUNREACH)?; + let iface = routes + .devices + .get(&(route.oif as usize)) + .ok_or(SystemError::ENETUNREACH)?; + if route.kind != RTN_LOCAL && !iface.flags().contains(InterfaceFlags::UP) { + return Err(SystemError::ENETDOWN); + } + if let Some(source) = fixed_source { + if !matches!(source, IpAddress::Ipv6(_)) + || !routes.devices.values().any(|candidate| { + crate::net::address::iface_accepts_local_address(candidate, source) + }) + { + return Err(SystemError::EADDRNOTAVAIL); + } + if super::is_ipv6_link_local(source) + && !crate::net::address::iface_accepts_local_address(iface, source) + { + return Err(SystemError::EADDRNOTAVAIL); + } + } + Ok(route) +} diff --git a/kernel/src/net/routing/mod.rs b/kernel/src/net/routing/mod.rs index 21c68b251..90c338b38 100644 --- a/kernel/src/net/routing/mod.rs +++ b/kernel/src/net/routing/mod.rs @@ -40,11 +40,11 @@ pub struct Router { pub(in crate::net) fib: RwSem, pub(self) nat_tracker: Arc, pub ns: RwSem>, - /// Execution mode for IPv4 output. It is derived from the committed FIB + /// Execution mode for namespace-routed output. It is derived from the committed FIB /// at rest and forced on while a writer publishes the FIB's per-interface /// smoltcp projections, making the complete control-plane transaction the /// only observable generation. - authoritative_ipv4_output: AtomicBool, + authoritative_output: AtomicBool, } /// Write access to the namespace FIB. @@ -55,7 +55,7 @@ pub struct Router { /// having to remember a separate generation handoff. pub(in crate::net) struct RouterFibWriteGuard<'a> { fib: RwSemWriteGuard<'a, crate::net::route::FibTable>, - authoritative_ipv4_output: &'a AtomicBool, + authoritative_output: &'a AtomicBool, } impl core::ops::Deref for RouterFibWriteGuard<'_> { @@ -74,10 +74,8 @@ impl core::ops::DerefMut for RouterFibWriteGuard<'_> { impl Drop for RouterFibWriteGuard<'_> { fn drop(&mut self) { - self.authoritative_ipv4_output.store( - self.fib.requires_authoritative_ipv4_output(), - Ordering::Release, - ); + self.authoritative_output + .store(self.fib.requires_authoritative_output(), Ordering::Release); } } @@ -87,7 +85,7 @@ impl Router { fib: RwSem::new(crate::net::route::FibTable::default()), nat_tracker: Arc::new(ConnTracker::default()), ns: RwSem::new(Weak::default()), - authoritative_ipv4_output: AtomicBool::new(false), + authoritative_output: AtomicBool::new(false), }) } @@ -98,7 +96,7 @@ impl Router { fib: RwSem::new(crate::net::route::FibTable::default()), ns: RwSem::new(Weak::default()), nat_tracker: Arc::new(ConnTracker::default()), - authoritative_ipv4_output: AtomicBool::new(false), + authoritative_output: AtomicBool::new(false), }) } @@ -110,16 +108,15 @@ impl Router { // restarts. Consequently no data-plane reader can enter between two // per-interface projection updates. Drop restores the mode derived // from the newly committed FIB before releasing the write lock. - self.authoritative_ipv4_output - .store(true, Ordering::Release); + self.authoritative_output.store(true, Ordering::Release); RouterFibWriteGuard { fib, - authoritative_ipv4_output: &self.authoritative_ipv4_output, + authoritative_output: &self.authoritative_output, } } - pub(crate) fn requires_authoritative_ipv4_output(&self) -> bool { - self.authoritative_ipv4_output.load(Ordering::Acquire) + pub(crate) fn requires_authoritative_output(&self) -> bool { + self.authoritative_output.load(Ordering::Acquire) } pub fn lookup_ingress_route( @@ -215,7 +212,7 @@ pub trait RouterEnableDevice: Iface { return Err(None); } interface - .inject_local_ipv4_packet( + .inject_local_ip_packet( self.nic_id() as u32, ether_frame.src_addr(), ipv4_packet_mut.as_ref(), @@ -229,7 +226,7 @@ pub trait RouterEnableDevice: Iface { return Err(None); } interface - .inject_local_ipv4_packet( + .inject_local_ip_packet( self.nic_id() as u32, ether_frame.src_addr(), ipv4_packet_mut.as_ref(), @@ -287,8 +284,40 @@ pub trait RouterEnableDevice: Iface { Err(None) } smoltcp::wire::EthernetProtocol::Ipv6 => { - log::warn!("IPv6 is not supported yet, ignoring packet"); - Err(None) + let packet = smoltcp::wire::Ipv6Packet::new_checked(ether_frame.payload()) + .map_err(|_| Some(SystemError::EINVAL))?; + let repr = smoltcp::wire::Ipv6Repr::parse(&packet) + .map_err(|_| Some(SystemError::EINVAL))?; + // Scoped destinations and neighbor discovery belong to the + // physical receiving interface, not the address owner. In + // particular a unicast NA must update that interface's cache. + if repr.dst_addr.is_multicast() + || repr.dst_addr.is_unicast_link_local() + || repr.dst_addr.is_loopback() + || repr.dst_addr.is_unspecified() + || ipv6_is_neighbor_discovery(&packet).map_err(Some)? + { + return Err(None); + } + let Some(IngressRouteDecision::Local(interface)) = self + .netns_router() + .lookup_ingress_route(repr.dst_addr.into(), self.nic_id() as u32) + else { + // This path implements host delivery only, not IPv6 + // forwarding. Let the physical stack reject other input. + return Err(None); + }; + if interface.nic_id() == self.nic_id() { + return Err(None); + } + interface + .inject_local_ip_packet( + self.nic_id() as u32, + ether_frame.src_addr(), + ðer_frame.payload()[..40 + repr.payload_len], + false, + ) + .map_err(Some) } _ => { log::warn!( @@ -444,6 +473,55 @@ pub trait RouterEnableDevice: Iface { } } +/// Recognize the IPv6 formats accepted by smoltcp without bypassing its +/// validation of NDP checksums, hop limit, or options. Hop-by-hop is the only +/// extension header currently dispatched by that stack. +fn ipv6_is_neighbor_discovery( + packet: &smoltcp::wire::Ipv6Packet<&[u8]>, +) -> Result { + use smoltcp::wire::{IpProtocol, Ipv6ExtHeader}; + let mut protocol = packet.next_header(); + let mut payload = packet.payload(); + if protocol == IpProtocol::HopByHop { + let header = Ipv6ExtHeader::new_checked(payload).map_err(|_| SystemError::EINVAL)?; + protocol = header.next_header(); + let header_len = (usize::from(header.header_len()) + 1) * 8; + payload = payload.get(header_len..).ok_or(SystemError::EINVAL)?; + } + Ok(protocol == IpProtocol::Icmpv6 + && payload + .first() + .is_some_and(|kind| (133..=137).contains(kind))) +} + +#[cfg(test)] +mod ipv6_input_tests { + use super::ipv6_is_neighbor_discovery; + use smoltcp::wire::Ipv6Packet; + + #[test] + fn neighbor_discovery_is_recognized_with_supported_extension_header() { + // The physical smoltcp interface still validates ICMPv6 and checksum. + let mut bytes = [0u8; 56]; + bytes[0] = 0x60; + bytes[4..6].copy_from_slice(&16u16.to_be_bytes()); + bytes[6] = 0; // Hop-by-hop. + bytes[40] = 58; // ICMPv6 follows the 8-byte extension. + bytes[48] = 136; // A unicast NA is still link-scoped protocol traffic. + assert!(ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).unwrap()); + bytes[48] = 129; // Ordinary echo replies must remain routable. + assert!( + !ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).unwrap() + ); + bytes[6] = 58; + bytes[40] = 135; // Direct NS, no extension. + assert!(ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).unwrap()); + bytes[6] = 0; + bytes[41] = 2; // Truncated extension cannot be indexed unchecked. + assert!(ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).is_err()); + } +} + fn sub_ttl_ipv4(ipv4_packet: &mut Ipv4Packet<&mut Vec>) { let new_ttl = ipv4_packet.hop_limit().saturating_sub(1); ipv4_packet.set_hop_limit(new_ttl); diff --git a/kernel/src/net/socket/inet/common/mod.rs b/kernel/src/net/socket/inet/common/mod.rs index 2d99f2fc6..deb009ee8 100644 --- a/kernel/src/net/socket/inet/common/mod.rs +++ b/kernel/src/net/socket/inet/common/mod.rs @@ -519,21 +519,18 @@ fn get_ephemeral_bind_target( /// Selects the SocketSet owner for a source chosen on `egress_iface`. /// -/// IPv4 local delivery follows the namespace local FIB and may therefore -/// target a different interface from egress. Native IPv6 output is still -/// interface-local, so IPv6 retains the egress stack until the routed-output -/// backend supports that family as well. +/// Local delivery follows the namespace local FIB and may therefore target +/// a different interface from egress, for either IP family. pub(crate) fn ephemeral_target_for_source( netns: &Arc, egress_iface: Arc, local_addr: smoltcp::wire::IpAddress, no_source_error: SystemError, ) -> Result { - let stack_owner = match local_addr { - smoltcp::wire::IpAddress::Ipv4(address) if !address.is_unspecified() => { - crate::net::route::local_address_owner(netns, local_addr).ok_or(no_source_error)? - } - _ => egress_iface, + let stack_owner = if !local_addr.is_unspecified() { + crate::net::route::local_address_owner(netns, local_addr).ok_or(no_source_error)? + } else { + egress_iface }; Ok(EphemeralBindTarget { stack_owner, diff --git a/kernel/src/net/socket/inet/datagram/mod.rs b/kernel/src/net/socket/inet/datagram/mod.rs index 4cc85f2e7..795e999ff 100644 --- a/kernel/src/net/socket/inet/datagram/mod.rs +++ b/kernel/src/net/socket/inet/datagram/mod.rs @@ -1026,6 +1026,28 @@ impl UdpSocket { Some((local, dest, connected_source)) => { let (required_oif, multicast_source) = output_flow::socket_constraints(self, dest.addr)?; + if matches!(dest.addr, smoltcp::wire::IpAddress::Ipv6(_)) + && !dest.addr.is_multicast() + { + let source = local + .addr + .filter(|source| !source.is_unspecified()) + .or(connected_source); + let route = crate::net::route::resolve_ipv6_output_route( + &self.netns, + dest.addr, + required_oif, + source, + )?; + // Native IPv6 source fragmentation is not implemented. + // Report the limit before enqueueing, rather than losing + // an accepted datagram at the physical egress. + if route.kind != crate::net::route::RTN_LOCAL + && buf.len() > route.ip_mtu.saturating_sub(40 + 8) + { + return Err(SystemError::EMSGSIZE); + } + } output_flow::resolve_ipv4_send_flow( &self.netns, local, diff --git a/kernel/src/net/socket/inet/stream/lifecycle.rs b/kernel/src/net/socket/inet/stream/lifecycle.rs index 219c19235..eb7becdfa 100644 --- a/kernel/src/net/socket/inet/stream/lifecycle.rs +++ b/kernel/src/net/socket/inet/stream/lifecycle.rs @@ -349,6 +349,17 @@ impl TcpSocket { // Explicit-source and self-connect paths need the same device/route // validation as implicit binding, before changing the socket state. if let Some(inner::Inner::Init(inner::Init::Bound((_, local, _)))) = writer.as_ref() { + if matches!(local.addr, smoltcp::wire::IpAddress::Ipv6(_)) + && !local.addr.is_unspecified() + { + let device = self.device_binding.resolve_iface(&self.netns)?; + crate::net::route::resolve_ipv6_output_route( + &self.netns, + remote_endpoint.addr, + device.map(|iface| iface.nic_id() as u32), + Some(local.addr), + )?; + } if local.addr.version() == smoltcp::wire::IpVersion::Ipv4 && !local.addr.is_unspecified() { diff --git a/user/apps/tests/dunitest/no_skip.txt b/user/apps/tests/dunitest/no_skip.txt index e73dd9e3f..bbb0fcede 100644 --- a/user/apps/tests/dunitest/no_skip.txt +++ b/user/apps/tests/dunitest/no_skip.txt @@ -26,6 +26,7 @@ normal/tcp_relisten normal/tcp_listener_overflow normal/tcp_accept_handshake normal/tcp_device_binding +normal/ipv6_routed_output normal/socket_nonblocking normal/poll_timeout_semantics diff --git a/user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc b/user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc new file mode 100644 index 000000000..89c862d5a --- /dev/null +++ b/user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc @@ -0,0 +1,430 @@ +// A synthetic peer observes physical ingress on veth1 for traffic emitted on +// veth2. The TCP source belongs to veth1: socket owner and egress must differ. +// No peer IP is assigned to the host, so the host cannot fake the handshake. +// Linux reference runs must disable veth TX checksum/GSO/GRO offloads first: +// ethtool -K veth{1,2} tx off rx off tso off gso off gro off (each device). +// Otherwise AF_PACKET exposes unfinished checksums and aggregated segments. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace { +class Fd { + public: + explicit Fd(int value = -1) : value_(value) {} + ~Fd() { reset(-1); } + Fd(const Fd&) = delete; + Fd& operator=(const Fd&) = delete; + int get() const { return value_; } + void reset(int value) { if (value_ >= 0) close(value_); value_ = value; } + private: + int value_; +}; + +uint16_t Get16(const unsigned char* p) { return (uint16_t(p[0]) << 8) | p[1]; } +uint32_t Get32(const unsigned char* p) { return (uint32_t(Get16(p)) << 16) | Get16(p + 2); } +void Put16(unsigned char* p, uint16_t n) { p[0] = n >> 8; p[1] = n; } +void Put32(unsigned char* p, uint32_t n) { Put16(p, n >> 16); Put16(p + 2, n); } +uint32_t Sum(const unsigned char* p, size_t n) { + uint32_t sum = 0; + for (size_t i = 0; i < n; i += 2) + sum += (uint16_t(p[i]) << 8) | (i + 1 < n ? p[i + 1] : 0); + return sum; +} +uint16_t Checksum(const unsigned char* ip, const unsigned char* tcp, size_t n, + uint8_t protocol = IPPROTO_TCP) { + uint32_t sum = Sum(ip + 8, 32) + n + protocol + Sum(tcp, n); + while (sum >> 16) sum = (sum & 0xffff) + (sum >> 16); + return ~sum; +} + +struct Request { + nlmsghdr header{}; + std::array body{}; + Request(uint16_t type, size_t size, uint16_t flags = 0) { + header.nlmsg_len = NLMSG_LENGTH(size); + header.nlmsg_type = type; + header.nlmsg_flags = NLM_F_REQUEST | NLM_F_ACK | flags; + } + void Attribute(uint16_t type, const void* data, size_t size) { + size_t offset = NLMSG_ALIGN(header.nlmsg_len); + ASSERT_LE(offset + RTA_SPACE(size), sizeof(*this)); + auto* attr = reinterpret_cast(reinterpret_cast(this) + offset); + attr->rta_type = type; + attr->rta_len = RTA_LENGTH(size); + memcpy(RTA_DATA(attr), data, size); + header.nlmsg_len = offset + RTA_SPACE(size); + } +}; + +class Ipv6RoutedOutput : public testing::Test { + protected: + Fd netlink_, packet_, client_; + unsigned int owner_ = 0, egress_ = 0, sequence_ = 0; + in6_addr source_{}, egress_address_{}, peer_{}; + std::array peer_mac_{}, egress_mac_{}; + bool source_added_ = false, egress_added_ = false, neighbor_added_ = false; + bool dynamic_neighbor_ = false, source_neighbor_ = false; + int original_mtu_ = 0; + unsigned int solicitations_ = 0; + uint16_t port_ = 0; + uint32_t client_next_ = 0; + static constexpr uint16_t kPeerPort = 19876; + static constexpr uint32_t kPeerSequence = 1234567; + + int Submit(Request& request) { + request.header.nlmsg_seq = ++sequence_; + sockaddr_nl kernel{}; + kernel.nl_family = AF_NETLINK; + if (sendto(netlink_.get(), &request, request.header.nlmsg_len, 0, + reinterpret_cast(&kernel), sizeof(kernel)) < 0) return -1; + for (;;) { + alignas(nlmsghdr) unsigned char reply[8192]; + ssize_t size = recv(netlink_.get(), reply, sizeof(reply), 0); + if (size < 0) { if (errno == EINTR) continue; return -1; } + for (auto* h = reinterpret_cast(reply); NLMSG_OK(h, size); + h = NLMSG_NEXT(h, size)) { + if (h->nlmsg_seq != sequence_ || h->nlmsg_type != NLMSG_ERROR) continue; + if (h->nlmsg_len < NLMSG_LENGTH(sizeof(nlmsgerr))) { errno = EPROTO; return -1; } + int error = reinterpret_cast(NLMSG_DATA(h))->error; + if (error) { errno = -error; return -1; } + return 0; + } + } + } + int Address(unsigned int index, const in6_addr& address, bool add) { + Request request(add ? RTM_NEWADDR : RTM_DELADDR, sizeof(ifaddrmsg), + add ? NLM_F_CREATE | NLM_F_EXCL : 0); + auto* msg = reinterpret_cast(NLMSG_DATA(&request.header)); + msg->ifa_family = AF_INET6; msg->ifa_prefixlen = 64; + msg->ifa_flags = IFA_F_NODAD; msg->ifa_index = index; + request.Attribute(IFA_ADDRESS, &address, sizeof(address)); + return Submit(request); + } + int Neighbor(bool add, const in6_addr* target = nullptr) { + Request request(add ? RTM_NEWNEIGH : RTM_DELNEIGH, sizeof(ndmsg), + add ? NLM_F_CREATE | NLM_F_EXCL : 0); + auto* msg = reinterpret_cast(NLMSG_DATA(&request.header)); + msg->ndm_family = AF_INET6; msg->ndm_ifindex = egress_; + msg->ndm_state = NUD_PERMANENT; msg->ndm_type = RTN_UNICAST; + request.Attribute(NDA_DST, target ? target : &peer_, sizeof(peer_)); + if (add) request.Attribute(NDA_LLADDR, peer_mac_.data(), peer_mac_.size()); + return Submit(request); + } + int SetMtu(int mtu) { + Request request(RTM_NEWLINK, sizeof(ifinfomsg)); + reinterpret_cast(NLMSG_DATA(&request.header))->ifi_index = egress_; + request.Attribute(IFLA_MTU, &mtu, sizeof(mtu)); + return Submit(request); + } + void SetUp() override { + owner_ = if_nametoindex("veth1"); egress_ = if_nametoindex("veth2"); + ASSERT_NE(owner_, 0u) << "requires the standard veth1/veth2 fixture"; + ASSERT_NE(egress_, 0u); + ASSERT_EQ(inet_pton(AF_INET6, "fd10:91::1", &source_), 1); + ASSERT_EQ(inet_pton(AF_INET6, "fd10:92::2", &egress_address_), 1); + ASSERT_EQ(inet_pton(AF_INET6, "fd10:92::99", &peer_), 1); + netlink_.reset(socket(AF_NETLINK, SOCK_RAW, NETLINK_ROUTE)); + ASSERT_GE(netlink_.get(), 0); + sockaddr_nl local{}; local.nl_family = AF_NETLINK; + ASSERT_EQ(bind(netlink_.get(), reinterpret_cast(&local), sizeof(local)), 0); + timeval timeout{5, 0}; + ASSERT_EQ(setsockopt(netlink_.get(), SOL_SOCKET, SO_RCVTIMEO, &timeout, sizeof(timeout)), 0); + ASSERT_EQ(Address(owner_, source_, true), 0) << strerror(errno); source_added_ = true; + ASSERT_EQ(Address(egress_, egress_address_, true), 0) << strerror(errno); egress_added_ = true; + packet_.reset(socket(AF_PACKET, SOCK_RAW | SOCK_NONBLOCK, htons(ETH_P_IPV6))); + ASSERT_GE(packet_.get(), 0); + ifreq request{}; + strcpy(request.ifr_name, "veth1"); + ASSERT_EQ(ioctl(packet_.get(), SIOCGIFHWADDR, &request), 0); + memcpy(peer_mac_.data(), request.ifr_hwaddr.sa_data, 6); + strcpy(request.ifr_name, "veth2"); + ASSERT_EQ(ioctl(packet_.get(), SIOCGIFHWADDR, &request), 0); + memcpy(egress_mac_.data(), request.ifr_hwaddr.sa_data, 6); + ASSERT_EQ(ioctl(packet_.get(), SIOCGIFMTU, &request), 0); + original_mtu_ = request.ifr_mtu; + sockaddr_ll link{}; link.sll_family = AF_PACKET; + link.sll_protocol = htons(ETH_P_IPV6); link.sll_ifindex = owner_; + ASSERT_EQ(bind(packet_.get(), reinterpret_cast(&link), sizeof(link)), 0); + ASSERT_EQ(Neighbor(true), 0) << strerror(errno); neighbor_added_ = true; + } + void TearDown() override { + client_.reset(-1); + if (neighbor_added_) { EXPECT_EQ(Neighbor(false), 0) << strerror(errno); } + if (dynamic_neighbor_) { + int result = Neighbor(false); + EXPECT_TRUE(result == 0 || errno == ENOENT) << strerror(errno); + } + if (source_neighbor_) { + int result = Neighbor(false, &source_); + EXPECT_TRUE(result == 0 || errno == ENOENT) << strerror(errno); + } + if (original_mtu_) { EXPECT_EQ(SetMtu(original_mtu_), 0) << strerror(errno); } + if (egress_added_) { EXPECT_EQ(Address(egress_, egress_address_, false), 0) << strerror(errno); } + if (source_added_) { EXPECT_EQ(Address(owner_, source_, false), 0) << strerror(errno); } + } + + void SendFrame(const std::vector& frame) { + sockaddr_ll link{}; link.sll_family = AF_PACKET; link.sll_ifindex = owner_; + link.sll_halen = 6; memcpy(link.sll_addr, egress_mac_.data(), 6); + ASSERT_EQ(sendto(packet_.get(), frame.data(), frame.size(), 0, + reinterpret_cast(&link), sizeof(link)), + static_cast(frame.size())) << strerror(errno); + } + void Ndisc(uint8_t type, const in6_addr& source, const in6_addr& destination, + const in6_addr& target) { + std::vector frame(14 + 40 + 32); + memcpy(frame.data(), egress_mac_.data(), 6); + memcpy(frame.data() + 6, peer_mac_.data(), 6); + Put16(frame.data() + 12, ETH_P_IPV6); + auto* ip = frame.data() + 14; + ip[0] = 0x60; Put16(ip + 4, 32); ip[6] = IPPROTO_ICMPV6; ip[7] = 255; + memcpy(ip + 8, &source, 16); memcpy(ip + 24, &destination, 16); + auto* icmp = ip + 40; + icmp[0] = type; icmp[4] = type == 136 ? 0x60 : 0; + memcpy(icmp + 8, &target, 16); + icmp[24] = type == 136 ? 2 : 1; icmp[25] = 1; + memcpy(icmp + 26, peer_mac_.data(), 6); + Put16(icmp + 2, Checksum(ip, icmp, 32, IPPROTO_ICMPV6)); + SendFrame(frame); + } + void Segment(uint8_t flags, uint32_t sequence, uint32_t ack, + const unsigned char* payload = nullptr, size_t length = 0) { + // Advertise a large MSS so the peer cannot accidentally mask egress MTU. + const size_t tcp_header = (flags & 2) ? 24 : 20; + std::vector frame(14 + 40 + tcp_header + length); + memcpy(frame.data(), egress_mac_.data(), 6); + memcpy(frame.data() + 6, peer_mac_.data(), 6); + Put16(frame.data() + 12, ETH_P_IPV6); + auto* ip = frame.data() + 14; + ip[0] = 0x60; Put16(ip + 4, tcp_header + length); ip[6] = IPPROTO_TCP; ip[7] = 64; + memcpy(ip + 8, &peer_, 16); memcpy(ip + 24, &source_, 16); + auto* tcp = ip + 40; + Put16(tcp, kPeerPort); Put16(tcp + 2, port_); + Put32(tcp + 4, sequence); Put32(tcp + 8, ack); + tcp[12] = (tcp_header / 4) << 4; tcp[13] = flags; Put16(tcp + 14, 65535); + if (flags & 2) { tcp[20] = 2; tcp[21] = 4; Put16(tcp + 22, 1460); } + if (length) memcpy(tcp + tcp_header, payload, length); + Put16(tcp + 16, Checksum(ip, tcp, tcp_header + length)); + SendFrame(frame); + } + bool Receive(std::vector& frame, int timeout_ms) { + auto deadline = std::chrono::steady_clock::now() + std::chrono::milliseconds(timeout_ms); + while (std::chrono::steady_clock::now() < deadline) { + pollfd event{packet_.get(), POLLIN, 0}; + if (poll(&event, 1, 50) < 0) { ADD_FAILURE() << strerror(errno); return false; } + std::array bytes{}; + sockaddr_ll link{}; socklen_t size = sizeof(link); + ssize_t n = recvfrom(packet_.get(), bytes.data(), bytes.size(), 0, + reinterpret_cast(&link), &size); + if (n < 74 || Get16(bytes.data() + 12) != ETH_P_IPV6) continue; + const auto* ip = bytes.data() + 14; + if (n >= 86 && ip[6] == IPPROTO_ICMPV6 && ip[40] == 135 && + !memcmp(ip + 48, &peer_, 16) && link.sll_pkttype != PACKET_OUTGOING) { + EXPECT_EQ(ip[7], 255); + in6_addr destination; + memcpy(&destination, ip + 8, 16); + Ndisc(136, peer_, destination, peer_); + ++solicitations_; + continue; + } + if (ip[6] != IPPROTO_TCP || memcmp(ip + 8, &source_, 16) || + memcmp(ip + 24, &peer_, 16)) continue; + const auto* tcp = ip + 40; + if (Get16(tcp) != port_ || Get16(tcp + 2) != kPeerPort) continue; + EXPECT_EQ(link.sll_pkttype, PACKET_HOST) << "must arrive from veth2, not owner output"; + if (link.sll_pkttype != PACKET_HOST) continue; + size_t payload_size = Get16(ip + 4); + if (payload_size < 20 || n < static_cast(54 + payload_size) || + (tcp[12] >> 4) * 4u < 20 || (tcp[12] >> 4) * 4u > payload_size) { + ADD_FAILURE() << "malformed TCP packet"; return false; + } + // RX checksum offload is not used by the synthetic peer. + EXPECT_EQ(Checksum(ip, tcp, payload_size), 0); + frame.assign(bytes.begin(), bytes.begin() + 54 + payload_size); + return true; + } + return false; + } + void Connect(bool bind_device, bool reuse_socket = false) { + sockaddr_in6 local{}; local.sin6_family = AF_INET6; local.sin6_addr = source_; + if (!reuse_socket) { + client_.reset(socket(AF_INET6, SOCK_STREAM | SOCK_NONBLOCK, 0)); + ASSERT_GE(client_.get(), 0); + ASSERT_EQ(bind(client_.get(), reinterpret_cast(&local), sizeof(local)), 0); + } + if (bind_device) { + const char device[] = "veth2"; + ASSERT_EQ(setsockopt(client_.get(), SOL_SOCKET, SO_BINDTODEVICE, device, sizeof(device)), 0); + } + sockaddr_in6 remote{}; remote.sin6_family = AF_INET6; + remote.sin6_addr = peer_; remote.sin6_port = htons(kPeerPort); + ASSERT_EQ(connect(client_.get(), reinterpret_cast(&remote), sizeof(remote)), -1); + ASSERT_EQ(errno, EINPROGRESS); + socklen_t size = sizeof(local); + ASSERT_EQ(getsockname(client_.get(), reinterpret_cast(&local), &size), 0); + port_ = ntohs(local.sin6_port); + std::vector frame; + ASSERT_TRUE(Receive(frame, 4000)) << "no routed SYN"; + ASSERT_EQ(frame[67] & 0x12, 2); + client_next_ = Get32(frame.data() + 58) + 1; + ASSERT_NO_FATAL_FAILURE(Segment(0x12, kPeerSequence, client_next_)); + pollfd event{client_.get(), POLLOUT, 0}; + ASSERT_EQ(poll(&event, 1, 4000), 1); + int error = -1; size = sizeof(error); + ASSERT_EQ(getsockopt(client_.get(), SOL_SOCKET, SO_ERROR, &error, &size), 0); + ASSERT_EQ(error, 0); + } + void Transfer(size_t length, size_t mtu) { + std::vector data(length); + for (size_t i = 0; i < length; ++i) data[i] = (i * 17 + 9) % 251; + ASSERT_EQ(send(client_.get(), data.data(), data.size(), MSG_NOSIGNAL), + static_cast(data.size())) << strerror(errno); + size_t received = 0; + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(8); + while (received < length && std::chrono::steady_clock::now() < deadline) { + std::vector frame; + ASSERT_TRUE(Receive(frame, 2000)) << "TCP data stalled at " << received; + ASSERT_LE(frame.size() - 14, mtu); + const auto* tcp = frame.data() + 54; + size_t header = (tcp[12] >> 4) * 4; + size_t count = frame.size() - 54 - header; + if (!count) continue; + uint32_t sequence = Get32(tcp + 4); + if (sequence == client_next_) { + ASSERT_LE(received + count, length); + EXPECT_EQ(memcmp(tcp + header, data.data() + received, count), 0); + received += count; client_next_ += count; + } + ASSERT_NO_FATAL_FAILURE(Segment(0x10, kPeerSequence + 1, client_next_)); + } + ASSERT_EQ(received, length); + // A reply arrives physically on veth2 but belongs to veth1's socket. + const unsigned char reply[] = "owner-handoff"; + ASSERT_NO_FATAL_FAILURE(Segment(0x18, kPeerSequence + 1, client_next_, reply, sizeof(reply))); + pollfd event{client_.get(), POLLIN, 0}; + ASSERT_EQ(poll(&event, 1, 4000), 1); + unsigned char buffer[sizeof(reply)]{}; + ASSERT_EQ(recv(client_.get(), buffer, sizeof(buffer), MSG_WAITALL), + static_cast(sizeof(reply))); + EXPECT_EQ(memcmp(buffer, reply, sizeof(reply)), 0); + ASSERT_NO_FATAL_FAILURE(Segment(0x14, kPeerSequence + 1 + sizeof(reply), client_next_)); + } +}; + +TEST_F(Ipv6RoutedOutput, ExplicitSourceUsesFibEgressAndReceivesOnOwner) { + ASSERT_NO_FATAL_FAILURE(Connect(false)); + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, BoundDeviceUsesEgressAndReceivesOnOwner) { + ASSERT_NO_FATAL_FAILURE(Connect(true)); + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, LargeTcpTransferUsesSmallerEgressMtu) { + ASSERT_EQ(SetMtu(1280), 0) << strerror(errno); + ASSERT_NO_FATAL_FAILURE(Connect(true)); + ASSERT_NO_FATAL_FAILURE(Transfer(8192, 1280)); +} +TEST_F(Ipv6RoutedOutput, ColdNeighborDiscoveryReleasesQueuedTcp) { + ASSERT_EQ(Neighbor(false), 0) << strerror(errno); + neighbor_added_ = false; + dynamic_neighbor_ = true; + ASSERT_NO_FATAL_FAILURE(Connect(true)); + ASSERT_GT(solicitations_, 0u) << "must exercise NDP, not a warm cache"; + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, UdpMtuSizedPayloadUsesPhysicalEgress) { + ASSERT_EQ(SetMtu(1280), 0) << strerror(errno); + client_.reset(socket(AF_INET6, SOCK_DGRAM, 0)); + ASSERT_GE(client_.get(), 0); + sockaddr_in6 local{}; local.sin6_family = AF_INET6; local.sin6_addr = source_; + ASSERT_EQ(bind(client_.get(), reinterpret_cast(&local), sizeof(local)), 0); + const char device[] = "veth2"; + ASSERT_EQ(setsockopt(client_.get(), SOL_SOCKET, SO_BINDTODEVICE, device, sizeof(device)), 0); + sockaddr_in6 remote{}; remote.sin6_family = AF_INET6; + remote.sin6_addr = peer_; remote.sin6_port = htons(kPeerPort); + std::vector payload(1232, 0x5a); + ASSERT_EQ(sendto(client_.get(), payload.data(), payload.size(), 0, + reinterpret_cast(&remote), sizeof(remote)), 1232); + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(4); + while (std::chrono::steady_clock::now() < deadline) { + pollfd event{packet_.get(), POLLIN, 0}; + ASSERT_GE(poll(&event, 1, 50), 0); + unsigned char frame[2048]; sockaddr_ll link{}; socklen_t size = sizeof(link); + ssize_t n = recvfrom(packet_.get(), frame, sizeof(frame), 0, + reinterpret_cast(&link), &size); + if (n < 62 || Get16(frame + 12) != ETH_P_IPV6) continue; + const auto* ip = frame + 14; + if (ip[6] != IPPROTO_UDP || memcmp(ip + 8, &source_, 16) || + memcmp(ip + 24, &peer_, 16) || Get16(ip + 42) != kPeerPort) continue; + ASSERT_EQ(link.sll_pkttype, PACKET_HOST); + ASSERT_EQ(n, 14 + 1280); + ASSERT_EQ(Get16(ip + 4), 1240); + ASSERT_EQ(Get16(ip + 44), 1240); + ASSERT_EQ(Checksum(ip, ip + 40, 1240, IPPROTO_UDP), 0); + ASSERT_EQ(memcmp(ip + 48, payload.data(), payload.size()), 0); + return; + } + FAIL() << "MTU-sized UDP datagram did not arrive physically from veth2"; +} +TEST_F(Ipv6RoutedOutput, MissingBoundDeviceRouteDoesNotConsumeSocket) { + client_.reset(socket(AF_INET6, SOCK_STREAM | SOCK_NONBLOCK, 0)); + ASSERT_GE(client_.get(), 0); + sockaddr_in6 local{}; local.sin6_family = AF_INET6; local.sin6_addr = source_; + ASSERT_EQ(bind(client_.get(), reinterpret_cast(&local), sizeof(local)), 0); + const char wrong_device[] = "veth1"; + ASSERT_EQ(setsockopt(client_.get(), SOL_SOCKET, SO_BINDTODEVICE, + wrong_device, sizeof(wrong_device)), 0); + sockaddr_in6 remote{}; remote.sin6_family = AF_INET6; + remote.sin6_addr = peer_; remote.sin6_port = htons(kPeerPort); + ASSERT_EQ(connect(client_.get(), reinterpret_cast(&remote), sizeof(remote)), -1); + ASSERT_EQ(errno, ENETUNREACH); + ASSERT_NO_FATAL_FAILURE(Connect(true, true)); + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, GlobalSourceNeighborAdvertisementStaysOnIngressInterface) { + // A global source owned by the other interface must not reroute the NDP + // reply through RTN_LOCAL. NDP is always scoped to the receiving link. + source_neighbor_ = true; + ASSERT_NO_FATAL_FAILURE(Ndisc(135, source_, egress_address_, egress_address_)); + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(4); + while (std::chrono::steady_clock::now() < deadline) { + pollfd event{packet_.get(), POLLIN, 0}; + ASSERT_GE(poll(&event, 1, 50), 0); + unsigned char frame[2048]; + sockaddr_ll link{}; socklen_t size = sizeof(link); + ssize_t n = recvfrom(packet_.get(), frame, sizeof(frame), 0, + reinterpret_cast(&link), &size); + if (n < 78 || Get16(frame + 12) != ETH_P_IPV6) continue; + const auto* ip = frame + 14; + if (ip[6] != IPPROTO_ICMPV6 || ip[40] != 136 || + memcmp(ip + 8, &egress_address_, 16) || memcmp(ip + 24, &source_, 16)) continue; + ASSERT_EQ(link.sll_pkttype, PACKET_HOST); + ASSERT_EQ(ip[7], 255); + ASSERT_LE(54 + Get16(ip + 4), n); + ASSERT_EQ(Checksum(ip, ip + 40, Get16(ip + 4), IPPROTO_ICMPV6), 0); + return; + } + FAIL() << "NDP response was not emitted back through physical ingress veth2"; +} +} // namespace + +int main(int argc, char** argv) { + testing::InitGoogleTest(&argc, argv); + return RUN_ALL_TESTS(); +} diff --git a/user/apps/tests/dunitest/whitelist.txt b/user/apps/tests/dunitest/whitelist.txt index 29d76b6a1..4c35c6926 100644 --- a/user/apps/tests/dunitest/whitelist.txt +++ b/user/apps/tests/dunitest/whitelist.txt @@ -61,6 +61,7 @@ normal/tcp_relisten normal/tcp_listener_overflow normal/tcp_accept_handshake normal/tcp_device_binding +normal/ipv6_routed_output normal/socket_nonblocking normal/tcp_self_connect_semantics normal/poll_timeout_semantics