From a5e2568742a75b69dc7031aee3b4a6ec2101f4e3 Mon Sep 17 00:00:00 2001 From: longjin Date: Tue, 22 Sep 2026 15:28:36 +0000 Subject: [PATCH 1/2] fix(net): route native IPv6 output across interfaces Use the namespace FIB and socket device constraint for native IPv6 output instead of implicitly transmitting on the SocketSet owner. Keep the owner stable and hand incoming local IPv6 packets to it while preserving the physical ingress interface. Generalize the existing bounded output and neighbor retry queues to complete IP addresses, including a full-width Patricia key and opaque route context. Activate configured-neighbor routing for both families and preserve existing IPv4 behavior. Use the selected egress MTU for TCP segmentation through smoltcp 4cfe0dd, keep NDP on the receiving link, and reject unsupported oversized native IPv6 UDP output before enqueueing. Validate fixed-source routes before changing TCP state. Add seven dunitest cases covering bidirectional cross-interface traffic, device constraints, 8192-byte TCP over MTU 1280, cold NDP, MTU-sized UDP, retry after ENETUNREACH, and global-source neighbor advertisements. Register them in whitelist and no_skip. Validation: make fmt and make kernel passed using the public dependency revision; Linux reference 21/21, DragonOS repeated focused tests 70/70 and existing networking regressions 161/161 passed. Real nginx IPv6 upstream proxy, dual-stack HTTP and graceful exit passed. IPv6 source fragmentation, PMTU socket options and full scope-id support remain outside this change. Signed-off-by: longjin --- kernel/Cargo.lock | 2 +- kernel/Cargo.toml | 2 +- kernel/src/driver/net/deferred_index.rs | 72 ++- kernel/src/driver/net/deferred_queue.rs | 58 ++- kernel/src/driver/net/iface.rs | 96 ++-- kernel/src/driver/net/iface_common.rs | 69 ++- kernel/src/driver/net/local_output.rs | 55 ++- kernel/src/driver/net/local_queue.rs | 132 ++++-- kernel/src/driver/net/loopback.rs | 2 +- kernel/src/net/neighbor/mod.rs | 12 +- kernel/src/net/neighbor/table.rs | 30 +- kernel/src/net/route/fib.rs | 10 +- kernel/src/net/route/fib_index.rs | 7 + kernel/src/net/route/mod.rs | 1 + kernel/src/net/route/source.rs | 44 +- kernel/src/net/routing/mod.rs | 114 ++++- kernel/src/net/socket/inet/common/mod.rs | 15 +- kernel/src/net/socket/inet/datagram/mod.rs | 22 + .../src/net/socket/inet/stream/lifecycle.rs | 11 + user/apps/tests/dunitest/no_skip.txt | 1 + .../suites/normal/ipv6_routed_output.cc | 430 ++++++++++++++++++ user/apps/tests/dunitest/whitelist.txt | 1 + 22 files changed, 974 insertions(+), 212 deletions(-) create mode 100644 user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index cfbe7af70..ec385f3f8 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -1561,7 +1561,7 @@ checksum = "7fcf8323ef1faaee30a44a340193b1ac6814fd9b7b4e88e9d4519a3e4abe1cfd" [[package]] name = "smoltcp" version = "0.12.0" -source = "git+https://github.com/DragonOS-Community/smoltcp?rev=9805dae47858c7247303d9d6d19ca504632c1e4a#9805dae47858c7247303d9d6d19ca504632c1e4a" +source = "git+https://github.com/DragonOS-Community/smoltcp?rev=4cfe0ddadd6c0f5a82a72245bf18c4df6f4bb8fd#4cfe0ddadd6c0f5a82a72245bf18c4df6f4bb8fd" dependencies = [ "bitflags 1.3.2", "byteorder", diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index 19848e32d..ef268d558 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -62,7 +62,7 @@ linkme = "=0.3.27" num = { version = "=0.4.0", default-features = false } num-derive = "=0.3" num-traits = { git = "https://git.mirrors.dragonos.org.cn/DragonOS-Community/num-traits.git", rev = "1597c1c", default-features = false } -smoltcp = { version = "=0.12.0", git = "https://github.com/DragonOS-Community/smoltcp", rev = "9805dae47858c7247303d9d6d19ca504632c1e4a", default-features = false, features = [ +smoltcp = { version = "=0.12.0", git = "https://github.com/DragonOS-Community/smoltcp", rev = "4cfe0ddadd6c0f5a82a72245bf18c4df6f4bb8fd", default-features = false, features = [ "alloc", "medium-ethernet", "socket-raw", diff --git a/kernel/src/driver/net/deferred_index.rs b/kernel/src/driver/net/deferred_index.rs index 86d589b72..3f150e388 100644 --- a/kernel/src/driver/net/deferred_index.rs +++ b/kernel/src/driver/net/deferred_index.rs @@ -2,15 +2,16 @@ use alloc::vec::Vec; use system_error::SystemError; pub(super) type NodeId = usize; +pub(super) type RouteIndexKey = [u8; 21]; #[derive(Clone, Copy, Debug)] enum Node { Free { next: Option }, - Leaf { key: u64, slot: usize }, + Leaf { key: RouteIndexKey, slot: usize }, Branch { bit: u8, children: [NodeId; 2] }, } -/// A fallibly allocated Patricia index for attacker-controlled 64-bit keys. +/// A fallibly allocated Patricia index for complete (family, ifindex, address) keys. /// /// Branch bits strictly increase from root to leaf, bounding every lookup by /// the key width without depending on secret hash entropy. Node IDs remain @@ -29,7 +30,7 @@ impl DeferredRouteIndex { self.leaves } - pub(super) fn get(&self, key: u64) -> Option<(NodeId, usize)> { + pub(super) fn get(&self, key: RouteIndexKey) -> Option<(NodeId, usize)> { let leaf = self.find_leaf(key)?; match self.nodes[leaf] { Node::Leaf { @@ -48,7 +49,7 @@ impl DeferredRouteIndex { } /// Inserts a key after `try_reserve_insert`; this method cannot allocate. - pub(super) fn insert_prepared(&mut self, key: u64, slot: usize) -> NodeId { + pub(super) fn insert_prepared(&mut self, key: RouteIndexKey, slot: usize) -> NodeId { debug_assert!(self.get(key).is_none()); let Some(root) = self.root else { let leaf = self.alloc_prepared(Node::Leaf { key, slot }); @@ -64,8 +65,14 @@ impl DeferredRouteIndex { Node::Leaf { key, .. } => key, _ => unreachable!("Patricia traversal ends at a leaf"), }; - let differing_bit = (key ^ existing_key).leading_zeros() as u8; - debug_assert!(differing_bit < 64); + let differing_byte = key + .iter() + .zip(existing_key) + .position(|(a, b)| *a != b) + .expect("inserted Patricia keys are distinct"); + let differing_bit = (differing_byte * 8 + + (key[differing_byte] ^ existing_key[differing_byte]).leading_zeros() as usize) + as u8; let mut parent = None; let mut current = root; @@ -95,7 +102,7 @@ impl DeferredRouteIndex { leaf } - pub(super) fn set_slot(&mut self, leaf: NodeId, key: u64, slot: usize) { + pub(super) fn set_slot(&mut self, leaf: NodeId, key: RouteIndexKey, slot: usize) { match &mut self.nodes[leaf] { Node::Leaf { key: leaf_key, @@ -108,7 +115,7 @@ impl DeferredRouteIndex { } } - pub(super) fn remove(&mut self, key: u64) -> Option { + pub(super) fn remove(&mut self, key: RouteIndexKey) -> Option { let root = self.root?; if let Node::Leaf { key: leaf_key, @@ -153,7 +160,7 @@ impl DeferredRouteIndex { Some(slot) } - fn find_leaf(&self, key: u64) -> Option { + fn find_leaf(&self, key: RouteIndexKey) -> Option { let mut current = self.root?; loop { match self.nodes[current] { @@ -166,8 +173,8 @@ impl DeferredRouteIndex { } } - fn direction(key: u64, bit: u8) -> usize { - ((key >> (63 - bit)) & 1) as usize + fn direction(key: RouteIndexKey, bit: u8) -> usize { + ((key[bit as usize / 8] >> (7 - bit % 8)) & 1) as usize } fn branch_children(&self, node: NodeId) -> [NodeId; 2] { @@ -209,3 +216,46 @@ impl DeferredRouteIndex { self.free_count += 1; } } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn every_key_bit_survives_insert_remove_and_slot_updates() { + let mut index = DeferredRouteIndex::default(); + // Include the all-zero key and every differing bit, especially the + // high IPv6 bytes which the old 64-bit key could not represent. + let mut keys = alloc::vec![[0; 21]]; + for bit in 0..168 { + let mut key = [0; 21]; + key[bit / 8] = 1 << (7 - bit % 8); + keys.push(key); + } + for (slot, key) in keys.iter().copied().enumerate() { + index.try_reserve_insert().unwrap(); + index.insert_prepared(key, slot); + } + assert_eq!(index.len(), keys.len()); + for (slot, key) in keys.iter().copied().enumerate() { + let (leaf, actual) = index.get(key).unwrap(); + assert_eq!(actual, slot); + index.set_slot(leaf, key, slot + 1000); + } + // Alternating removals exercise root replacement and stable leaf IDs. + for parity in 0..2 { + for slot in (parity..keys.len()).step_by(2) { + assert_eq!(index.remove(keys[slot]), Some(slot + 1000)); + assert!(index.get(keys[slot]).is_none()); + } + } + assert_eq!(index.len(), 0); + for (slot, key) in keys.iter().copied().enumerate().rev() { + index.try_reserve_insert().unwrap(); + index.insert_prepared(key, slot); + } + for (slot, key) in keys.iter().copied().enumerate() { + assert_eq!(index.get(key).unwrap().1, slot); + } + } +} diff --git a/kernel/src/driver/net/deferred_queue.rs b/kernel/src/driver/net/deferred_queue.rs index f84dd4eba..1ecc7074d 100644 --- a/kernel/src/driver/net/deferred_queue.rs +++ b/kernel/src/driver/net/deferred_queue.rs @@ -1,5 +1,5 @@ use super::{ - deferred_index::{DeferredRouteIndex, NodeId}, + deferred_index::{DeferredRouteIndex, NodeId, RouteIndexKey}, local_queue::{LocalOutputDisposition, LocalOutputPacket}, }; use alloc::{collections::VecDeque, vec::Vec}; @@ -7,12 +7,58 @@ use alloc::{collections::VecDeque, vec::Vec}; #[derive(Clone, Copy, Debug, Eq, PartialEq)] pub(super) struct DeferredRouteKey { pub(super) oif: u32, - pub(super) next_hop: smoltcp::wire::Ipv4Address, + pub(super) next_hop: smoltcp::wire::IpAddress, } impl DeferredRouteKey { - fn packed(self) -> u64 { - ((self.oif as u64) << 32) | u32::from_be_bytes(self.next_hop.octets()) as u64 + fn packed(self) -> RouteIndexKey { + let mut key = [0; 21]; + key[1..5].copy_from_slice(&self.oif.to_be_bytes()); + match self.next_hop { + smoltcp::wire::IpAddress::Ipv4(address) => { + key[0] = 4; + key[17..].copy_from_slice(&address.octets()); + } + smoltcp::wire::IpAddress::Ipv6(address) => { + key[0] = 6; + key[5..].copy_from_slice(&address.octets()); + } + } + key + } +} + +#[cfg(test)] +mod key_tests { + use super::*; + use smoltcp::wire::{IpAddress, Ipv4Address, Ipv6Address}; + + #[test] + fn family_device_and_all_address_bytes_identify_a_neighbor() { + let v4 = DeferredRouteKey { + oif: 2, + next_hop: IpAddress::Ipv4(Ipv4Address::new(192, 0, 2, 1)), + }; + let mut bytes = [0; 16]; + bytes[12..].copy_from_slice(&[192, 0, 2, 1]); + let v6 = DeferredRouteKey { + oif: 2, + next_hop: IpAddress::Ipv6(Ipv6Address::from(bytes)), + }; + assert_ne!(v4.packed(), v6.packed()); + assert_ne!(v6.packed(), DeferredRouteKey { oif: 3, ..v6 }.packed()); + for byte in 0..16 { + let mut different = bytes; + different[byte] ^= 0x80; + assert_ne!( + v6.packed(), + DeferredRouteKey { + next_hop: IpAddress::Ipv6(Ipv6Address::from(different)), + ..v6 + } + .packed() + ); + } } } @@ -43,7 +89,7 @@ pub(super) struct DeferredRouteLimits { /// Per-interface neighbor-resolution backlog. /// -/// The hash index gives direct access by `(oif, next_hop)`, while `heap` is an +/// The Patricia index gives bounded access by `(oif, next_hop)`, while `heap` is an /// indexed min-heap whose root is the next schedulable neighbor. In-flight /// buckets sort after every schedulable bucket and therefore never publish a /// stale retry deadline. Both structures are protected by the containing @@ -300,7 +346,7 @@ impl DeferredRouteQueue { /// compaction and heap rebuilding are each performed only once. pub(super) fn release_resolved( &mut self, - mut is_resolved: impl FnMut(smoltcp::wire::Ipv4Address) -> bool, + mut is_resolved: impl FnMut(smoltcp::wire::IpAddress) -> bool, ready: &mut VecDeque, ) { let mut removed_any = false; diff --git a/kernel/src/driver/net/iface.rs b/kernel/src/driver/net/iface.rs index 4b6fa3532..6539ccf7b 100644 --- a/kernel/src/driver/net/iface.rs +++ b/kernel/src/driver/net/iface.rs @@ -181,7 +181,7 @@ pub trait Iface: crate::driver::base::device::Device { self.raw_transmit(&frame) } - /// Sends an already-routed IPv4 packet through this interface. + /// Sends an already-routed IP packet through this interface. /// /// Ethernet devices share this implementation so FIB eligibility cannot /// drift from driver-specific forwarding hooks. smoltcp resolves the @@ -193,10 +193,9 @@ pub trait Iface: crate::driver::base::device::Device { next_hop: &smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result<(), RouteSendError> { - let smoltcp::wire::IpAddress::Ipv4(next_hop) = *next_hop else { - return Err(SystemError::EAFNOSUPPORT.into()); - }; - let frame_capacity = (14usize + ip_packet.len()).max(42); + // A short original packet may instead emit an ARP request or an + // IPv6 Neighbor Solicitation (Ethernet + IPv6 + NS with SLLAO). + let frame_capacity = (14usize + ip_packet.len()).max(86); let mut frame = Vec::new(); frame .try_reserve_exact(frame_capacity) @@ -204,24 +203,18 @@ pub trait Iface: crate::driver::base::device::Device { let frame_prepared = Cell::new(false); let permanent_neighbor = self .net_namespace() - .and_then(|netns| { - crate::net::neighbor::lookup( - &netns, - self.nic_id() as u32, - smoltcp::wire::IpAddress::Ipv4(next_hop), - ) - }) + .and_then(|netns| crate::net::neighbor::lookup(&netns, self.nic_id() as u32, *next_hop)) .map(smoltcp::wire::HardwareAddress::Ethernet); let dispatch = { let mut interface = self.smol_iface().lock(); - interface.dispatch_ipv4_packet( + interface.dispatch_ip_packet( crate::time::Instant::now().into(), PreparedFrameTxToken { frame: &mut frame, prepared: &frame_prepared, }, - next_hop, + *next_hop, permanent_neighbor, ip_packet, ) @@ -234,20 +227,20 @@ pub trait Iface: crate::driver::base::device::Device { } match dispatch { Ok(()) => Ok(()), - Err(smoltcp::iface::Ipv4PacketDispatchError::NeighborPending { retry_at }) => { + Err(smoltcp::iface::IpPacketDispatchError::NeighborPending { retry_at }) => { Err(RouteSendError::RetryAt { retry_at, probe_sent, }) } - Err(smoltcp::iface::Ipv4PacketDispatchError::NoRoute) => { + Err(smoltcp::iface::IpPacketDispatchError::NoRoute) => { Err(SystemError::ENETUNREACH.into()) } - Err(smoltcp::iface::Ipv4PacketDispatchError::Exhausted) => { + Err(smoltcp::iface::IpPacketDispatchError::Exhausted) => { Err(SystemError::EAGAIN_OR_EWOULDBLOCK.into()) } - Err(smoltcp::iface::Ipv4PacketDispatchError::Malformed) - | Err(smoltcp::iface::Ipv4PacketDispatchError::InvalidHardwareAddress) => { + Err(smoltcp::iface::IpPacketDispatchError::Malformed) + | Err(smoltcp::iface::IpPacketDispatchError::InvalidHardwareAddress) => { Err(SystemError::EINVAL.into()) } } @@ -261,9 +254,7 @@ pub trait Iface: crate::driver::base::device::Device { next_hop: &smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result<(), SystemError> { - let smoltcp::wire::IpAddress::Ipv4(next_hop) = *next_hop else { - return Err(SystemError::EAFNOSUPPORT); - }; + let next_hop = *next_hop; if ip_packet.len() > self.mtu() { return Err(SystemError::EMSGSIZE); } @@ -283,35 +274,34 @@ pub trait Iface: crate::driver::base::device::Device { return Ok(()); } let tx_generation = self.common().tx_completion_generation(); - let (retry_at, probe_sent) = - match self.route_and_send(&smoltcp::wire::IpAddress::Ipv4(next_hop), ip_packet) { - Ok(()) => return Ok(()), - Err(RouteSendError::RetryAt { - retry_at, - probe_sent, - }) => (retry_at, probe_sent), - Err(RouteSendError::Failed(SystemError::ENOBUFS)) - | Err(RouteSendError::Failed(SystemError::EAGAIN_OR_EWOULDBLOCK)) => { - let now: smoltcp::time::Instant = crate::time::Instant::now().into(); - let delay_us = self.common().next_local_output_tx_backoff_us(); - let retry_at = now + smoltcp::time::Duration::from_micros(delay_us); - let (packet, reservation) = self.common().prepare_routed_output( - self.nic_id() as u32, - next_hop, - ip_packet, - )?; - reservation.requeue_backpressured(packet, retry_at); - let retry_at = if self.common().release_tx_backpressure_after(tx_generation) { - now - } else { - retry_at - }; - self.common() - .schedule_local_output(retry_at, napi, scheduler_netns); - return Ok(()); - } - Err(RouteSendError::Failed(error)) => return Err(error), - }; + let (retry_at, probe_sent) = match self.route_and_send(&next_hop, ip_packet) { + Ok(()) => return Ok(()), + Err(RouteSendError::RetryAt { + retry_at, + probe_sent, + }) => (retry_at, probe_sent), + Err(RouteSendError::Failed(SystemError::ENOBUFS)) + | Err(RouteSendError::Failed(SystemError::EAGAIN_OR_EWOULDBLOCK)) => { + let now: smoltcp::time::Instant = crate::time::Instant::now().into(); + let delay_us = self.common().next_local_output_tx_backoff_us(); + let retry_at = now + smoltcp::time::Duration::from_micros(delay_us); + let (packet, reservation) = self.common().prepare_routed_output( + self.nic_id() as u32, + next_hop, + ip_packet, + )?; + reservation.requeue_backpressured(packet, retry_at); + let retry_at = if self.common().release_tx_backpressure_after(tx_generation) { + now + } else { + retry_at + }; + self.common() + .schedule_local_output(retry_at, napi, scheduler_netns); + return Ok(()); + } + Err(RouteSendError::Failed(error)) => return Err(error), + }; self.common().enqueue_routed_output( self.nic_id() as u32, @@ -337,11 +327,11 @@ pub trait Iface: crate::driver::base::device::Device { Ok(()) } - /// Hands a namespace-local IPv4 packet to this interface's protocol stack + /// Hands a namespace-local IP packet to this interface's protocol stack /// without emitting it on the link. The shared queue in `IfaceCommon` /// makes local delivery a protocol-stack capability rather than an /// optional device-driver feature. - fn inject_local_ipv4_packet( + fn inject_local_ip_packet( &self, ingress_ifindex: u32, source_mac: smoltcp::wire::EthernetAddress, diff --git a/kernel/src/driver/net/iface_common.rs b/kernel/src/driver/net/iface_common.rs index 71d52f0f9..a60469e28 100644 --- a/kernel/src/driver/net/iface_common.rs +++ b/kernel/src/driver/net/iface_common.rs @@ -194,7 +194,7 @@ impl IfaceCommon { pub(super) fn enqueue_routed_output( &self, oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_packet: &[u8], retry_at: smoltcp::time::Instant, probe_sent: bool, @@ -211,7 +211,7 @@ impl IfaceCommon { pub(super) fn prepare_routed_output( &self, oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result<(LocalOutputPacket, LocalOutputReservation<'_>), SystemError> { if ip_packet.len() > self.mtu.load(Ordering::Acquire) { @@ -253,7 +253,7 @@ impl IfaceCommon { pub(super) fn enqueue_existing_routed_output( &self, oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_packet: &[u8], ) -> Result, SystemError> { let pending = self @@ -381,12 +381,8 @@ impl IfaceCommon { } let ifindex = self.iface_id as u32; self.local_input_queue.release_resolved_outputs(|next_hop| { - configured.is_some_and(|neighbors| { - neighbors - .lookup(ifindex, smoltcp::wire::IpAddress::Ipv4(next_hop)) - .is_some() - }) || interface - .is_neighbor_resolved(timestamp, smoltcp::wire::IpAddress::Ipv4(next_hop)) + configured.is_some_and(|neighbors| neighbors.lookup(ifindex, next_hop).is_some()) + || interface.is_neighbor_resolved(timestamp, next_hop) }); } @@ -419,16 +415,15 @@ impl IfaceCommon { &self, netns: Option<&Arc>, needs_routed_poll: bool, - authoritative_ipv4_output: bool, - configured_ipv4_output: bool, + authoritative_output: bool, + configured_output: bool, ) -> PollModeRecheck { - if !authoritative_ipv4_output - && netns.is_some_and(|netns| netns.router().requires_authoritative_ipv4_output()) + if !authoritative_output + && netns.is_some_and(|netns| netns.router().requires_authoritative_output()) { PollModeRecheck::Authoritative } else if (!needs_routed_poll && self.needs_namespace_routing()) - || (!configured_ipv4_output - && netns.is_some_and(crate::net::neighbor::has_ipv4_entries)) + || (!configured_output && netns.is_some_and(crate::net::neighbor::has_ethernet_entries)) { PollModeRecheck::Routed } else { @@ -514,17 +509,17 @@ impl IfaceCommon { } let netns = self.net_namespace(); - let authoritative_ipv4_output = force_authoritative + let authoritative_output = force_authoritative || netns .as_ref() - .is_some_and(|netns| netns.router().requires_authoritative_ipv4_output()); - let configured_ipv4_output = netns + .is_some_and(|netns| netns.router().requires_authoritative_output()); + let configured_output = netns .as_ref() - .is_some_and(crate::net::neighbor::has_ipv4_entries); + .is_some_and(crate::net::neighbor::has_ethernet_entries); let needs_routed_poll = self.needs_namespace_routing() || scope == IfacePollScope::LocalOnly - || authoritative_ipv4_output - || configured_ipv4_output; + || authoritative_output + || configured_output; let router = if needs_routed_poll { netns.as_ref().map(|netns| netns.router()) } else { @@ -549,7 +544,7 @@ impl IfaceCommon { continue; } let configured_neighbors = netns.as_ref().and_then(|netns| { - crate::net::neighbor::has_ipv4_entries(netns) + crate::net::neighbor::has_ethernet_entries(netns) .then(|| crate::net::neighbor::read(netns)) }); @@ -559,8 +554,8 @@ impl IfaceCommon { let restart = self.recheck_poll_mode( netns.as_ref(), needs_routed_poll, - authoritative_ipv4_output, - configured_ipv4_output, + authoritative_output, + configured_output, ); if restart != PollModeRecheck::Current { drop(configured_neighbors); @@ -576,7 +571,7 @@ impl IfaceCommon { configured_neighbors: configured_neighbors.as_ref(), owner_ifindex: self.iface_id as u32, owner_is_up, - authoritative_ipv4_output, + authoritative_output, }); let (has_events, poll_again, deadline_rearm) = { @@ -701,17 +696,17 @@ impl IfaceCommon { } let netns = self.net_namespace(); - let authoritative_ipv4_output = force_authoritative + let authoritative_output = force_authoritative || netns .as_ref() - .is_some_and(|netns| netns.router().requires_authoritative_ipv4_output()); - let configured_ipv4_output = netns + .is_some_and(|netns| netns.router().requires_authoritative_output()); + let configured_output = netns .as_ref() - .is_some_and(crate::net::neighbor::has_ipv4_entries); + .is_some_and(crate::net::neighbor::has_ethernet_entries); let needs_routed_poll = self.needs_namespace_routing() || scope == IfacePollScope::LocalOnly - || authoritative_ipv4_output - || configured_ipv4_output; + || authoritative_output + || configured_output; let router = if needs_routed_poll { netns.as_ref().map(|netns| netns.router()) } else { @@ -736,15 +731,15 @@ impl IfaceCommon { continue; } let configured_neighbors = netns.as_ref().and_then(|netns| { - crate::net::neighbor::has_ipv4_entries(netns) + crate::net::neighbor::has_ethernet_entries(netns) .then(|| crate::net::neighbor::read(netns)) }); let restart = self.recheck_poll_mode( netns.as_ref(), needs_routed_poll, - authoritative_ipv4_output, - configured_ipv4_output, + authoritative_output, + configured_output, ); if restart != PollModeRecheck::Current { drop(configured_neighbors); @@ -760,7 +755,7 @@ impl IfaceCommon { configured_neighbors: configured_neighbors.as_ref(), owner_ifindex: self.iface_id as u32, owner_is_up, - authoritative_ipv4_output, + authoritative_output, }); let mut processed = 0usize; @@ -1417,7 +1412,7 @@ impl IfaceCommon { pub(crate) fn release_configured_neighbor( &self, ifindex: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ) -> bool { self.local_input_queue.release_neighbor(ifindex, next_hop) } @@ -1428,7 +1423,7 @@ impl IfaceCommon { pub(crate) fn configured_neighbor_committed( &self, ifindex: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ) { if self.release_configured_neighbor(ifindex, next_hop) { self.schedule_registered_local_output(crate::time::Instant::now().into()); diff --git a/kernel/src/driver/net/local_output.rs b/kernel/src/driver/net/local_output.rs index 8e87df476..c78171a70 100644 --- a/kernel/src/driver/net/local_output.rs +++ b/kernel/src/driver/net/local_output.rs @@ -64,9 +64,8 @@ pub(super) fn defer_native_output_after_tx_backpressure( /// A namespace-local view over the target interface's transport stack. /// Ingress retains the physical ifindex. Output is staged until the smoltcp -/// locks are released: IPv4 may then select another device through the -/// namespace FIB, while native same-interface and non-IPv4 traffic keeps the -/// underlying device path. +/// locks are released: unicast IP may then select another device through the +/// namespace FIB, while native link-local control traffic keeps the device path. pub(super) struct LocalInputDevice<'a, D: SmolDevice + ?Sized> { pub(super) device: &'a mut D, pub(super) common: &'a IfaceCommon, @@ -74,7 +73,7 @@ pub(super) struct LocalInputDevice<'a, D: SmolDevice + ?Sized> { } /// Delegates receive to the physical device while routing every response and -/// standalone IPv4 transmission through the same deferred output FIFO as +/// standalone routed IP transmission through the same deferred output FIFO as /// namespace-local input. pub(super) struct RoutedTxDevice<'a, D: SmolDevice + ?Sized> { pub(super) device: &'a mut D, @@ -125,17 +124,31 @@ pub(super) struct OutputBackendPolicy<'a> { pub(super) configured_neighbors: Option<&'a crate::net::neighbor::NeighborReadGuard<'a>>, pub(super) owner_ifindex: u32, pub(super) owner_is_up: bool, - pub(super) authoritative_ipv4_output: bool, + pub(super) authoritative_output: bool, } impl OutputBackendPolicy<'_> { + fn outbound_ip_mtu( + self, + destination: smoltcp::wire::IpAddress, + meta: PacketMeta, + native_mtu: usize, + ) -> usize { + match self.classify(destination.version(), destination, meta) { + OutputBackendDecision::Deferred(Some(route)) => route.ip_mtu.min(u16::MAX as usize), + // A missing route must not prevent TCP from advancing its timers. + // Actual packet dispatch still rejects the missing route. + _ => native_mtu, + } + } + pub(super) fn classify( self, version: smoltcp::wire::IpVersion, destination: smoltcp::wire::IpAddress, meta: PacketMeta, ) -> OutputBackendDecision { - if version != smoltcp::wire::IpVersion::Ipv4 { + if version == smoltcp::wire::IpVersion::Ipv6 && destination.is_multicast() { return OutputBackendDecision::NativeOwner; } let constrained_oif = (meta.id != 0).then_some(meta.id); @@ -149,7 +162,7 @@ impl OutputBackendPolicy<'_> { && !self.configured_neighbors.is_some_and(|neighbors| { neighbors.lookup(route.oif, route.next_hop).is_some() }) - && (!self.authoritative_ipv4_output + && (!self.authoritative_output || route.table != crate::net::route::RT_TABLE_DEFAULT) => { OutputBackendDecision::NativeOwner @@ -344,15 +357,7 @@ impl LocalInputTxToken<'_> { context: LocalOutputDisposition::local_context(route.oif), })); } - let smoltcp::wire::IpAddress::Ipv4(next_hop) = route.next_hop else { - self.medium = smoltcp::phy::Medium::Ip; - self.disposition = LocalOutputDisposition::Drop; - return Ok(Some(smoltcp::phy::TxEgressOverride { - medium: smoltcp::phy::Medium::Ip, - ip_mtu: self.owner_ip_mtu, - context: LocalOutputDisposition::DROP_CONTEXT, - })); - }; + let next_hop = route.next_hop; self.medium = smoltcp::phy::Medium::Ip; // The address owner and selected egress may have different MTUs. // Grow only the token that actually needs the larger route MTU; @@ -506,6 +511,11 @@ pub(super) fn local_tx_token<'a>( } impl SmolDevice for LocalInputDevice<'_, D> { + fn outbound_ip_mtu(&self, destination: smoltcp::wire::IpAddress, meta: PacketMeta) -> usize { + self.backend_policy + .outbound_ip_mtu(destination, meta, self.device.capabilities().ip_mtu()) + } + type RxToken<'a> = LocalInputRxToken where @@ -541,6 +551,11 @@ impl SmolDevice for LocalInputDevice<'_, D> { } impl SmolDevice for RoutedTxDevice<'_, D> { + fn outbound_ip_mtu(&self, destination: smoltcp::wire::IpAddress, meta: PacketMeta) -> usize { + self.backend_policy + .outbound_ip_mtu(destination, meta, self.device.capabilities().ip_mtu()) + } + type RxToken<'a> = RoutedRxToken> where @@ -623,7 +638,7 @@ pub(super) fn transmit_routed_stack_output( }; if packet.medium != smoltcp::phy::Medium::Ip || packet.frame.len() > ip_mtu - || packet.frame.first().map(|byte| byte >> 4) != Some(4) + || smoltcp::wire::IpVersion::of_packet(&packet.frame).ok() != Some(next_hop.version()) { return LocalOutputTransmitResult::Drop(packet, SystemError::EINVAL); } @@ -635,7 +650,7 @@ pub(super) fn transmit_routed_stack_output( }; return LocalOutputTransmitResult::Drop(packet, error); } - match iface.route_and_send(&smoltcp::wire::IpAddress::Ipv4(next_hop), &packet.frame) { + match iface.route_and_send(&next_hop, &packet.frame) { Ok(()) => LocalOutputTransmitResult::Sent(packet), Err(RouteSendError::RetryAt { retry_at, @@ -738,7 +753,7 @@ where LocalOutputDisposition::Local { oif, ip_mtu } => { if packet.medium != smoltcp::phy::Medium::Ip || packet.frame.len() > ip_mtu - || packet.frame.first().map(|byte| byte >> 4) != Some(4) + || smoltcp::wire::IpVersion::of_packet(&packet.frame).is_err() { return LocalOutputTransmitResult::Drop(packet, SystemError::EINVAL); } @@ -748,7 +763,7 @@ where if packet.frame.len() > iface.mtu() { return LocalOutputTransmitResult::Drop(packet, SystemError::EMSGSIZE); } - match iface.inject_local_ipv4_packet(oif, iface.mac(), &packet.frame, false) { + match iface.inject_local_ip_packet(oif, iface.mac(), &packet.frame, false) { Ok(()) => LocalOutputTransmitResult::Sent(packet), // This is receive-backlog congestion, not physical TX // backpressure. Linux may drop locally delivered packets when diff --git a/kernel/src/driver/net/local_queue.rs b/kernel/src/driver/net/local_queue.rs index 100039592..f90d6b8e1 100644 --- a/kernel/src/driver/net/local_queue.rs +++ b/kernel/src/driver/net/local_queue.rs @@ -34,6 +34,12 @@ impl LocalInputPacket { } pub(super) fn into_frame(self, medium: smoltcp::phy::Medium) -> Result, SystemError> { + let ethertype = match smoltcp::wire::IpVersion::of_packet(&self.ip_packet) + .map_err(|_| SystemError::EINVAL)? + { + smoltcp::wire::IpVersion::Ipv4 => [0x08, 0x00], + smoltcp::wire::IpVersion::Ipv6 => [0x86, 0xdd], + }; if medium == smoltcp::phy::Medium::Ip { return Ok(self.ip_packet); } @@ -46,7 +52,7 @@ impl LocalInputPacket { .map_err(|_| SystemError::ENOMEM)?; frame.extend_from_slice(&self.destination_mac.0); frame.extend_from_slice(&self.source_mac.0); - frame.extend_from_slice(&[0x08, 0x00]); + frame.extend_from_slice(ðertype); frame.extend_from_slice(&self.ip_packet); Ok(frame) } @@ -83,55 +89,127 @@ pub(super) enum LocalOutputDisposition { }, Routed { oif: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: smoltcp::wire::IpAddress, ip_mtu: usize, }, Drop, } impl LocalOutputDisposition { - pub(super) const DROP_CONTEXT: u64 = 0; - const LOCAL_CONTEXT: u64 = 1 << 63; - // Reserved opaque smoltcp fragment context. Valid ifindices use the - // positive i32 range, so neither routed nor local encodings can collide - // with the all-ones value. - pub(super) const NATIVE_CONTEXT: u64 = u64::MAX; - - pub(super) fn routed_context(oif: u32, next_hop: smoltcp::wire::Ipv4Address) -> u64 { + pub(super) const DROP_CONTEXT: [u64; 3] = [0; 3]; + pub(super) const NATIVE_CONTEXT: [u64; 3] = [1, 0, 0]; + const LOCAL_TAG: u64 = 2; + const IPV4_TAG: u64 = 4; + const IPV6_TAG: u64 = 6; + + pub(super) fn routed_context(oif: u32, next_hop: smoltcp::wire::IpAddress) -> [u64; 3] { debug_assert_ne!(oif, 0); - debug_assert_eq!(oif & (1 << 31), 0); - ((oif as u64) << 32) | u32::from_be_bytes(next_hop.octets()) as u64 + match next_hop { + smoltcp::wire::IpAddress::Ipv4(address) => [ + ((oif as u64) << 32) | Self::IPV4_TAG, + 0, + u32::from_be_bytes(address.octets()) as u64, + ], + smoltcp::wire::IpAddress::Ipv6(address) => { + let bytes = address.octets(); + [ + ((oif as u64) << 32) | Self::IPV6_TAG, + u64::from_be_bytes(bytes[..8].try_into().unwrap()), + u64::from_be_bytes(bytes[8..].try_into().unwrap()), + ] + } + } } - pub(super) fn local_context(oif: u32) -> u64 { + pub(super) fn local_context(oif: u32) -> [u64; 3] { debug_assert_ne!(oif, 0); - debug_assert_eq!(oif & (1 << 31), 0); - Self::LOCAL_CONTEXT | ((oif as u64) << 32) + [((oif as u64) << 32) | Self::LOCAL_TAG, 0, 0] } - pub(super) fn from_context(context: u64, ip_mtu: usize) -> Self { + pub(super) fn from_context(context: [u64; 3], ip_mtu: usize) -> Self { if context == Self::NATIVE_CONTEXT { return Self::NativeOwner; } - if context & Self::LOCAL_CONTEXT != 0 { - return Self::Local { - oif: ((context >> 32) as u32) & !(1 << 31), - ip_mtu, - }; - } - let oif = (context >> 32) as u32; + let oif = (context[0] >> 32) as u32; if oif == 0 { return Self::Drop; } - let octets = (context as u32).to_be_bytes(); + let next_hop = match context[0] & u32::MAX as u64 { + Self::LOCAL_TAG if context[1..] == [0, 0] => return Self::Local { oif, ip_mtu }, + Self::IPV4_TAG if context[1] == 0 && context[2] <= u32::MAX as u64 => { + smoltcp::wire::IpAddress::Ipv4(smoltcp::wire::Ipv4Address::from( + (context[2] as u32).to_be_bytes(), + )) + } + Self::IPV6_TAG => { + let mut bytes = [0; 16]; + bytes[..8].copy_from_slice(&context[1].to_be_bytes()); + bytes[8..].copy_from_slice(&context[2].to_be_bytes()); + smoltcp::wire::IpAddress::Ipv6(smoltcp::wire::Ipv6Address::from(bytes)) + } + _ => return Self::Drop, + }; Self::Routed { oif, - next_hop: smoltcp::wire::Ipv4Address::new(octets[0], octets[1], octets[2], octets[3]), + next_hop, ip_mtu, } } } +#[cfg(test)] +mod context_tests { + use super::*; + use smoltcp::wire::{IpAddress, Ipv4Address, Ipv6Address}; + + #[test] + fn routed_context_preserves_both_families_and_full_ipv6_address() { + let addresses = [ + IpAddress::Ipv4(Ipv4Address::new(192, 0, 2, 1)), + IpAddress::Ipv6(Ipv6Address::from([ + 0xfd, 0x12, 0x34, 0x56, 0x78, 0x9a, 0xbc, 0xde, 0xf0, 0x11, 0x22, 0x33, 0xc0, 0, 2, + 1, + ])), + ]; + for address in addresses { + for ifindex in [1, 42, i32::MAX as u32] { + let context = LocalOutputDisposition::routed_context(ifindex, address); + let LocalOutputDisposition::Routed { + oif, + next_hop, + ip_mtu, + } = LocalOutputDisposition::from_context(context, 1280) + else { + panic!("routed context changed disposition") + }; + assert_eq!((oif, next_hop, ip_mtu), (ifindex, address, 1280)); + } + } + assert!(matches!( + LocalOutputDisposition::from_context(LocalOutputDisposition::local_context(42), 1500), + LocalOutputDisposition::Local { + oif: 42, + ip_mtu: 1500 + } + )); + assert!(matches!( + LocalOutputDisposition::from_context(LocalOutputDisposition::NATIVE_CONTEXT, 1500), + LocalOutputDisposition::NativeOwner + )); + assert!(matches!( + LocalOutputDisposition::from_context(LocalOutputDisposition::DROP_CONTEXT, 1500), + LocalOutputDisposition::Drop + )); + // Unknown tags and malformed IPv4 payloads fail closed. + for context in [[(42 << 32) | 3, 0, 0], [(42 << 32) | 4, 1, 0], [6, 1, 2]] { + assert!(matches!( + LocalOutputDisposition::from_context(context, 1500), + LocalOutputDisposition::Drop + )); + } + } +} + #[derive(Debug, Default)] pub(super) struct LocalOutputQueueState { pub(super) packets: VecDeque, @@ -604,7 +682,7 @@ impl LocalInputQueue { pub(super) fn release_resolved_outputs( &self, - mut is_resolved: impl FnMut(smoltcp::wire::Ipv4Address) -> bool, + mut is_resolved: impl FnMut(smoltcp::wire::IpAddress) -> bool, ) { let mut output = self.output.lock(); let LocalOutputQueueState { @@ -615,7 +693,7 @@ impl LocalInputQueue { deferred_routes.release_resolved(&mut is_resolved, packets); } - pub(super) fn release_neighbor(&self, oif: u32, next_hop: smoltcp::wire::Ipv4Address) -> bool { + pub(super) fn release_neighbor(&self, oif: u32, next_hop: smoltcp::wire::IpAddress) -> bool { let mut output = self.output.lock(); let LocalOutputQueueState { packets, diff --git a/kernel/src/driver/net/loopback.rs b/kernel/src/driver/net/loopback.rs index aac4a1507..9d7633939 100644 --- a/kernel/src/driver/net/loopback.rs +++ b/kernel/src/driver/net/loopback.rs @@ -602,7 +602,7 @@ impl Iface for LoopbackInterface { if ip_packet.len() > self.mtu() { return Err(SystemError::EMSGSIZE.into()); } - self.inject_local_ipv4_packet(self.nic_id() as u32, self.mac(), ip_packet, false) + self.inject_local_ip_packet(self.nic_id() as u32, self.mac(), ip_packet, false) .map_err(Into::into) } diff --git a/kernel/src/net/neighbor/mod.rs b/kernel/src/net/neighbor/mod.rs index 9436f35f2..a8cd70116 100644 --- a/kernel/src/net/neighbor/mod.rs +++ b/kernel/src/net/neighbor/mod.rs @@ -114,8 +114,8 @@ pub(crate) fn read(netns: &Arc) -> NeighborReadGuard<'_> { netns.neighbor_table().read() } -pub(crate) fn has_ipv4_entries(netns: &Arc) -> bool { - netns.neighbor_table().has_ipv4_entries() +pub(crate) fn has_ethernet_entries(netns: &Arc) -> bool { + netns.neighbor_table().has_ethernet_entries() } /// Allocation-complete removal of all configured neighbors owned by one @@ -185,9 +185,7 @@ fn release_deferred_neighbor(iface: &Arc, entry: NeighborEntry) { if !entry.ethernet_output { return; } - let IpAddress::Ipv4(next_hop) = entry.destination else { - return; - }; + let next_hop = entry.destination; iface .common() .configured_neighbor_committed(entry.ifindex, next_hop); @@ -199,9 +197,9 @@ pub(crate) fn release_deferred_after_enqueue( netns: &Arc, common: &IfaceCommon, ifindex: u32, - next_hop: smoltcp::wire::Ipv4Address, + next_hop: IpAddress, ) -> bool { - if lookup(netns, ifindex, IpAddress::Ipv4(next_hop)).is_none() { + if lookup(netns, ifindex, next_hop).is_none() { return false; } common.release_configured_neighbor(ifindex, next_hop) diff --git a/kernel/src/net/neighbor/table.rs b/kernel/src/net/neighbor/table.rs index 31c5a8938..0e9dfd87d 100644 --- a/kernel/src/net/neighbor/table.rs +++ b/kernel/src/net/neighbor/table.rs @@ -26,9 +26,9 @@ use super::{ #[derive(Debug)] pub(crate) struct NeighborTable { entries: RwSem>, - /// Lock-free gate for the IPv4 egress path. The exact mapping remains in + /// Lock-free gate for configured Ethernet egress. The exact mapping remains in /// `entries`; this count only decides whether a poll must consult it. - ipv4_entries: AtomicUsize, + ethernet_entries: AtomicUsize, } /// Allocation-free view for packet paths that perform multiple lookups while @@ -92,7 +92,7 @@ impl NeighborTable { pub(crate) fn new() -> Self { Self { entries: RwSem::new(Vec::new()), - ipv4_entries: AtomicUsize::new(0), + ethernet_entries: AtomicUsize::new(0), } } @@ -157,11 +157,11 @@ impl NeighborTable { flags: update.flags, kind: RTN_UNICAST, }; - if entry.ethernet_output && matches!(entry.destination, IpAddress::Ipv4(_)) { + if entry.ethernet_output { // Publish the conservative slow-path gate before the // non-fallible insert, so a packet cannot miss a committed // configured mapping. - self.ipv4_entries.fetch_add(1, AtomicOrdering::Release); + self.ethernet_entries.fetch_add(1, AtomicOrdering::Release); } entries.insert(index, entry); Ok(NeighborMutationOutcome::Added(entry)) @@ -178,8 +178,8 @@ impl NeighborTable { let mut entries = self.entries.write(); let index = find(&entries, ifindex, destination).map_err(|_| SystemError::ENOENT)?; let removed = entries.remove(index); - if removed.ethernet_output && matches!(removed.destination, IpAddress::Ipv4(_)) { - self.ipv4_entries.fetch_sub(1, AtomicOrdering::Release); + if removed.ethernet_output { + self.ethernet_entries.fetch_sub(1, AtomicOrdering::Release); } Ok(removed) } @@ -208,8 +208,8 @@ impl NeighborTable { Ok(NeighborSnapshot { entries: snapshot }) } - pub(super) fn has_ipv4_entries(&self) -> bool { - self.ipv4_entries.load(AtomicOrdering::Acquire) != 0 + pub(super) fn has_ethernet_entries(&self) -> bool { + self.ethernet_entries.load(AtomicOrdering::Acquire) != 0 } pub(super) fn prepare_iface_purge( @@ -258,19 +258,17 @@ impl NeighborTable { entries: &mut Vec, range: core::ops::Range, ) { - let removed_ipv4 = entries[range.clone()] + let removed_ethernet = entries[range.clone()] .iter() - .filter(|entry| { - entry.ethernet_output && matches!(entry.destination, IpAddress::Ipv4(_)) - }) + .filter(|entry| entry.ethernet_output) .count(); entries.drain(range); - if removed_ipv4 != 0 { + if removed_ethernet != 0 { // Keep a stale-positive gate until the mappings are gone. A packet // may take the slow path unnecessarily, but can never miss a // configured mapping that is still published. - self.ipv4_entries - .fetch_sub(removed_ipv4, AtomicOrdering::Release); + self.ethernet_entries + .fetch_sub(removed_ethernet, AtomicOrdering::Release); } } } diff --git a/kernel/src/net/route/fib.rs b/kernel/src/net/route/fib.rs index 59b1e318b..08382d36b 100644 --- a/kernel/src/net/route/fib.rs +++ b/kernel/src/net/route/fib.rs @@ -8,7 +8,7 @@ use super::{ fib_index::{projection_key, AliasPlacement, BroadcastLookup, FibIndex, ProjectionKey}, is_ipv4, RouteDeleteSelector, RouteEntry, RouteLookupResult, RouteMutationOutcome, RouteNewFlags, RouteNotifications, RouteSourcePolicy, RT_SCOPE_LINK, RT_TABLE_DEFAULT, - RT_TABLE_LOCAL, + RT_TABLE_LOCAL, RT_TABLE_MAIN, }; #[derive(Debug, Default, PartialEq, Eq)] @@ -621,9 +621,11 @@ impl FibTable { /// A flat smoltcp LPM projection cannot preserve RPDB table priority when /// a more-specific default-table route overlaps a main-table route. Keep - /// such namespaces on the authoritative IPv4 egress path instead. - pub(in crate::net) fn requires_authoritative_ipv4_output(&self) -> bool { - self.index.has_ipv4_routes(RT_TABLE_DEFAULT) + /// such namespaces on the authoritative egress path instead. IPv6 also + /// needs that path when main-table routing is configured: a socketless + /// ICMPv6 reply may leave through a different interface from its owner. + pub(in crate::net) fn requires_authoritative_output(&self) -> bool { + self.index.has_ipv4_routes(RT_TABLE_DEFAULT) || self.index.has_ipv6_routes(RT_TABLE_MAIN) } pub(super) fn resolve_gateway( diff --git a/kernel/src/net/route/fib_index.rs b/kernel/src/net/route/fib_index.rs index b72c4d06d..d190ccbbe 100644 --- a/kernel/src/net/route/fib_index.rs +++ b/kernel/src/net/route/fib_index.rs @@ -325,6 +325,13 @@ impl FibIndex { }) } + pub(super) fn has_ipv6_routes(&self, table: u32) -> bool { + self.prefix_counts.contains_key(&PrefixDomain { + table, + family: AddressFamily::Ipv6, + }) + } + /// Reserve every bucket needed by a later infallible commit. pub(super) fn prepare_insert(&mut self, route: RouteEntry) -> Result<(), SystemError> { let result = (|| { diff --git a/kernel/src/net/route/mod.rs b/kernel/src/net/route/mod.rs index 25a0348df..9ce0183d3 100644 --- a/kernel/src/net/route/mod.rs +++ b/kernel/src/net/route/mod.rs @@ -29,6 +29,7 @@ pub(crate) use lifecycle::{ commit_addresses, prepare_address_link_change, prepare_link_state_change, register_iface, unregister_iface, AddressLinkChange, PreparedAddressRouteCommit, PreparedLinkStateChange, }; +pub(crate) use source::resolve_ipv6_output_route; pub(crate) use source::{resolve_ipv4_output_flow, resolve_ipv4_route, Ipv4OutputFlow}; use transaction::{ prepare_with_devices, projection_for_iface, transact_single, transact_with_devices, diff --git a/kernel/src/net/route/source.rs b/kernel/src/net/route/source.rs index 6ecbce9ae..4bc4d95af 100644 --- a/kernel/src/net/route/source.rs +++ b/kernel/src/net/route/source.rs @@ -1,4 +1,4 @@ -//! IPv4 output-route and source-address resolution. +//! Output-route and source-address validation for socket and routing callers. //! //! Socket and rtnetlink callers consume this module's result instead of //! independently interpreting FIB source policy. @@ -132,3 +132,45 @@ pub(crate) fn resolve_ipv4_output_flow( source: resolved.source, }) } + +/// Validate an already selected native IPv6 source against the output route. +/// Source selection itself remains in the existing socket source policy. +/// In particular, IPv6 never uses IPv4's explicit-device on-link fallback. +pub(crate) fn resolve_ipv6_output_route( + netns: &Arc, + destination: IpAddress, + required_oif: Option, + fixed_source: Option, +) -> Result { + if !matches!(destination, IpAddress::Ipv6(_)) { + return Err(SystemError::EAFNOSUPPORT); + } + let devices = netns.device_list(); + let router = netns.router(); + let routes = super::lock_output_routes(&router, devices); + let route = routes + .lookup(destination, required_oif) + .ok_or(SystemError::ENETUNREACH)?; + let iface = routes + .devices + .get(&(route.oif as usize)) + .ok_or(SystemError::ENETUNREACH)?; + if route.kind != RTN_LOCAL && !iface.flags().contains(InterfaceFlags::UP) { + return Err(SystemError::ENETDOWN); + } + if let Some(source) = fixed_source { + if !matches!(source, IpAddress::Ipv6(_)) + || !routes.devices.values().any(|candidate| { + crate::net::address::iface_accepts_local_address(candidate, source) + }) + { + return Err(SystemError::EADDRNOTAVAIL); + } + if super::is_ipv6_link_local(source) + && !crate::net::address::iface_accepts_local_address(iface, source) + { + return Err(SystemError::EADDRNOTAVAIL); + } + } + Ok(route) +} diff --git a/kernel/src/net/routing/mod.rs b/kernel/src/net/routing/mod.rs index 21c68b251..90c338b38 100644 --- a/kernel/src/net/routing/mod.rs +++ b/kernel/src/net/routing/mod.rs @@ -40,11 +40,11 @@ pub struct Router { pub(in crate::net) fib: RwSem, pub(self) nat_tracker: Arc, pub ns: RwSem>, - /// Execution mode for IPv4 output. It is derived from the committed FIB + /// Execution mode for namespace-routed output. It is derived from the committed FIB /// at rest and forced on while a writer publishes the FIB's per-interface /// smoltcp projections, making the complete control-plane transaction the /// only observable generation. - authoritative_ipv4_output: AtomicBool, + authoritative_output: AtomicBool, } /// Write access to the namespace FIB. @@ -55,7 +55,7 @@ pub struct Router { /// having to remember a separate generation handoff. pub(in crate::net) struct RouterFibWriteGuard<'a> { fib: RwSemWriteGuard<'a, crate::net::route::FibTable>, - authoritative_ipv4_output: &'a AtomicBool, + authoritative_output: &'a AtomicBool, } impl core::ops::Deref for RouterFibWriteGuard<'_> { @@ -74,10 +74,8 @@ impl core::ops::DerefMut for RouterFibWriteGuard<'_> { impl Drop for RouterFibWriteGuard<'_> { fn drop(&mut self) { - self.authoritative_ipv4_output.store( - self.fib.requires_authoritative_ipv4_output(), - Ordering::Release, - ); + self.authoritative_output + .store(self.fib.requires_authoritative_output(), Ordering::Release); } } @@ -87,7 +85,7 @@ impl Router { fib: RwSem::new(crate::net::route::FibTable::default()), nat_tracker: Arc::new(ConnTracker::default()), ns: RwSem::new(Weak::default()), - authoritative_ipv4_output: AtomicBool::new(false), + authoritative_output: AtomicBool::new(false), }) } @@ -98,7 +96,7 @@ impl Router { fib: RwSem::new(crate::net::route::FibTable::default()), ns: RwSem::new(Weak::default()), nat_tracker: Arc::new(ConnTracker::default()), - authoritative_ipv4_output: AtomicBool::new(false), + authoritative_output: AtomicBool::new(false), }) } @@ -110,16 +108,15 @@ impl Router { // restarts. Consequently no data-plane reader can enter between two // per-interface projection updates. Drop restores the mode derived // from the newly committed FIB before releasing the write lock. - self.authoritative_ipv4_output - .store(true, Ordering::Release); + self.authoritative_output.store(true, Ordering::Release); RouterFibWriteGuard { fib, - authoritative_ipv4_output: &self.authoritative_ipv4_output, + authoritative_output: &self.authoritative_output, } } - pub(crate) fn requires_authoritative_ipv4_output(&self) -> bool { - self.authoritative_ipv4_output.load(Ordering::Acquire) + pub(crate) fn requires_authoritative_output(&self) -> bool { + self.authoritative_output.load(Ordering::Acquire) } pub fn lookup_ingress_route( @@ -215,7 +212,7 @@ pub trait RouterEnableDevice: Iface { return Err(None); } interface - .inject_local_ipv4_packet( + .inject_local_ip_packet( self.nic_id() as u32, ether_frame.src_addr(), ipv4_packet_mut.as_ref(), @@ -229,7 +226,7 @@ pub trait RouterEnableDevice: Iface { return Err(None); } interface - .inject_local_ipv4_packet( + .inject_local_ip_packet( self.nic_id() as u32, ether_frame.src_addr(), ipv4_packet_mut.as_ref(), @@ -287,8 +284,40 @@ pub trait RouterEnableDevice: Iface { Err(None) } smoltcp::wire::EthernetProtocol::Ipv6 => { - log::warn!("IPv6 is not supported yet, ignoring packet"); - Err(None) + let packet = smoltcp::wire::Ipv6Packet::new_checked(ether_frame.payload()) + .map_err(|_| Some(SystemError::EINVAL))?; + let repr = smoltcp::wire::Ipv6Repr::parse(&packet) + .map_err(|_| Some(SystemError::EINVAL))?; + // Scoped destinations and neighbor discovery belong to the + // physical receiving interface, not the address owner. In + // particular a unicast NA must update that interface's cache. + if repr.dst_addr.is_multicast() + || repr.dst_addr.is_unicast_link_local() + || repr.dst_addr.is_loopback() + || repr.dst_addr.is_unspecified() + || ipv6_is_neighbor_discovery(&packet).map_err(Some)? + { + return Err(None); + } + let Some(IngressRouteDecision::Local(interface)) = self + .netns_router() + .lookup_ingress_route(repr.dst_addr.into(), self.nic_id() as u32) + else { + // This path implements host delivery only, not IPv6 + // forwarding. Let the physical stack reject other input. + return Err(None); + }; + if interface.nic_id() == self.nic_id() { + return Err(None); + } + interface + .inject_local_ip_packet( + self.nic_id() as u32, + ether_frame.src_addr(), + ðer_frame.payload()[..40 + repr.payload_len], + false, + ) + .map_err(Some) } _ => { log::warn!( @@ -444,6 +473,55 @@ pub trait RouterEnableDevice: Iface { } } +/// Recognize the IPv6 formats accepted by smoltcp without bypassing its +/// validation of NDP checksums, hop limit, or options. Hop-by-hop is the only +/// extension header currently dispatched by that stack. +fn ipv6_is_neighbor_discovery( + packet: &smoltcp::wire::Ipv6Packet<&[u8]>, +) -> Result { + use smoltcp::wire::{IpProtocol, Ipv6ExtHeader}; + let mut protocol = packet.next_header(); + let mut payload = packet.payload(); + if protocol == IpProtocol::HopByHop { + let header = Ipv6ExtHeader::new_checked(payload).map_err(|_| SystemError::EINVAL)?; + protocol = header.next_header(); + let header_len = (usize::from(header.header_len()) + 1) * 8; + payload = payload.get(header_len..).ok_or(SystemError::EINVAL)?; + } + Ok(protocol == IpProtocol::Icmpv6 + && payload + .first() + .is_some_and(|kind| (133..=137).contains(kind))) +} + +#[cfg(test)] +mod ipv6_input_tests { + use super::ipv6_is_neighbor_discovery; + use smoltcp::wire::Ipv6Packet; + + #[test] + fn neighbor_discovery_is_recognized_with_supported_extension_header() { + // The physical smoltcp interface still validates ICMPv6 and checksum. + let mut bytes = [0u8; 56]; + bytes[0] = 0x60; + bytes[4..6].copy_from_slice(&16u16.to_be_bytes()); + bytes[6] = 0; // Hop-by-hop. + bytes[40] = 58; // ICMPv6 follows the 8-byte extension. + bytes[48] = 136; // A unicast NA is still link-scoped protocol traffic. + assert!(ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).unwrap()); + bytes[48] = 129; // Ordinary echo replies must remain routable. + assert!( + !ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).unwrap() + ); + bytes[6] = 58; + bytes[40] = 135; // Direct NS, no extension. + assert!(ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).unwrap()); + bytes[6] = 0; + bytes[41] = 2; // Truncated extension cannot be indexed unchecked. + assert!(ipv6_is_neighbor_discovery(&Ipv6Packet::new_checked(&bytes[..]).unwrap()).is_err()); + } +} + fn sub_ttl_ipv4(ipv4_packet: &mut Ipv4Packet<&mut Vec>) { let new_ttl = ipv4_packet.hop_limit().saturating_sub(1); ipv4_packet.set_hop_limit(new_ttl); diff --git a/kernel/src/net/socket/inet/common/mod.rs b/kernel/src/net/socket/inet/common/mod.rs index 2d99f2fc6..deb009ee8 100644 --- a/kernel/src/net/socket/inet/common/mod.rs +++ b/kernel/src/net/socket/inet/common/mod.rs @@ -519,21 +519,18 @@ fn get_ephemeral_bind_target( /// Selects the SocketSet owner for a source chosen on `egress_iface`. /// -/// IPv4 local delivery follows the namespace local FIB and may therefore -/// target a different interface from egress. Native IPv6 output is still -/// interface-local, so IPv6 retains the egress stack until the routed-output -/// backend supports that family as well. +/// Local delivery follows the namespace local FIB and may therefore target +/// a different interface from egress, for either IP family. pub(crate) fn ephemeral_target_for_source( netns: &Arc, egress_iface: Arc, local_addr: smoltcp::wire::IpAddress, no_source_error: SystemError, ) -> Result { - let stack_owner = match local_addr { - smoltcp::wire::IpAddress::Ipv4(address) if !address.is_unspecified() => { - crate::net::route::local_address_owner(netns, local_addr).ok_or(no_source_error)? - } - _ => egress_iface, + let stack_owner = if !local_addr.is_unspecified() { + crate::net::route::local_address_owner(netns, local_addr).ok_or(no_source_error)? + } else { + egress_iface }; Ok(EphemeralBindTarget { stack_owner, diff --git a/kernel/src/net/socket/inet/datagram/mod.rs b/kernel/src/net/socket/inet/datagram/mod.rs index 4cc85f2e7..795e999ff 100644 --- a/kernel/src/net/socket/inet/datagram/mod.rs +++ b/kernel/src/net/socket/inet/datagram/mod.rs @@ -1026,6 +1026,28 @@ impl UdpSocket { Some((local, dest, connected_source)) => { let (required_oif, multicast_source) = output_flow::socket_constraints(self, dest.addr)?; + if matches!(dest.addr, smoltcp::wire::IpAddress::Ipv6(_)) + && !dest.addr.is_multicast() + { + let source = local + .addr + .filter(|source| !source.is_unspecified()) + .or(connected_source); + let route = crate::net::route::resolve_ipv6_output_route( + &self.netns, + dest.addr, + required_oif, + source, + )?; + // Native IPv6 source fragmentation is not implemented. + // Report the limit before enqueueing, rather than losing + // an accepted datagram at the physical egress. + if route.kind != crate::net::route::RTN_LOCAL + && buf.len() > route.ip_mtu.saturating_sub(40 + 8) + { + return Err(SystemError::EMSGSIZE); + } + } output_flow::resolve_ipv4_send_flow( &self.netns, local, diff --git a/kernel/src/net/socket/inet/stream/lifecycle.rs b/kernel/src/net/socket/inet/stream/lifecycle.rs index 219c19235..eb7becdfa 100644 --- a/kernel/src/net/socket/inet/stream/lifecycle.rs +++ b/kernel/src/net/socket/inet/stream/lifecycle.rs @@ -349,6 +349,17 @@ impl TcpSocket { // Explicit-source and self-connect paths need the same device/route // validation as implicit binding, before changing the socket state. if let Some(inner::Inner::Init(inner::Init::Bound((_, local, _)))) = writer.as_ref() { + if matches!(local.addr, smoltcp::wire::IpAddress::Ipv6(_)) + && !local.addr.is_unspecified() + { + let device = self.device_binding.resolve_iface(&self.netns)?; + crate::net::route::resolve_ipv6_output_route( + &self.netns, + remote_endpoint.addr, + device.map(|iface| iface.nic_id() as u32), + Some(local.addr), + )?; + } if local.addr.version() == smoltcp::wire::IpVersion::Ipv4 && !local.addr.is_unspecified() { diff --git a/user/apps/tests/dunitest/no_skip.txt b/user/apps/tests/dunitest/no_skip.txt index e73dd9e3f..bbb0fcede 100644 --- a/user/apps/tests/dunitest/no_skip.txt +++ b/user/apps/tests/dunitest/no_skip.txt @@ -26,6 +26,7 @@ normal/tcp_relisten normal/tcp_listener_overflow normal/tcp_accept_handshake normal/tcp_device_binding +normal/ipv6_routed_output normal/socket_nonblocking normal/poll_timeout_semantics diff --git a/user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc b/user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc new file mode 100644 index 000000000..89c862d5a --- /dev/null +++ b/user/apps/tests/dunitest/suites/normal/ipv6_routed_output.cc @@ -0,0 +1,430 @@ +// A synthetic peer observes physical ingress on veth1 for traffic emitted on +// veth2. The TCP source belongs to veth1: socket owner and egress must differ. +// No peer IP is assigned to the host, so the host cannot fake the handshake. +// Linux reference runs must disable veth TX checksum/GSO/GRO offloads first: +// ethtool -K veth{1,2} tx off rx off tso off gso off gro off (each device). +// Otherwise AF_PACKET exposes unfinished checksums and aggregated segments. +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include +#include + +namespace { +class Fd { + public: + explicit Fd(int value = -1) : value_(value) {} + ~Fd() { reset(-1); } + Fd(const Fd&) = delete; + Fd& operator=(const Fd&) = delete; + int get() const { return value_; } + void reset(int value) { if (value_ >= 0) close(value_); value_ = value; } + private: + int value_; +}; + +uint16_t Get16(const unsigned char* p) { return (uint16_t(p[0]) << 8) | p[1]; } +uint32_t Get32(const unsigned char* p) { return (uint32_t(Get16(p)) << 16) | Get16(p + 2); } +void Put16(unsigned char* p, uint16_t n) { p[0] = n >> 8; p[1] = n; } +void Put32(unsigned char* p, uint32_t n) { Put16(p, n >> 16); Put16(p + 2, n); } +uint32_t Sum(const unsigned char* p, size_t n) { + uint32_t sum = 0; + for (size_t i = 0; i < n; i += 2) + sum += (uint16_t(p[i]) << 8) | (i + 1 < n ? p[i + 1] : 0); + return sum; +} +uint16_t Checksum(const unsigned char* ip, const unsigned char* tcp, size_t n, + uint8_t protocol = IPPROTO_TCP) { + uint32_t sum = Sum(ip + 8, 32) + n + protocol + Sum(tcp, n); + while (sum >> 16) sum = (sum & 0xffff) + (sum >> 16); + return ~sum; +} + +struct Request { + nlmsghdr header{}; + std::array body{}; + Request(uint16_t type, size_t size, uint16_t flags = 0) { + header.nlmsg_len = NLMSG_LENGTH(size); + header.nlmsg_type = type; + header.nlmsg_flags = NLM_F_REQUEST | NLM_F_ACK | flags; + } + void Attribute(uint16_t type, const void* data, size_t size) { + size_t offset = NLMSG_ALIGN(header.nlmsg_len); + ASSERT_LE(offset + RTA_SPACE(size), sizeof(*this)); + auto* attr = reinterpret_cast(reinterpret_cast(this) + offset); + attr->rta_type = type; + attr->rta_len = RTA_LENGTH(size); + memcpy(RTA_DATA(attr), data, size); + header.nlmsg_len = offset + RTA_SPACE(size); + } +}; + +class Ipv6RoutedOutput : public testing::Test { + protected: + Fd netlink_, packet_, client_; + unsigned int owner_ = 0, egress_ = 0, sequence_ = 0; + in6_addr source_{}, egress_address_{}, peer_{}; + std::array peer_mac_{}, egress_mac_{}; + bool source_added_ = false, egress_added_ = false, neighbor_added_ = false; + bool dynamic_neighbor_ = false, source_neighbor_ = false; + int original_mtu_ = 0; + unsigned int solicitations_ = 0; + uint16_t port_ = 0; + uint32_t client_next_ = 0; + static constexpr uint16_t kPeerPort = 19876; + static constexpr uint32_t kPeerSequence = 1234567; + + int Submit(Request& request) { + request.header.nlmsg_seq = ++sequence_; + sockaddr_nl kernel{}; + kernel.nl_family = AF_NETLINK; + if (sendto(netlink_.get(), &request, request.header.nlmsg_len, 0, + reinterpret_cast(&kernel), sizeof(kernel)) < 0) return -1; + for (;;) { + alignas(nlmsghdr) unsigned char reply[8192]; + ssize_t size = recv(netlink_.get(), reply, sizeof(reply), 0); + if (size < 0) { if (errno == EINTR) continue; return -1; } + for (auto* h = reinterpret_cast(reply); NLMSG_OK(h, size); + h = NLMSG_NEXT(h, size)) { + if (h->nlmsg_seq != sequence_ || h->nlmsg_type != NLMSG_ERROR) continue; + if (h->nlmsg_len < NLMSG_LENGTH(sizeof(nlmsgerr))) { errno = EPROTO; return -1; } + int error = reinterpret_cast(NLMSG_DATA(h))->error; + if (error) { errno = -error; return -1; } + return 0; + } + } + } + int Address(unsigned int index, const in6_addr& address, bool add) { + Request request(add ? RTM_NEWADDR : RTM_DELADDR, sizeof(ifaddrmsg), + add ? NLM_F_CREATE | NLM_F_EXCL : 0); + auto* msg = reinterpret_cast(NLMSG_DATA(&request.header)); + msg->ifa_family = AF_INET6; msg->ifa_prefixlen = 64; + msg->ifa_flags = IFA_F_NODAD; msg->ifa_index = index; + request.Attribute(IFA_ADDRESS, &address, sizeof(address)); + return Submit(request); + } + int Neighbor(bool add, const in6_addr* target = nullptr) { + Request request(add ? RTM_NEWNEIGH : RTM_DELNEIGH, sizeof(ndmsg), + add ? NLM_F_CREATE | NLM_F_EXCL : 0); + auto* msg = reinterpret_cast(NLMSG_DATA(&request.header)); + msg->ndm_family = AF_INET6; msg->ndm_ifindex = egress_; + msg->ndm_state = NUD_PERMANENT; msg->ndm_type = RTN_UNICAST; + request.Attribute(NDA_DST, target ? target : &peer_, sizeof(peer_)); + if (add) request.Attribute(NDA_LLADDR, peer_mac_.data(), peer_mac_.size()); + return Submit(request); + } + int SetMtu(int mtu) { + Request request(RTM_NEWLINK, sizeof(ifinfomsg)); + reinterpret_cast(NLMSG_DATA(&request.header))->ifi_index = egress_; + request.Attribute(IFLA_MTU, &mtu, sizeof(mtu)); + return Submit(request); + } + void SetUp() override { + owner_ = if_nametoindex("veth1"); egress_ = if_nametoindex("veth2"); + ASSERT_NE(owner_, 0u) << "requires the standard veth1/veth2 fixture"; + ASSERT_NE(egress_, 0u); + ASSERT_EQ(inet_pton(AF_INET6, "fd10:91::1", &source_), 1); + ASSERT_EQ(inet_pton(AF_INET6, "fd10:92::2", &egress_address_), 1); + ASSERT_EQ(inet_pton(AF_INET6, "fd10:92::99", &peer_), 1); + netlink_.reset(socket(AF_NETLINK, SOCK_RAW, NETLINK_ROUTE)); + ASSERT_GE(netlink_.get(), 0); + sockaddr_nl local{}; local.nl_family = AF_NETLINK; + ASSERT_EQ(bind(netlink_.get(), reinterpret_cast(&local), sizeof(local)), 0); + timeval timeout{5, 0}; + ASSERT_EQ(setsockopt(netlink_.get(), SOL_SOCKET, SO_RCVTIMEO, &timeout, sizeof(timeout)), 0); + ASSERT_EQ(Address(owner_, source_, true), 0) << strerror(errno); source_added_ = true; + ASSERT_EQ(Address(egress_, egress_address_, true), 0) << strerror(errno); egress_added_ = true; + packet_.reset(socket(AF_PACKET, SOCK_RAW | SOCK_NONBLOCK, htons(ETH_P_IPV6))); + ASSERT_GE(packet_.get(), 0); + ifreq request{}; + strcpy(request.ifr_name, "veth1"); + ASSERT_EQ(ioctl(packet_.get(), SIOCGIFHWADDR, &request), 0); + memcpy(peer_mac_.data(), request.ifr_hwaddr.sa_data, 6); + strcpy(request.ifr_name, "veth2"); + ASSERT_EQ(ioctl(packet_.get(), SIOCGIFHWADDR, &request), 0); + memcpy(egress_mac_.data(), request.ifr_hwaddr.sa_data, 6); + ASSERT_EQ(ioctl(packet_.get(), SIOCGIFMTU, &request), 0); + original_mtu_ = request.ifr_mtu; + sockaddr_ll link{}; link.sll_family = AF_PACKET; + link.sll_protocol = htons(ETH_P_IPV6); link.sll_ifindex = owner_; + ASSERT_EQ(bind(packet_.get(), reinterpret_cast(&link), sizeof(link)), 0); + ASSERT_EQ(Neighbor(true), 0) << strerror(errno); neighbor_added_ = true; + } + void TearDown() override { + client_.reset(-1); + if (neighbor_added_) { EXPECT_EQ(Neighbor(false), 0) << strerror(errno); } + if (dynamic_neighbor_) { + int result = Neighbor(false); + EXPECT_TRUE(result == 0 || errno == ENOENT) << strerror(errno); + } + if (source_neighbor_) { + int result = Neighbor(false, &source_); + EXPECT_TRUE(result == 0 || errno == ENOENT) << strerror(errno); + } + if (original_mtu_) { EXPECT_EQ(SetMtu(original_mtu_), 0) << strerror(errno); } + if (egress_added_) { EXPECT_EQ(Address(egress_, egress_address_, false), 0) << strerror(errno); } + if (source_added_) { EXPECT_EQ(Address(owner_, source_, false), 0) << strerror(errno); } + } + + void SendFrame(const std::vector& frame) { + sockaddr_ll link{}; link.sll_family = AF_PACKET; link.sll_ifindex = owner_; + link.sll_halen = 6; memcpy(link.sll_addr, egress_mac_.data(), 6); + ASSERT_EQ(sendto(packet_.get(), frame.data(), frame.size(), 0, + reinterpret_cast(&link), sizeof(link)), + static_cast(frame.size())) << strerror(errno); + } + void Ndisc(uint8_t type, const in6_addr& source, const in6_addr& destination, + const in6_addr& target) { + std::vector frame(14 + 40 + 32); + memcpy(frame.data(), egress_mac_.data(), 6); + memcpy(frame.data() + 6, peer_mac_.data(), 6); + Put16(frame.data() + 12, ETH_P_IPV6); + auto* ip = frame.data() + 14; + ip[0] = 0x60; Put16(ip + 4, 32); ip[6] = IPPROTO_ICMPV6; ip[7] = 255; + memcpy(ip + 8, &source, 16); memcpy(ip + 24, &destination, 16); + auto* icmp = ip + 40; + icmp[0] = type; icmp[4] = type == 136 ? 0x60 : 0; + memcpy(icmp + 8, &target, 16); + icmp[24] = type == 136 ? 2 : 1; icmp[25] = 1; + memcpy(icmp + 26, peer_mac_.data(), 6); + Put16(icmp + 2, Checksum(ip, icmp, 32, IPPROTO_ICMPV6)); + SendFrame(frame); + } + void Segment(uint8_t flags, uint32_t sequence, uint32_t ack, + const unsigned char* payload = nullptr, size_t length = 0) { + // Advertise a large MSS so the peer cannot accidentally mask egress MTU. + const size_t tcp_header = (flags & 2) ? 24 : 20; + std::vector frame(14 + 40 + tcp_header + length); + memcpy(frame.data(), egress_mac_.data(), 6); + memcpy(frame.data() + 6, peer_mac_.data(), 6); + Put16(frame.data() + 12, ETH_P_IPV6); + auto* ip = frame.data() + 14; + ip[0] = 0x60; Put16(ip + 4, tcp_header + length); ip[6] = IPPROTO_TCP; ip[7] = 64; + memcpy(ip + 8, &peer_, 16); memcpy(ip + 24, &source_, 16); + auto* tcp = ip + 40; + Put16(tcp, kPeerPort); Put16(tcp + 2, port_); + Put32(tcp + 4, sequence); Put32(tcp + 8, ack); + tcp[12] = (tcp_header / 4) << 4; tcp[13] = flags; Put16(tcp + 14, 65535); + if (flags & 2) { tcp[20] = 2; tcp[21] = 4; Put16(tcp + 22, 1460); } + if (length) memcpy(tcp + tcp_header, payload, length); + Put16(tcp + 16, Checksum(ip, tcp, tcp_header + length)); + SendFrame(frame); + } + bool Receive(std::vector& frame, int timeout_ms) { + auto deadline = std::chrono::steady_clock::now() + std::chrono::milliseconds(timeout_ms); + while (std::chrono::steady_clock::now() < deadline) { + pollfd event{packet_.get(), POLLIN, 0}; + if (poll(&event, 1, 50) < 0) { ADD_FAILURE() << strerror(errno); return false; } + std::array bytes{}; + sockaddr_ll link{}; socklen_t size = sizeof(link); + ssize_t n = recvfrom(packet_.get(), bytes.data(), bytes.size(), 0, + reinterpret_cast(&link), &size); + if (n < 74 || Get16(bytes.data() + 12) != ETH_P_IPV6) continue; + const auto* ip = bytes.data() + 14; + if (n >= 86 && ip[6] == IPPROTO_ICMPV6 && ip[40] == 135 && + !memcmp(ip + 48, &peer_, 16) && link.sll_pkttype != PACKET_OUTGOING) { + EXPECT_EQ(ip[7], 255); + in6_addr destination; + memcpy(&destination, ip + 8, 16); + Ndisc(136, peer_, destination, peer_); + ++solicitations_; + continue; + } + if (ip[6] != IPPROTO_TCP || memcmp(ip + 8, &source_, 16) || + memcmp(ip + 24, &peer_, 16)) continue; + const auto* tcp = ip + 40; + if (Get16(tcp) != port_ || Get16(tcp + 2) != kPeerPort) continue; + EXPECT_EQ(link.sll_pkttype, PACKET_HOST) << "must arrive from veth2, not owner output"; + if (link.sll_pkttype != PACKET_HOST) continue; + size_t payload_size = Get16(ip + 4); + if (payload_size < 20 || n < static_cast(54 + payload_size) || + (tcp[12] >> 4) * 4u < 20 || (tcp[12] >> 4) * 4u > payload_size) { + ADD_FAILURE() << "malformed TCP packet"; return false; + } + // RX checksum offload is not used by the synthetic peer. + EXPECT_EQ(Checksum(ip, tcp, payload_size), 0); + frame.assign(bytes.begin(), bytes.begin() + 54 + payload_size); + return true; + } + return false; + } + void Connect(bool bind_device, bool reuse_socket = false) { + sockaddr_in6 local{}; local.sin6_family = AF_INET6; local.sin6_addr = source_; + if (!reuse_socket) { + client_.reset(socket(AF_INET6, SOCK_STREAM | SOCK_NONBLOCK, 0)); + ASSERT_GE(client_.get(), 0); + ASSERT_EQ(bind(client_.get(), reinterpret_cast(&local), sizeof(local)), 0); + } + if (bind_device) { + const char device[] = "veth2"; + ASSERT_EQ(setsockopt(client_.get(), SOL_SOCKET, SO_BINDTODEVICE, device, sizeof(device)), 0); + } + sockaddr_in6 remote{}; remote.sin6_family = AF_INET6; + remote.sin6_addr = peer_; remote.sin6_port = htons(kPeerPort); + ASSERT_EQ(connect(client_.get(), reinterpret_cast(&remote), sizeof(remote)), -1); + ASSERT_EQ(errno, EINPROGRESS); + socklen_t size = sizeof(local); + ASSERT_EQ(getsockname(client_.get(), reinterpret_cast(&local), &size), 0); + port_ = ntohs(local.sin6_port); + std::vector frame; + ASSERT_TRUE(Receive(frame, 4000)) << "no routed SYN"; + ASSERT_EQ(frame[67] & 0x12, 2); + client_next_ = Get32(frame.data() + 58) + 1; + ASSERT_NO_FATAL_FAILURE(Segment(0x12, kPeerSequence, client_next_)); + pollfd event{client_.get(), POLLOUT, 0}; + ASSERT_EQ(poll(&event, 1, 4000), 1); + int error = -1; size = sizeof(error); + ASSERT_EQ(getsockopt(client_.get(), SOL_SOCKET, SO_ERROR, &error, &size), 0); + ASSERT_EQ(error, 0); + } + void Transfer(size_t length, size_t mtu) { + std::vector data(length); + for (size_t i = 0; i < length; ++i) data[i] = (i * 17 + 9) % 251; + ASSERT_EQ(send(client_.get(), data.data(), data.size(), MSG_NOSIGNAL), + static_cast(data.size())) << strerror(errno); + size_t received = 0; + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(8); + while (received < length && std::chrono::steady_clock::now() < deadline) { + std::vector frame; + ASSERT_TRUE(Receive(frame, 2000)) << "TCP data stalled at " << received; + ASSERT_LE(frame.size() - 14, mtu); + const auto* tcp = frame.data() + 54; + size_t header = (tcp[12] >> 4) * 4; + size_t count = frame.size() - 54 - header; + if (!count) continue; + uint32_t sequence = Get32(tcp + 4); + if (sequence == client_next_) { + ASSERT_LE(received + count, length); + EXPECT_EQ(memcmp(tcp + header, data.data() + received, count), 0); + received += count; client_next_ += count; + } + ASSERT_NO_FATAL_FAILURE(Segment(0x10, kPeerSequence + 1, client_next_)); + } + ASSERT_EQ(received, length); + // A reply arrives physically on veth2 but belongs to veth1's socket. + const unsigned char reply[] = "owner-handoff"; + ASSERT_NO_FATAL_FAILURE(Segment(0x18, kPeerSequence + 1, client_next_, reply, sizeof(reply))); + pollfd event{client_.get(), POLLIN, 0}; + ASSERT_EQ(poll(&event, 1, 4000), 1); + unsigned char buffer[sizeof(reply)]{}; + ASSERT_EQ(recv(client_.get(), buffer, sizeof(buffer), MSG_WAITALL), + static_cast(sizeof(reply))); + EXPECT_EQ(memcmp(buffer, reply, sizeof(reply)), 0); + ASSERT_NO_FATAL_FAILURE(Segment(0x14, kPeerSequence + 1 + sizeof(reply), client_next_)); + } +}; + +TEST_F(Ipv6RoutedOutput, ExplicitSourceUsesFibEgressAndReceivesOnOwner) { + ASSERT_NO_FATAL_FAILURE(Connect(false)); + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, BoundDeviceUsesEgressAndReceivesOnOwner) { + ASSERT_NO_FATAL_FAILURE(Connect(true)); + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, LargeTcpTransferUsesSmallerEgressMtu) { + ASSERT_EQ(SetMtu(1280), 0) << strerror(errno); + ASSERT_NO_FATAL_FAILURE(Connect(true)); + ASSERT_NO_FATAL_FAILURE(Transfer(8192, 1280)); +} +TEST_F(Ipv6RoutedOutput, ColdNeighborDiscoveryReleasesQueuedTcp) { + ASSERT_EQ(Neighbor(false), 0) << strerror(errno); + neighbor_added_ = false; + dynamic_neighbor_ = true; + ASSERT_NO_FATAL_FAILURE(Connect(true)); + ASSERT_GT(solicitations_, 0u) << "must exercise NDP, not a warm cache"; + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, UdpMtuSizedPayloadUsesPhysicalEgress) { + ASSERT_EQ(SetMtu(1280), 0) << strerror(errno); + client_.reset(socket(AF_INET6, SOCK_DGRAM, 0)); + ASSERT_GE(client_.get(), 0); + sockaddr_in6 local{}; local.sin6_family = AF_INET6; local.sin6_addr = source_; + ASSERT_EQ(bind(client_.get(), reinterpret_cast(&local), sizeof(local)), 0); + const char device[] = "veth2"; + ASSERT_EQ(setsockopt(client_.get(), SOL_SOCKET, SO_BINDTODEVICE, device, sizeof(device)), 0); + sockaddr_in6 remote{}; remote.sin6_family = AF_INET6; + remote.sin6_addr = peer_; remote.sin6_port = htons(kPeerPort); + std::vector payload(1232, 0x5a); + ASSERT_EQ(sendto(client_.get(), payload.data(), payload.size(), 0, + reinterpret_cast(&remote), sizeof(remote)), 1232); + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(4); + while (std::chrono::steady_clock::now() < deadline) { + pollfd event{packet_.get(), POLLIN, 0}; + ASSERT_GE(poll(&event, 1, 50), 0); + unsigned char frame[2048]; sockaddr_ll link{}; socklen_t size = sizeof(link); + ssize_t n = recvfrom(packet_.get(), frame, sizeof(frame), 0, + reinterpret_cast(&link), &size); + if (n < 62 || Get16(frame + 12) != ETH_P_IPV6) continue; + const auto* ip = frame + 14; + if (ip[6] != IPPROTO_UDP || memcmp(ip + 8, &source_, 16) || + memcmp(ip + 24, &peer_, 16) || Get16(ip + 42) != kPeerPort) continue; + ASSERT_EQ(link.sll_pkttype, PACKET_HOST); + ASSERT_EQ(n, 14 + 1280); + ASSERT_EQ(Get16(ip + 4), 1240); + ASSERT_EQ(Get16(ip + 44), 1240); + ASSERT_EQ(Checksum(ip, ip + 40, 1240, IPPROTO_UDP), 0); + ASSERT_EQ(memcmp(ip + 48, payload.data(), payload.size()), 0); + return; + } + FAIL() << "MTU-sized UDP datagram did not arrive physically from veth2"; +} +TEST_F(Ipv6RoutedOutput, MissingBoundDeviceRouteDoesNotConsumeSocket) { + client_.reset(socket(AF_INET6, SOCK_STREAM | SOCK_NONBLOCK, 0)); + ASSERT_GE(client_.get(), 0); + sockaddr_in6 local{}; local.sin6_family = AF_INET6; local.sin6_addr = source_; + ASSERT_EQ(bind(client_.get(), reinterpret_cast(&local), sizeof(local)), 0); + const char wrong_device[] = "veth1"; + ASSERT_EQ(setsockopt(client_.get(), SOL_SOCKET, SO_BINDTODEVICE, + wrong_device, sizeof(wrong_device)), 0); + sockaddr_in6 remote{}; remote.sin6_family = AF_INET6; + remote.sin6_addr = peer_; remote.sin6_port = htons(kPeerPort); + ASSERT_EQ(connect(client_.get(), reinterpret_cast(&remote), sizeof(remote)), -1); + ASSERT_EQ(errno, ENETUNREACH); + ASSERT_NO_FATAL_FAILURE(Connect(true, true)); + ASSERT_NO_FATAL_FAILURE(Transfer(256, original_mtu_)); +} +TEST_F(Ipv6RoutedOutput, GlobalSourceNeighborAdvertisementStaysOnIngressInterface) { + // A global source owned by the other interface must not reroute the NDP + // reply through RTN_LOCAL. NDP is always scoped to the receiving link. + source_neighbor_ = true; + ASSERT_NO_FATAL_FAILURE(Ndisc(135, source_, egress_address_, egress_address_)); + auto deadline = std::chrono::steady_clock::now() + std::chrono::seconds(4); + while (std::chrono::steady_clock::now() < deadline) { + pollfd event{packet_.get(), POLLIN, 0}; + ASSERT_GE(poll(&event, 1, 50), 0); + unsigned char frame[2048]; + sockaddr_ll link{}; socklen_t size = sizeof(link); + ssize_t n = recvfrom(packet_.get(), frame, sizeof(frame), 0, + reinterpret_cast(&link), &size); + if (n < 78 || Get16(frame + 12) != ETH_P_IPV6) continue; + const auto* ip = frame + 14; + if (ip[6] != IPPROTO_ICMPV6 || ip[40] != 136 || + memcmp(ip + 8, &egress_address_, 16) || memcmp(ip + 24, &source_, 16)) continue; + ASSERT_EQ(link.sll_pkttype, PACKET_HOST); + ASSERT_EQ(ip[7], 255); + ASSERT_LE(54 + Get16(ip + 4), n); + ASSERT_EQ(Checksum(ip, ip + 40, Get16(ip + 4), IPPROTO_ICMPV6), 0); + return; + } + FAIL() << "NDP response was not emitted back through physical ingress veth2"; +} +} // namespace + +int main(int argc, char** argv) { + testing::InitGoogleTest(&argc, argv); + return RUN_ALL_TESTS(); +} diff --git a/user/apps/tests/dunitest/whitelist.txt b/user/apps/tests/dunitest/whitelist.txt index 29d76b6a1..4c35c6926 100644 --- a/user/apps/tests/dunitest/whitelist.txt +++ b/user/apps/tests/dunitest/whitelist.txt @@ -61,6 +61,7 @@ normal/tcp_relisten normal/tcp_listener_overflow normal/tcp_accept_handshake normal/tcp_device_binding +normal/ipv6_routed_output normal/socket_nonblocking normal/tcp_self_connect_semantics normal/poll_timeout_semantics From d22e43070098563da9f1877939e557240bc7b10d Mon Sep 17 00:00:00 2001 From: longjin Date: Tue, 22 Sep 2026 15:40:35 +0000 Subject: [PATCH 2/2] chore(deps): pin merged smoltcp IPv6 routing revision Update the manifest and lockfile to afde7359455a5b9feacd2b8684994232acbf8903 after smoltcp PR #32 merged into dragonos/v0.12.0. Verified that the merged commit has the same Git tree as the previously validated 4cfe0dd revision. No protocol implementation changes are introduced. make kernel passed with the merged dependency. Signed-off-by: longjin --- kernel/Cargo.lock | 2 +- kernel/Cargo.toml | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/kernel/Cargo.lock b/kernel/Cargo.lock index ec385f3f8..647e8d0be 100644 --- a/kernel/Cargo.lock +++ b/kernel/Cargo.lock @@ -1561,7 +1561,7 @@ checksum = "7fcf8323ef1faaee30a44a340193b1ac6814fd9b7b4e88e9d4519a3e4abe1cfd" [[package]] name = "smoltcp" version = "0.12.0" -source = "git+https://github.com/DragonOS-Community/smoltcp?rev=4cfe0ddadd6c0f5a82a72245bf18c4df6f4bb8fd#4cfe0ddadd6c0f5a82a72245bf18c4df6f4bb8fd" +source = "git+https://github.com/DragonOS-Community/smoltcp?rev=afde7359455a5b9feacd2b8684994232acbf8903#afde7359455a5b9feacd2b8684994232acbf8903" dependencies = [ "bitflags 1.3.2", "byteorder", diff --git a/kernel/Cargo.toml b/kernel/Cargo.toml index ef268d558..21bf0ea11 100644 --- a/kernel/Cargo.toml +++ b/kernel/Cargo.toml @@ -62,7 +62,7 @@ linkme = "=0.3.27" num = { version = "=0.4.0", default-features = false } num-derive = "=0.3" num-traits = { git = "https://git.mirrors.dragonos.org.cn/DragonOS-Community/num-traits.git", rev = "1597c1c", default-features = false } -smoltcp = { version = "=0.12.0", git = "https://github.com/DragonOS-Community/smoltcp", rev = "4cfe0ddadd6c0f5a82a72245bf18c4df6f4bb8fd", default-features = false, features = [ +smoltcp = { version = "=0.12.0", git = "https://github.com/DragonOS-Community/smoltcp", rev = "afde7359455a5b9feacd2b8684994232acbf8903", default-features = false, features = [ "alloc", "medium-ethernet", "socket-raw",