From 5dcb3d7922e86eeeed50d4a3222ad75e4a7dadf4 Mon Sep 17 00:00:00 2001 From: Nicolas Kagami Date: Thu, 17 Sep 2026 11:03:13 -0300 Subject: [PATCH 1/2] RFD 662: per-router V2B tables with per-port prioritized router lists (API 42) --- Cargo.lock | 6 +- Cargo.toml | 2 +- bin/opteadm/Cargo.toml | 1 + bin/opteadm/src/bin/opteadm.rs | 123 +++++++- crates/opte-api/Cargo.toml | 1 + crates/opte-api/src/cmd.rs | 8 + crates/opte-api/src/ip.rs | 24 ++ crates/opte-api/src/lib.rs | 2 +- lib/opte-ioctl/src/lib.rs | 21 ++ lib/opte-test-utils/src/lib.rs | 13 +- lib/oxide-vpc/src/api.rs | 116 ++++++- lib/oxide-vpc/src/engine/mod.rs | 11 +- lib/oxide-vpc/src/engine/overlay.rs | 468 +++++++++++++++++++++++----- lib/oxide-vpc/src/print.rs | 40 ++- xde/src/xde.rs | 65 +++- 15 files changed, 789 insertions(+), 112 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index f49773b1..f0cb2853 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1237,6 +1237,7 @@ dependencies = [ "illumos-sys-hdrs", "ingot", "ipnetwork", + "poptrie", "postcard", "serde", "smoltcp", @@ -1312,6 +1313,7 @@ dependencies = [ "serde", "tabwriter", "thiserror 2.0.18", + "uuid", ] [[package]] @@ -1447,8 +1449,8 @@ dependencies = [ [[package]] name = "poptrie" -version = "0.1.0" -source = "git+https://github.com/oxidecomputer/poptrie?branch=main#5bf62f6b889c61e0608d8463ed11da28e130cb34" +version = "0.2.0" +source = "git+https://github.com/nicolaskagami/poptrie?branch=main#55fadabd7eb12ef4306f16a8812d5b4ed99dcc91" [[package]] name = "postcard" diff --git a/Cargo.toml b/Cargo.toml index 08cdc498..90487f61 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -81,7 +81,7 @@ version_check = "0.9" zerocopy = { version = "0.8", features = ["derive"] } zone = { git = "https://github.com/oxidecomputer/zone" } ztest = { git = "https://github.com/oxidecomputer/falcon", branch = "main" } -poptrie = { git = "https://github.com/oxidecomputer/poptrie", branch = "main" } +poptrie = { git = "https://github.com/nicolaskagami/poptrie", branch = "main" } [profile.dev] opt-level = 1 diff --git a/bin/opteadm/Cargo.toml b/bin/opteadm/Cargo.toml index c0e5ac19..d11a7fd6 100644 --- a/bin/opteadm/Cargo.toml +++ b/bin/opteadm/Cargo.toml @@ -23,6 +23,7 @@ postcard.workspace = true serde.workspace = true tabwriter.workspace = true thiserror.workspace = true +uuid.workspace = true [build-dependencies] anyhow.workspace = true diff --git a/bin/opteadm/src/bin/opteadm.rs b/bin/opteadm/src/bin/opteadm.rs index 39398ca8..b81be543 100644 --- a/bin/opteadm/src/bin/opteadm.rs +++ b/bin/opteadm/src/bin/opteadm.rs @@ -36,6 +36,7 @@ use oxide_vpc::api::DEFAULT_MULTICAST_VNI; use oxide_vpc::api::DelRouterEntryReq; use oxide_vpc::api::DelRouterEntryResp; use oxide_vpc::api::DhcpCfg; +use oxide_vpc::api::DumpRouterListReq; use oxide_vpc::api::ExternalIpCfg; use oxide_vpc::api::Filters as FirewallFilters; use oxide_vpc::api::FirewallAction; @@ -56,6 +57,8 @@ use oxide_vpc::api::RemFwRuleReq; use oxide_vpc::api::RemoveCidrResp; use oxide_vpc::api::Replication; use oxide_vpc::api::RouterClass; +use oxide_vpc::api::RouterId; +use oxide_vpc::api::RouterList; use oxide_vpc::api::RouterTarget; use oxide_vpc::api::SNat4Cfg; use oxide_vpc::api::SNat6Cfg; @@ -63,6 +66,7 @@ use oxide_vpc::api::SetExternalIpsReq; use oxide_vpc::api::SetFwRulesReq; use oxide_vpc::api::SetMcast2PhysReq; use oxide_vpc::api::SetMcastForwardingReq; +use oxide_vpc::api::SetRouterListReq; use oxide_vpc::api::SetVirt2BoundaryReq; use oxide_vpc::api::SetVirt2PhysReq; use oxide_vpc::api::SourceFilter; @@ -78,6 +82,7 @@ use std::io; use std::io::Write; use std::str::FromStr; use tabwriter::TabWriter; +use uuid::Uuid; /// Administer the Oxide Packet Transformation Engine (OPTE) #[derive(Debug, Parser)] @@ -240,10 +245,39 @@ enum Command { }, /// Set a virtual-to-boundary mapping - SetV2B { prefix: IpCidr, tunnel_endpoint: Vec }, + SetV2B { + prefix: IpCidr, + tunnel_endpoint: Vec, + /// The router whose table to modify (default router if omitted) + #[arg(long)] + router_id: Option, + }, /// Clear a virtual-to-boundary mapping - ClearV2B { prefix: IpCidr, tunnel_endpoint: Vec }, + ClearV2B { + prefix: IpCidr, + tunnel_endpoint: Vec, + /// The router whose table to modify (default router if omitted) + #[arg(long)] + router_id: Option, + }, + + /// Replace a port's prioritized router list for boundary TEP + /// selection. + /// + /// Entries are `=` or `=default`, + /// e.g. `set-router-list -p opte0 10=6fe93b02-… 1000=default`. + SetRouterList { + #[arg(short)] + port: String, + entries: Vec, + }, + + /// Show a port's prioritized router list. + DumpRouterList { + #[arg(short)] + port: String, + }, /// Set a multicast-to-physical (M2P) mapping /// @@ -470,6 +504,35 @@ impl From for FirewallFilters { } } +/// One router-list entry: `=` or +/// `=default`. +#[derive(Clone, Debug)] +struct RouterListEntryArg { + priority: u16, + router: RouterId, +} + +impl FromStr for RouterListEntryArg { + type Err = String; + + fn from_str(s: &str) -> Result { + let (prio, rtr) = s.split_once('=').ok_or_else(|| { + format!("expected =, got {s}") + })?; + let priority = prio + .parse::() + .map_err(|e| format!("bad priority {prio}: {e}"))?; + let router = match rtr { + "default" => None, + uuid => Some( + uuid.parse::() + .map_err(|e| format!("bad router id {uuid}: {e}"))?, + ), + }; + Ok(Self { priority, router }) + } +} + #[derive(Debug, Parser)] struct RouterRule { /// The OPTE port to which the route change is applied. @@ -912,7 +975,7 @@ fn main() -> anyhow::Result<()> { hdl.clear_v2p(&req)?; } - Command::SetV2B { prefix, tunnel_endpoint } => { + Command::SetV2B { prefix, tunnel_endpoint, router_id } => { let tep = tunnel_endpoint .into_iter() .map(|ip| TunnelEndpoint { @@ -920,11 +983,11 @@ fn main() -> anyhow::Result<()> { vni: Vni::new(BOUNDARY_SERVICES_VNI).unwrap(), }) .collect(); - let req = SetVirt2BoundaryReq { vip: prefix, tep }; + let req = SetVirt2BoundaryReq { router_id, vip: prefix, tep }; hdl.set_v2b(&req)?; } - Command::ClearV2B { prefix, tunnel_endpoint } => { + Command::ClearV2B { prefix, tunnel_endpoint, router_id } => { let tep = tunnel_endpoint .into_iter() .map(|ip| TunnelEndpoint { @@ -932,10 +995,32 @@ fn main() -> anyhow::Result<()> { vni: Vni::new(BOUNDARY_SERVICES_VNI).unwrap(), }) .collect(); - let req = ClearVirt2BoundaryReq { vip: prefix, tep }; + let req = ClearVirt2BoundaryReq { router_id, vip: prefix, tep }; hdl.clear_v2b(&req)?; } + Command::SetRouterList { port, entries } => { + let list = RouterList::new( + entries.into_iter().map(|e| (e.priority, e.router)), + ) + .map_err(|e| anyhow::anyhow!("invalid router list: {e}"))?; + let req = SetRouterListReq { port_name: port, list }; + hdl.set_router_list(&req)?; + } + + Command::DumpRouterList { port } => { + let resp = + hdl.dump_router_list(&DumpRouterListReq { port_name: port })?; + println!("PRIORITY\tROUTER"); + for (prio, rtr) in resp.list.entries() { + let rtr = match rtr { + None => "default".to_string(), + Some(id) => id.to_string(), + }; + println!("{prio}\t{rtr}"); + } + } + Command::SetM2P { group, underlay } => { let req = SetMcast2PhysReq { group, underlay }; hdl.set_m2p(&req)?; @@ -1123,3 +1208,29 @@ fn main() -> anyhow::Result<()> { Ok(()) } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn router_list_entry_parses() { + let e: RouterListEntryArg = "1000=default".parse().unwrap(); + assert_eq!(e.priority, 1000); + assert_eq!(e.router, None); + + let id = "6fe93b02-9b02-4c40-9dbd-33b50b3ba9f4"; + let e: RouterListEntryArg = format!("10={id}").parse().unwrap(); + assert_eq!(e.priority, 10); + assert_eq!(e.router, Some(id.parse().unwrap())); + } + + #[test] + fn router_list_entry_rejects_bad_input() { + assert!("10".parse::().is_err()); + assert!("=default".parse::().is_err()); + assert!("banana=default".parse::().is_err()); + assert!("70000=default".parse::().is_err()); + assert!("10=not-a-uuid".parse::().is_err()); + } +} diff --git a/crates/opte-api/Cargo.toml b/crates/opte-api/Cargo.toml index 7c4d2e60..bbb3a781 100644 --- a/crates/opte-api/Cargo.toml +++ b/crates/opte-api/Cargo.toml @@ -15,6 +15,7 @@ illumos-sys-hdrs.workspace = true ingot.workspace = true ipnetwork = { workspace = true, optional = true } +poptrie.workspace = true postcard.workspace = true serde.workspace = true diff --git a/crates/opte-api/src/cmd.rs b/crates/opte-api/src/cmd.rs index 675a5a26..ea03c9c2 100644 --- a/crates/opte-api/src/cmd.rs +++ b/crates/opte-api/src/cmd.rs @@ -116,6 +116,12 @@ pub enum OpteCmd { McastUnsubscribeAll = 108, /// Read out all M2P (multicast group -> underlay multicast) mappings. DumpMcast2Phys = 109, + + /// Replace a port's prioritized router list for boundary TEP + /// selection. + SetRouterList = 110, + /// Read a port's prioritized router list. + DumpRouterList = 111, } impl TryFrom for OpteCmd { @@ -158,6 +164,8 @@ impl TryFrom for OpteCmd { 107 => Ok(Self::DumpMcastSubscriptions), 108 => Ok(Self::McastUnsubscribeAll), 109 => Ok(Self::DumpMcast2Phys), + 110 => Ok(Self::SetRouterList), + 111 => Ok(Self::DumpRouterList), _ => Err(()), } } diff --git a/crates/opte-api/src/ip.rs b/crates/opte-api/src/ip.rs index 04c59140..4037997e 100644 --- a/crates/opte-api/src/ip.rs +++ b/crates/opte-api/src/ip.rs @@ -1346,6 +1346,18 @@ impl From for ipnetwork::Ipv4Network { } } +impl poptrie::Prefix for Ipv4Cidr { + type ADDRESS = u32; + + fn address(&self) -> Self::ADDRESS { + u32::from_be_bytes(self.ip.inner) + } + + fn prefix_length(&self) -> u8 { + self.prefix_len.val() + } +} + /// An IPv6 CIDR. #[derive(Clone, Copy, Debug, Deserialize, Eq, PartialEq, Serialize)] pub struct Ipv6Cidr { @@ -1533,6 +1545,18 @@ impl NetworkRepr<[u8; 16]> for Ipv6Addr { } } +impl poptrie::Prefix for Ipv6Cidr { + type ADDRESS = u128; + + fn address(&self) -> Self::ADDRESS { + u128::from_be_bytes(self.ip.bytes()) + } + + fn prefix_length(&self) -> u8 { + self.prefix_len.val() + } +} + #[cfg(test)] mod test { use super::*; diff --git a/crates/opte-api/src/lib.rs b/crates/opte-api/src/lib.rs index 0ea97614..c3fef6e6 100644 --- a/crates/opte-api/src/lib.rs +++ b/crates/opte-api/src/lib.rs @@ -51,7 +51,7 @@ pub use ulp::*; /// /// We rely on CI and the check-api-version.sh script to verify that /// this number is incremented anytime the oxide-api code changes. -pub const API_VERSION: u64 = 41; +pub const API_VERSION: u64 = 42; /// Major version of the OPTE package. pub const MAJOR_VERSION: u64 = 0; diff --git a/lib/opte-ioctl/src/lib.rs b/lib/opte-ioctl/src/lib.rs index ffb5a92c..c52ef3b8 100644 --- a/lib/opte-ioctl/src/lib.rs +++ b/lib/opte-ioctl/src/lib.rs @@ -42,6 +42,8 @@ use oxide_vpc::api::DetachSubnetResp; use oxide_vpc::api::DumpMcast2PhysResp; use oxide_vpc::api::DumpMcastForwardingResp; use oxide_vpc::api::DumpMcastSubscriptionsResp; +use oxide_vpc::api::DumpRouterListReq; +use oxide_vpc::api::DumpRouterListResp; use oxide_vpc::api::DumpVirt2BoundaryResp; use oxide_vpc::api::DumpVirt2PhysResp; use oxide_vpc::api::IpCidr; @@ -56,6 +58,7 @@ use oxide_vpc::api::SetExternalIpsReq; use oxide_vpc::api::SetFwRulesReq; use oxide_vpc::api::SetMcast2PhysReq; use oxide_vpc::api::SetMcastForwardingReq; +use oxide_vpc::api::SetRouterListReq; use oxide_vpc::api::SetVirt2BoundaryReq; use oxide_vpc::api::SetVirt2PhysReq; use oxide_vpc::api::VpcCfg; @@ -252,6 +255,24 @@ impl OpteHdl { run_cmd_ioctl(self.device.as_raw_fd(), cmd, None::<&()>) } + /// Replace a port's prioritized router list. + pub fn set_router_list( + &self, + req: &SetRouterListReq, + ) -> Result { + let cmd = OpteCmd::SetRouterList; + run_cmd_ioctl(self.device.as_raw_fd(), cmd, Some(&req)) + } + + /// Read a port's prioritized router list. + pub fn dump_router_list( + &self, + req: &DumpRouterListReq, + ) -> Result { + let cmd = OpteCmd::DumpRouterList; + run_cmd_ioctl(self.device.as_raw_fd(), cmd, Some(&req)) + } + /// Set a multicast forwarding entry. pub fn set_mcast_fwd( &self, diff --git a/lib/opte-test-utils/src/lib.rs b/lib/opte-test-utils/src/lib.rs index d7d78dfe..56b8aa23 100644 --- a/lib/opte-test-utils/src/lib.rs +++ b/lib/opte-test-utils/src/lib.rs @@ -22,6 +22,7 @@ pub use opte::api::Direction::*; pub use opte::api::MacAddr; pub use opte::ddi::mblk::MsgBlk; pub use opte::ddi::mblk::MsgBlkIterMut; +pub use opte::ddi::sync::KRwLock; pub use opte::engine::GenericUlp; pub use opte::engine::NetworkParser; pub use opte::engine::ether::EtherMeta; @@ -72,6 +73,7 @@ pub use oxide_vpc::api::Ipv4Cfg; pub use oxide_vpc::api::Ipv6Cfg; pub use oxide_vpc::api::PhysNet; pub use oxide_vpc::api::RouterClass; +pub use oxide_vpc::api::RouterList; pub use oxide_vpc::api::RouterTarget; pub use oxide_vpc::api::SNat4Cfg; pub use oxide_vpc::api::SNat6Cfg; @@ -270,6 +272,7 @@ fn oxide_net_builder( v2p: Arc, m2p: Arc, v2b: Arc, + router_list: Arc>, ) -> PortBuilder { #[allow(clippy::arc_with_non_send_sync)] let ectx = Arc::new(ExecCtx { log: Box::new(opte::PrintlnLog {}) }); @@ -291,7 +294,7 @@ fn oxide_net_builder( .expect("failed to setup gateway layer"); router::setup(&pb, cfg, one_limit).expect("failed to add router layer"); nat::setup(&mut pb, cfg, snat_limit).expect("failed to add nat layer"); - overlay::setup(&pb, cfg, v2p, m2p, v2b, one_limit) + overlay::setup(&pb, cfg, v2p, m2p, v2b, router_list, one_limit) .expect("failed to add overlay layer"); pb } @@ -364,9 +367,14 @@ pub fn oxide_net_setup2( let v2b = Arc::new(Virt2Boundary::new()); let m2p = Arc::new(Mcast2Phys::new()); + let router_list = Arc::new(KRwLock::new(RouterList::default_only())); let converted_cfg = oxide_vpc::cfg::VpcCfg::with_mtu(cfg.clone(), 1500); - let vpc_net = VpcNetwork { cfg: converted_cfg.clone(), v2b: v2b.clone() }; + let vpc_net = VpcNetwork { + cfg: converted_cfg.clone(), + v2b: v2b.clone(), + router_list: router_list.clone(), + }; let uft_limit = flow_table_limits.unwrap_or(UFT_LIMIT.unwrap()); let tcp_limit = flow_table_limits.unwrap_or(TCP_LIMIT.unwrap()); @@ -392,6 +400,7 @@ pub fn oxide_net_setup2( port_v2p, m2p.clone(), v2b, + router_list, ) .create(vpc_net, uft_limit, tcp_limit) .unwrap(); diff --git a/lib/oxide-vpc/src/api.rs b/lib/oxide-vpc/src/api.rs index eb2176f2..10617823 100644 --- a/lib/oxide-vpc/src/api.rs +++ b/lib/oxide-vpc/src/api.rs @@ -472,6 +472,91 @@ impl PartialEq for TunnelEndpoint { impl Eq for TunnelEndpoint {} +/// Identifies a router whose boundary table (V2B) may be consulted for +/// TEP selection. `None` is the default router, i.e. tunnel routes +/// advertised without a router id. +pub type RouterId = Option; + +/// The prioritized list of routers consulted for boundary TEP selection. +#[derive(Clone, Debug, Deserialize, Serialize, PartialEq, Eq)] +#[serde(try_from = "Vec<(u16, RouterId)>", into = "Vec<(u16, RouterId)>")] +pub struct RouterList(BTreeMap); + +/// The priority given to the default router when a port has no +/// explicit router list. +pub const DEFAULT_ROUTER_PRIORITY: u16 = 1000; + +/// The maximum number of routers that can be associated with a single port. +pub const MAX_ROUTER_LIST_ENTRIES: usize = 64; + +impl RouterList { + /// Build a list from `(priority, router)` entries, in any order. + /// Duplicate priorities and duplicate routers are rejected. + pub fn new( + entries: impl IntoIterator, + ) -> Result { + let mut list = BTreeMap::new(); + for (prio, router) in entries { + if list.len() >= MAX_ROUTER_LIST_ENTRIES { + return Err(format!( + "router list has more than {MAX_ROUTER_LIST_ENTRIES} entries" + )); + } + if list.contains_key(&prio) { + return Err(format!( + "duplicate priority {prio} in router list" + )); + } + if list.values().any(|r| *r == router) { + return Err(match router { + Some(id) => format!("duplicate router {id} in router list"), + None => { + "duplicate default router in router list".to_string() + } + }); + } + list.insert(prio, router); + } + Ok(Self(list)) + } + + /// The list used when nothing has been configured: just the + /// default router. + pub fn default_only() -> Self { + Self(BTreeMap::from([(DEFAULT_ROUTER_PRIORITY, None)])) + } + + /// Iterate over routers in priority order. + pub fn iter(&self) -> impl Iterator { + self.0.values() + } + + /// Iterate over `(priority, router)` entries in priority order. + pub fn entries(&self) -> impl Iterator + '_ { + self.0.iter().map(|(p, r)| (*p, *r)) + } +} + +impl Default for RouterList { + fn default() -> Self { + Self::default_only() + } +} + +impl TryFrom> for RouterList { + type Error = String; + + fn try_from(entries: Vec<(u16, RouterId)>) -> Result { + Self::new(entries) + } +} + +impl From for Vec<(u16, RouterId)> { + fn from(list: RouterList) -> Self { + list.0.into_iter().collect() + } +} + /// The physical address for a guest, minus the VNI. /// /// We save space in the VPC mappings by grouping guest @@ -684,7 +769,8 @@ pub struct V2bMapResp { #[derive(Debug, Deserialize, Serialize)] pub struct DumpVirt2BoundaryResp { - pub mappings: V2bMapResp, + /// Per-router mappings; `None` is the default router. + pub routers: Vec<(RouterId, V2bMapResp)>, } impl CmdOk for DumpVirt2BoundaryResp {} @@ -744,20 +830,44 @@ pub struct ClearMcast2PhysReq { pub underlay: MulticastUnderlay, } -/// Set a mapping from a VPC IP to boundary tunnel endpoint destination. +/// Set a mapping from a VPC IP to boundary tunnel endpoint destination +/// in the given router's table (`None` = default router). #[derive(Clone, Debug, Deserialize, Serialize)] pub struct SetVirt2BoundaryReq { + pub router_id: RouterId, pub vip: IpCidr, pub tep: Vec, } -/// Clear a mapping from VPC IP to a boundary tunnel endpoint destination. +/// Clear a mapping from VPC IP to a boundary tunnel endpoint destination +/// in the given router's table (`None` = default router). #[derive(Clone, Debug, Deserialize, Serialize)] pub struct ClearVirt2BoundaryReq { + pub router_id: RouterId, pub vip: IpCidr, pub tep: Vec, } +/// Replace a port's prioritized router list for boundary TEP selection. +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct SetRouterListReq { + pub port_name: String, + pub list: RouterList, +} + +/// Read a port's prioritized router list. +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct DumpRouterListReq { + pub port_name: String, +} + +#[derive(Clone, Debug, Deserialize, Serialize)] +pub struct DumpRouterListResp { + pub list: RouterList, +} + +impl CmdOk for DumpRouterListResp {} + /// Add an entry to the router. Addresses may be either IPv4 or IPv6, though the /// destination and target must match in protocol version. #[derive(Clone, Debug, Deserialize, Serialize)] diff --git a/lib/oxide-vpc/src/engine/mod.rs b/lib/oxide-vpc/src/engine/mod.rs index 9d810b8c..d157fdb3 100644 --- a/lib/oxide-vpc/src/engine/mod.rs +++ b/lib/oxide-vpc/src/engine/mod.rs @@ -13,6 +13,7 @@ pub mod overlay; pub mod router; use crate::api::BOUNDARY_SERVICES_VNI; +use crate::api::RouterList; use crate::cfg::VpcCfg; use crate::engine::geneve::OxideOptions; use crate::engine::geneve::ValidOxideOption; @@ -32,6 +33,7 @@ use ingot::types::HeaderLen; use opte::api::IpAddr; use opte::api::Vni; use opte::ddi::mblk::MsgBlk; +use opte::ddi::sync::KRwLock; use opte::engine::Direction; use opte::engine::HdlErrAction; use opte::engine::HdlPktAction; @@ -115,6 +117,9 @@ impl VpcParser { pub struct VpcNetwork { pub cfg: VpcCfg, pub v2b: Arc, + /// The port's prioritized router list for boundary TEP selection. + /// Shared with the overlay layer's `EncapAction`. + pub router_list: Arc>, } impl core::fmt::Debug for VpcNetwork { @@ -562,8 +567,10 @@ impl NetworkImpl for VpcNetwork { )); } - let Some(nhs) = - self.v2b.get(&recipient).filter(|v| !v.is_empty()) + let Some(nhs) = self + .v2b + .lookup_best(&self.router_list.read(), &recipient) + .filter(|v| !v.is_empty()) else { return Err(HdlPktError( "no external nexthop for ICMP reply", diff --git a/lib/oxide-vpc/src/engine/overlay.rs b/lib/oxide-vpc/src/engine/overlay.rs index a68d2a40..56c9b9e3 100644 --- a/lib/oxide-vpc/src/engine/overlay.rs +++ b/lib/oxide-vpc/src/engine/overlay.rs @@ -16,6 +16,8 @@ use crate::api::DumpVirt2PhysResp; use crate::api::GuestPhysAddr; use crate::api::PhysNet; use crate::api::Replication; +use crate::api::RouterId; +use crate::api::RouterList; use crate::api::TunnelEndpoint; use crate::api::V2bMapResp; use crate::api::VpcMapResp; @@ -38,7 +40,6 @@ use opte::api::MacAddr; use opte::api::MulticastUnderlay; use opte::api::OpteError; use opte::ddi::sync::KMutex; -use opte::ddi::sync::KMutexGuard; use opte::ddi::sync::KRwLock; use opte::engine::ether::EtherMeta; use opte::engine::ether::EtherMod; @@ -92,6 +93,7 @@ pub fn setup( v2p: Arc, m2p: Arc, v2b: Arc, + router_list: Arc>, ft_limit: core::num::NonZeroU32, ) -> core::result::Result<(), OpteError> { // Action Index 0 @@ -101,6 +103,7 @@ pub fn setup( v2p, m2p, v2b, + router_list, ))); // Action Index 1 @@ -209,6 +212,9 @@ pub struct EncapAction { v2p: Arc, m2p: Arc, v2b: Arc, + // The port's prioritized list of routers for boundary TEP + // selection. + router_list: Arc>, } impl EncapAction { @@ -218,8 +224,9 @@ impl EncapAction { v2p: Arc, m2p: Arc, v2b: Arc, + router_list: Arc>, ) -> Self { - Self { phys_ip_src, vni, v2p, m2p, v2b } + Self { phys_ip_src, vni, v2p, m2p, v2b, router_list } } } @@ -307,7 +314,11 @@ impl StaticAction for EncapAction { // // It's a possible optimisation, but it'd need more thought. RouterTargetInternal::InternetGateway(_) => { - match self.v2b.get(&recipient) { + let teps = { + let list = self.router_list.read(); + self.v2b.lookup_best(&list, &recipient) + }; + match teps { Some(phys) if sent_from_eip => { // Hash the packet onto a route target. This is a very // rudimentary mechanism. Should level-up to an ECMP @@ -774,26 +785,25 @@ pub struct Virt2Phys { ip6: KMutex>, } -/// A mapping from virtual IPs to boundary services addresses. +/// A mapping from virtual IPs to boundary services addresses, +/// partitioned by router. pub struct Virt2Boundary { + routers: KRwLock>>, +} + +/// A single router's mapping from IP prefixes to boundary TEPs. +pub struct RouterV2b { // The BTreeMap-based representation of the v2b table is a representation // that is easily updated. ip4: KMutex>>, ip6: KMutex>>, // The Poptrie-based representation of the v2b table is a data structure - // optimized for fast query times. It's not easily updated in-place. It's - // rebuilt each time an update is made. The heuristic being applied here is - // we expect table churn to be highly-infrequent compared to lookups. - // Lookups may happen millions of times per second and and we want those to - // be as fast as possible. At the time of writing, poptrie is the fastest - // LPM lookup data structure known to the author. - // - // The poptrie is under an read-write lock to allow multiple concurrent - // readers. When we update we hold the lock just long enough to do a swap - // with a poptrie that was pre-built out of band. - pt4: KRwLock>>, - pt6: KRwLock>>, + // optimized for fast query times. It's under an read-write lock to allow + // multiple concurrent readers. When we update we hold the lock just long + // enough to swap the entry with one that was pre-built out of band. + pt4: KRwLock>>, + pt6: KRwLock>>, } /// A mapping from inner multicast destination IPs to underlay multicast groups. @@ -808,35 +818,114 @@ pub struct Mcast2Phys { pub const TUNNEL_ENDPOINT_MAC: [u8; 6] = [0xA8, 0x40, 0x25, 0x77, 0x77, 0x77]; impl Virt2Boundary { - pub fn dump_ip4(&self) -> Vec<(Ipv4Cidr, BTreeSet)> { - self.ip4 - .lock() - .iter() - .map(|(vip, baddr)| (*vip, baddr.clone())) - .collect() - } - - pub fn dump_ip6(&self) -> Vec<(Ipv6Cidr, BTreeSet)> { - self.ip6 - .lock() - .iter() - .map(|(vip, baddr)| (*vip, baddr.clone())) - .collect() + pub fn new() -> Self { + let mut routers = BTreeMap::new(); + // Start with the default router. + routers.insert(None, Arc::new(RouterV2b::new())); + Virt2Boundary { routers: KRwLock::new(routers) } } + /// Dump every router's mappings. pub fn dump(&self) -> DumpVirt2BoundaryResp { DumpVirt2BoundaryResp { - mappings: V2bMapResp { ip4: self.dump_ip4(), ip6: self.dump_ip6() }, + routers: self + .routers + .read() + .iter() + .map(|(id, rtr)| { + ( + *id, + V2bMapResp { ip4: rtr.dump_ip4(), ip6: rtr.dump_ip6() }, + ) + }) + .collect(), } } - pub fn new() -> Self { - Virt2Boundary { - ip4: KMutex::new(BTreeMap::new()), - ip6: KMutex::new(BTreeMap::new()), - pt4: KRwLock::new(Poptrie::default()), - pt6: KRwLock::new(Poptrie::default()), + pub fn router(&self, id: &RouterId) -> Option> { + self.routers.read().get(id).cloned() + } + + /// Set a prefix's TEPs in the default router's table. + pub fn set>( + &self, + vip: IpCidr, + tep: I, + ) -> Option> { + self.set_for(None, vip, tep) + } + + /// Set a prefix's TEPs in the given router's table, creating the + /// router's table if it does not yet exist. + pub fn set_for>( + &self, + router: RouterId, + vip: IpCidr, + tep: I, + ) -> Option> { + let mut routers = self.routers.write(); + let rtr = routers + .entry(router) + .or_insert_with(|| Arc::new(RouterV2b::new())) + .clone(); + rtr.set(vip, tep) + } + + /// Remove TEPs from a prefix in the default router's table. + pub fn remove>( + &self, + vip: IpCidr, + tep: I, + ) -> Option> { + self.remove_for(None, vip, tep) + } + + /// Remove TEPs from a prefix in the given router's table. + /// + /// When a named router's last prefix clears, its table is dropped + /// from the map. The default router's (empty) table always exists. + pub fn remove_for>( + &self, + router: RouterId, + vip: IpCidr, + tep: I, + ) -> Option> { + let mut routers = self.routers.write(); + let rtr = routers.get(&router)?.clone(); + let orig = rtr.remove(vip, tep); + if router.is_some() && rtr.is_empty() { + routers.remove(&router); } + orig + } + + /// Find the best TEP set for `vip` across the routers in `list`. + /// + /// Routers are consulted in priority order; the longest matching + /// prefix wins, and equal-length matches are won by the earliest + /// (highest-priority) router. + pub fn lookup_best( + &self, + list: &RouterList, + vip: &IpAddr, + ) -> Option> { + let routers = self.routers.read(); + let mut best: Option<(u8, BTreeSet)> = None; + for id in list.iter() { + let Some(rtr) = routers.get(id) else { + continue; + }; + let Some((len, teps)) = rtr.get_with_len(vip) else { + continue; + }; + if teps.is_empty() { + continue; + } + if best.as_ref().is_none_or(|(blen, _)| len > *blen) { + best = Some((len, teps)); + } + } + best.map(|(_, teps)| teps) } } @@ -853,11 +942,58 @@ impl ResourceEntry for PhysNet {} // different type than the query argument. Keys are prefixes and query arguments // are IPs. The mapping resource trait requires that the keys and query // arguments be of the same type. -impl Virt2Boundary { +impl RouterV2b { + pub fn new() -> Self { + RouterV2b { + ip4: KMutex::new(BTreeMap::new()), + ip6: KMutex::new(BTreeMap::new()), + pt4: KRwLock::new(Poptrie::default()), + pt6: KRwLock::new(Poptrie::default()), + } + } + + pub fn dump_ip4(&self) -> Vec<(Ipv4Cidr, BTreeSet)> { + self.ip4 + .lock() + .iter() + .map(|(vip, baddr)| (*vip, baddr.clone())) + .collect() + } + + pub fn dump_ip6(&self) -> Vec<(Ipv6Cidr, BTreeSet)> { + self.ip6 + .lock() + .iter() + .map(|(vip, baddr)| (*vip, baddr.clone())) + .collect() + } + pub fn get(&self, vip: &IpAddr) -> Option> { + self.get_with_len(vip).map(|(_, teps)| teps) + } + + /// True when the router has no prefixes in either address family. + pub fn is_empty(&self) -> bool { + self.ip4.lock().is_empty() && self.ip6.lock().is_empty() + } + + /// Longest-prefix match returning the matched prefix length + /// alongside the TEP set. + pub fn get_with_len( + &self, + vip: &IpAddr, + ) -> Option<(u8, BTreeSet)> { match vip { - IpAddr::Ip4(ip4) => self.pt4.read().match_v4(u32::from(*ip4)), - IpAddr::Ip6(ip6) => self.pt6.read().match_v6(u128::from(*ip6)), + IpAddr::Ip4(ip4) => self + .pt4 + .read() + .lookup_with_prefix(u32::from(*ip4)) + .map(|(cidr, teps)| (cidr.prefix_len(), teps.clone())), + IpAddr::Ip6(ip6) => self + .pt6 + .read() + .lookup_with_prefix(u128::from(*ip6)) + .map(|(cidr, teps)| (cidr.prefix_len(), teps.clone())), } } @@ -875,14 +1011,15 @@ impl Virt2Boundary { for t in tep.into_iter() { entry.remove(&t); } + self.pt4.write().insert(ip4, entry.clone()); (entry.is_empty(), Some(orig)) } None => (false, None), }; if clear { tbl.remove(&ip4); + self.pt4.write().remove(ip4); } - self.update_poptrie_v4(&tbl); orig } IpCidr::Ip6(ip6) => { @@ -893,14 +1030,15 @@ impl Virt2Boundary { for t in tep.into_iter() { entry.remove(&t); } + self.pt6.write().insert(ip6, entry.clone()); (entry.is_empty(), Some(orig)) } None => (false, None), }; if clear { tbl.remove(&ip6); + self.pt6.write().remove(ip6); } - self.update_poptrie_v6(&tbl); orig } } @@ -914,55 +1052,43 @@ impl Virt2Boundary { match vip { IpCidr::Ip4(ip4) => { let mut tbl = self.ip4.lock(); - let result = match tbl.get_mut(&ip4) { + let updated = match tbl.get_mut(&ip4) { Some(entry) => { - let orig = entry.clone(); entry.extend(tep); - Some(orig) + entry.clone() + } + None => { + let updated: BTreeSet = + tep.into_iter().collect(); + tbl.insert(ip4, updated.clone()); + updated } - None => tbl.insert(ip4, tep.into_iter().collect()), }; - self.update_poptrie_v4(&tbl); - result + self.pt4.write().insert(ip4, updated) } IpCidr::Ip6(ip6) => { let mut tbl = self.ip6.lock(); - let result = match tbl.get_mut(&ip6) { + let updated = match tbl.get_mut(&ip6) { Some(entry) => { - let orig = entry.clone(); entry.extend(tep); - Some(orig) + entry.clone() + } + None => { + let updated: BTreeSet = + tep.into_iter().collect(); + tbl.insert(ip6, updated.clone()); + updated } - None => tbl.insert(ip6, tep.into_iter().collect()), }; - self.update_poptrie_v6(&tbl); - result + self.pt6.write().insert(ip6, updated) } } } +} - fn update_poptrie_v4( - &self, - tree: &KMutexGuard>>, - ) { - let table = poptrie::Ipv4RoutingTable( - tree.iter() - .map(|(k, v)| ((k.ip().bytes(), k.prefix_len()), v.clone())) - .collect(), - ); - *self.pt4.write() = poptrie::Poptrie::from(table); - } - - fn update_poptrie_v6( - &self, - tree: &KMutexGuard>>, - ) { - let table = poptrie::Ipv6RoutingTable( - tree.iter() - .map(|(k, v)| ((k.ip().bytes(), k.prefix_len()), v.clone())) - .collect(), - ); - *self.pt6.write() = poptrie::Poptrie::from(table); +impl Default for RouterV2b { + fn default() -> Self { + Self::new() } } @@ -1086,3 +1212,193 @@ impl MappingResource for Mcast2Phys { } } } + +#[cfg(test)] +mod tests { + use super::*; + use uuid::Uuid; + + fn tep(last: u16) -> TunnelEndpoint { + TunnelEndpoint { + ip: Ipv6Addr::from([0xfd00, 0, 0, 0, 0, 0, 0, last]), + vni: Vni::new(99u32).unwrap(), + } + } + + fn v4(ip: &str) -> IpAddr { + IpAddr::Ip4(ip.parse().unwrap()) + } + + #[test] + fn default_router_compat() { + let v2b = Virt2Boundary::new(); + v2b.set("0.0.0.0/0".parse().unwrap(), [tep(1)]); + + let list = RouterList::default_only(); + let teps = v2b.lookup_best(&list, &v4("10.1.2.3")).unwrap(); + assert_eq!(teps, BTreeSet::from([tep(1)])); + } + + #[test] + fn lpm_wins_across_routers() { + let v2b = Virt2Boundary::new(); + let r1 = Some(Uuid::from_u128(1)); + let r2 = Some(Uuid::from_u128(2)); + + v2b.set_for(r1, "10.0.0.0/8".parse().unwrap(), [tep(1)]); + v2b.set_for(r2, "10.1.0.0/16".parse().unwrap(), [tep(2)]); + + let list = RouterList::new([(10, r1), (20, r2)]).unwrap(); + let best = v2b.lookup_best(&list, &v4("10.1.2.3")).unwrap(); + assert_eq!(best, BTreeSet::from([tep(2)])); + + let best = v2b.lookup_best(&list, &v4("10.2.0.1")).unwrap(); + assert_eq!(best, BTreeSet::from([tep(1)])); + } + + #[test] + fn priority_breaks_lpm_ties() { + let v2b = Virt2Boundary::new(); + let r1 = Some(Uuid::from_u128(1)); + let r2 = Some(Uuid::from_u128(2)); + v2b.set_for(r1, "10.0.0.0/8".parse().unwrap(), [tep(1)]); + v2b.set_for(r2, "10.0.0.0/8".parse().unwrap(), [tep(2)]); + + let list = RouterList::new([(20, r2), (10, r1)]).unwrap(); + let best = v2b.lookup_best(&list, &v4("10.1.2.3")).unwrap(); + assert_eq!(best, BTreeSet::from([tep(1)])); + } + + #[test] + fn unlisted_and_unknown_routers_are_ignored() { + let v2b = Virt2Boundary::new(); + let r1 = Some(Uuid::from_u128(1)); + let r2 = Some(Uuid::from_u128(2)); + let ghost = Some(Uuid::from_u128(3)); + // r2 has the longest match but is not in the list. The default + // router has a match but is likewise not listed. + v2b.set("10.0.0.0/8".parse().unwrap(), [tep(0)]); + v2b.set_for(r1, "10.0.0.0/8".parse().unwrap(), [tep(1)]); + v2b.set_for(r2, "10.1.0.0/16".parse().unwrap(), [tep(2)]); + + let list = RouterList::new([(10, r1), (20, ghost)]).unwrap(); + let best = v2b.lookup_best(&list, &v4("10.1.2.3")).unwrap(); + assert_eq!(best, BTreeSet::from([tep(1)])); + + // A list of only unknown routers matches nothing. + let list = RouterList::new([(10, ghost)]).unwrap(); + assert!(v2b.lookup_best(&list, &v4("10.1.2.3")).is_none()); + } + + #[test] + fn router_list_rejects_duplicate_priorities() { + let r1 = Some(Uuid::from_u128(1)); + let r2 = Some(Uuid::from_u128(2)); + assert!(RouterList::new([(10, r1), (10, r2)]).is_err()); + + let list = RouterList::new([(20, r2), (10, r1)]).unwrap(); + assert_eq!(list.entries().collect::>(), &[(10, r1), (20, r2)]); + } + + #[test] + fn router_list_rejects_duplicate_routers() { + let r1 = Some(Uuid::from_u128(1)); + let r2 = Some(Uuid::from_u128(2)); + // The same router at two different priorities. + let err = RouterList::new([(10, r1), (20, r1)]).unwrap_err(); + assert!(err.contains("duplicate router"), "{err}"); + // The default router listed twice. + let err = RouterList::new([(10, None), (20, None)]).unwrap_err(); + assert!(err.contains("duplicate default router"), "{err}"); + // Duplicates hidden among valid entries, in any order. + assert!( + RouterList::new([(30, r2), (10, None), (20, r1), (5, r2)]).is_err() + ); + // Distinct routers, including one default entry, are fine. + let list = RouterList::new([(20, r2), (1000, None), (10, r1)]).unwrap(); + assert_eq!( + list.entries().collect::>(), + &[(10, r1), (20, r2), (1000, None)] + ); + + let err = RouterList::new([(10, r1), (10, None)]).unwrap_err(); + assert!(err.contains("duplicate priority"), "{err}"); + } + + #[test] + fn router_list_rejects_oversized_lists() { + let entry = |i: usize| (i as u16, Some(Uuid::from_u128(i as u128))); + let max = crate::api::MAX_ROUTER_LIST_ENTRIES; + assert!(RouterList::new((0..=max).map(entry)).is_err()); + assert!(RouterList::new((1..=max).map(entry)).is_ok()); + } + + #[test] + fn empty_tep_sets_are_skipped() { + let v2b = Virt2Boundary::new(); + let r1 = Some(Uuid::from_u128(1)); + let r2 = Some(Uuid::from_u128(2)); + // r1's longer prefix has an empty TEP set (Set with no TEPs); it + // must not win the LPM nor shadow r2's valid shorter prefix. + v2b.set_for(r1, "10.1.0.0/16".parse().unwrap(), []); + v2b.set_for(r2, "10.0.0.0/8".parse().unwrap(), [tep(2)]); + + let list = RouterList::new([(10, r1), (20, r2)]).unwrap(); + let best = v2b.lookup_best(&list, &v4("10.1.2.3")).unwrap(); + assert_eq!(best, BTreeSet::from([tep(2)])); + + // With no other match at all, an empty set means no match. + let list = RouterList::new([(10, r1)]).unwrap(); + assert!(v2b.lookup_best(&list, &v4("10.1.2.3")).is_none()); + } + + #[test] + fn emptied_router_tables_are_reaped() { + let v2b = Virt2Boundary::new(); + let r1 = Some(Uuid::from_u128(1)); + let pfx4 = "10.0.0.0/8".parse().unwrap(); + let pfx6 = "fd00::/8".parse().unwrap(); + v2b.set_for(r1, pfx4, [tep(1)]); + v2b.set_for(r1, pfx6, [tep(2)]); + assert!(v2b.router(&r1).is_some()); + + // One family emptied: the table stays. + v2b.remove_for(r1, pfx4, [tep(1)]); + assert!(v2b.router(&r1).is_some()); + + // Last prefix gone: the table goes with it, and dump shows no + // empty section for r1. + v2b.remove_for(r1, pfx6, [tep(2)]); + assert!(v2b.router(&r1).is_none()); + assert!(!v2b.dump().routers.iter().any(|(id, _)| *id == r1)); + + // The default router is never reaped. + let dflt = "0.0.0.0/0".parse().unwrap(); + v2b.set(dflt, [tep(3)]); + v2b.remove(dflt, [tep(3)]); + assert!(v2b.router(&None).is_some()); + } + + #[test] + fn get_with_len_reports_prefix_length() { + let rtr = RouterV2b::new(); + rtr.set("10.0.0.0/8".parse().unwrap(), [tep(1)]); + rtr.set("10.1.0.0/16".parse().unwrap(), [tep(2)]); + rtr.set("::/0".parse().unwrap(), [tep(3)]); + rtr.set("fd00::/8".parse().unwrap(), [tep(4)]); + + let (len, _) = rtr.get_with_len(&v4("10.1.2.3")).unwrap(); + assert_eq!(len, 16); + let (len, _) = rtr.get_with_len(&v4("10.2.0.1")).unwrap(); + assert_eq!(len, 8); + assert!(rtr.get_with_len(&v4("192.168.0.1")).is_none()); + + let ip6 = IpAddr::Ip6("fd00::1".parse().unwrap()); + let (len, teps) = rtr.get_with_len(&ip6).unwrap(); + assert_eq!(len, 8); + assert_eq!(teps, BTreeSet::from([tep(4)])); + let ip6 = IpAddr::Ip6("2001:db8::1".parse().unwrap()); + let (len, _) = rtr.get_with_len(&ip6).unwrap(); + assert_eq!(len, 0); + } +} diff --git a/lib/oxide-vpc/src/print.rs b/lib/oxide-vpc/src/print.rs index c2f2ac30..ac4c4dc5 100644 --- a/lib/oxide-vpc/src/print.rs +++ b/lib/oxide-vpc/src/print.rs @@ -22,6 +22,7 @@ use opte::api::IpCidr; use opte::api::Vni; use opte::print::*; use std::io::Write; +use std::string::ToString; use tabwriter::TabWriter; /// Print the header for the [`print_v2p()`] output. @@ -91,25 +92,34 @@ pub fn print_v2b_into( let mut t = TabWriter::new(writer); writeln!(t, "Virtual to Boundary Mappings")?; write_hrb(&mut t)?; - writeln!(t, "\nIPv4 mappings")?; - write_hr(&mut t)?; - print_v2b_header(&mut t)?; - for x in &resp.mappings.ip4 { - for tep in &x.1 { - print_v2b_entry(&mut t, x.0.into(), tep.ip, tep.vni)?; + + for (router, mappings) in &resp.routers { + let router = match router { + None => "default".to_string(), + Some(id) => id.to_string(), + }; + writeln!(t, "\nRouter: {router}")?; + + writeln!(t, "\nIPv4 mappings")?; + write_hr(&mut t)?; + print_v2b_header(&mut t)?; + for x in &mappings.ip4 { + for tep in &x.1 { + print_v2b_entry(&mut t, x.0.into(), tep.ip, tep.vni)?; + } } - } - t.flush()?; + t.flush()?; - writeln!(t, "\nIPv6 mappings")?; - write_hr(&mut t)?; - print_v2b_header(&mut t)?; - for x in &resp.mappings.ip6 { - for tep in &x.1 { - print_v2b_entry(&mut t, x.0.into(), tep.ip, tep.vni)?; + writeln!(t, "\nIPv6 mappings")?; + write_hr(&mut t)?; + print_v2b_header(&mut t)?; + for x in &mappings.ip6 { + for tep in &x.1 { + print_v2b_entry(&mut t, x.0.into(), tep.ip, tep.vni)?; + } } + writeln!(t)?; } - writeln!(t)?; t.flush() } diff --git a/xde/src/xde.rs b/xde/src/xde.rs index 6281f323..598c2862 100644 --- a/xde/src/xde.rs +++ b/xde/src/xde.rs @@ -285,6 +285,8 @@ use oxide_vpc::api::DetachSubnetResp; use oxide_vpc::api::DumpMcast2PhysResp; use oxide_vpc::api::DumpMcastForwardingResp; use oxide_vpc::api::DumpMcastSubscriptionsResp; +use oxide_vpc::api::DumpRouterListReq; +use oxide_vpc::api::DumpRouterListResp; use oxide_vpc::api::DumpVirt2BoundaryResp; use oxide_vpc::api::DumpVirt2PhysResp; use oxide_vpc::api::InternetGatewayMap; @@ -303,9 +305,11 @@ use oxide_vpc::api::PortInfo; use oxide_vpc::api::RemFwRuleReq; use oxide_vpc::api::RemoveCidrResp; use oxide_vpc::api::Replication; +use oxide_vpc::api::RouterList; use oxide_vpc::api::SetFwRulesReq; use oxide_vpc::api::SetMcast2PhysReq; use oxide_vpc::api::SetMcastForwardingReq; +use oxide_vpc::api::SetRouterListReq; use oxide_vpc::api::SetVirt2BoundaryReq; use oxide_vpc::api::SetVirt2PhysReq; use oxide_vpc::api::SourceFilter; @@ -1052,6 +1056,16 @@ unsafe extern "C" fn xde_ioc_opte_cmd(karg: *mut c_void, mode: c_int) -> c_int { hdlr_resp(&mut env, resp) } + OpteCmd::SetRouterList => { + let resp = set_router_list_hdlr(&mut env); + hdlr_resp(&mut env, resp) + } + + OpteCmd::DumpRouterList => { + let resp = dump_router_list_hdlr(&mut env); + hdlr_resp(&mut env, resp) + } + OpteCmd::AddRouterEntry => { let resp = add_router_entry_hdlr(&mut env); hdlr_resp(&mut env, resp) @@ -3455,7 +3469,16 @@ fn new_port( gateway::setup(&pb, &cfg, vpc_map, FT_LIMIT_ONE)?; router::setup(&pb, &cfg, FT_LIMIT_ONE)?; nat::setup(&mut pb, &cfg, nat_ft_limit)?; - overlay::setup(&pb, &cfg, v2p, m2p, v2b.clone(), FT_LIMIT_ONE)?; + let router_list = Arc::new(KRwLock::new(RouterList::default_only())); + overlay::setup( + &pb, + &cfg, + v2p, + m2p, + v2b.clone(), + router_list.clone(), + FT_LIMIT_ONE, + )?; // Set the overall unified flow and TCP flow table limits based on the total // configuration above, by taking the maximum of size of the individual @@ -3466,7 +3489,7 @@ fn new_port( // construct a new one, so the unwrap is safe. let limit = NonZeroU32::new(FW_FT_LIMIT.get().max(nat_ft_limit.get())).unwrap(); - let net = VpcNetwork { cfg, v2b }; + let net = VpcNetwork { cfg, v2b, router_list }; let port = Arc::new(pb.create(net, limit, limit)?); Ok(port) } @@ -4057,7 +4080,7 @@ fn dump_m2p_hdlr() -> Result { fn set_v2b_hdlr(env: &mut IoctlEnvelope) -> Result { let req: SetVirt2BoundaryReq = env.copy_in_req()?; let state = get_xde_state(); - state.v2b.set(req.vip, req.tep); + state.v2b.set_for(req.router_id, req.vip, req.tep); Ok(NoResp::default()) } @@ -4065,7 +4088,7 @@ fn set_v2b_hdlr(env: &mut IoctlEnvelope) -> Result { fn clear_v2b_hdlr(env: &mut IoctlEnvelope) -> Result { let req: ClearVirt2BoundaryReq = env.copy_in_req()?; let state = get_xde_state(); - state.v2b.remove(req.vip, req.tep); + state.v2b.remove_for(req.router_id, req.vip, req.tep); Ok(NoResp::default()) } @@ -4075,6 +4098,40 @@ fn dump_v2b_hdlr() -> Result { Ok(state.v2b.dump()) } +#[unsafe(no_mangle)] +fn set_router_list_hdlr(env: &mut IoctlEnvelope) -> Result { + let req: SetRouterListReq = env.copy_in_req()?; + let state = get_xde_state(); + let devs = state.devs.read(); + let dev = devs + .get_by_name(&req.port_name) + .ok_or_else(|| OpteError::PortNotFound(req.port_name.clone()))?; + *dev.port.network().router_list.write() = req.list; + // UFT expiry is idle-based, so active flows never re-evaluate the + // list; flush them or a removed router keeps serving live traffic. + match dev.port.clear_uft() { + Ok(()) => {} + // A port that isn't running has no flows to invalidate. + Err(OpteError::BadState(_)) => {} + Err(e) => return Err(e), + } + Ok(NoResp::default()) +} + +#[unsafe(no_mangle)] +fn dump_router_list_hdlr( + env: &mut IoctlEnvelope, +) -> Result { + let req: DumpRouterListReq = env.copy_in_req()?; + let state = get_xde_state(); + let devs = state.devs.read(); + let dev = devs + .get_by_name(&req.port_name) + .ok_or_else(|| OpteError::PortNotFound(req.port_name.clone()))?; + let list = dev.port.network().router_list.read().clone(); + Ok(DumpRouterListResp { list }) +} + #[unsafe(no_mangle)] fn set_mcast_forwarding_hdlr( env: &mut IoctlEnvelope, From 8c7b60efd5bcbff67badf67de2de726dc9c0accb Mon Sep 17 00:00:00 2001 From: Nicolas Kagami Date: Thu, 17 Sep 2026 11:04:07 -0300 Subject: [PATCH 2/2] (temp) truncate helios build number --- pkg/build.sh | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/pkg/build.sh b/pkg/build.sh index b51d594f..345b4b19 100755 --- a/pkg/build.sh +++ b/pkg/build.sh @@ -45,6 +45,12 @@ pkgdepend generate -d proto opte.base.p5m > opte.generate.p5m mkdir -p packages pkgdepend resolve -d packages -s resolve.p5m opte.generate.p5m +# RFD 662 POC: pkgdepend resolve pins deps to the build host's exact branch +# (e.g. system/kernel@0.5.11-3.0.24068), which makes the p5p uninstallable on +# any system with an older helios build. Strip the branch so only the release +# component constrains. +sed -i 's/@\([0-9][0-9.]*\)-[0-9][0-9.]*/@\1/g' packages/opte.generate.p5m.resolve.p5m + cat opte.base.p5m packages/opte.generate.p5m.resolve.p5m > opte.final.p5m pkgrepo create $REPO