mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-06 03:28:24 +00:00
The drain deadline for the `previous` slot slides forward on every inbound frame that authenticates against it. That is deliberate: it stops the old epoch being erased out from under a peer that lost msg3 and is still sealing in it. But the only party that can push the deadline out is the authenticated peer holding that key, so a peer that keeps using the old epoch keeps the retired key resident indefinitely. Add an absolute ceiling measured from the cutover, so the sliding grace delays erasure by a bounded amount rather than preventing it. The ceiling has to clear the worst-case legitimate recovery of a peer that lost msg3, which is the msg3 resend ladder plus the responder's handshake timeout plus the rekey dampening window, about 90 seconds at stock settings. It defaults to 120 seconds and is raised to the budget the configured handshake timers actually imply, so tightening a timer cannot push the ceiling under the recovery it has to leave room for. This does not close the related gap where an FSP rekey we initiate and the peer never answers is never abandoned, which leaves the session's current epoch pinned and not rotating. The cap erases the old epoch on schedule regardless, which is a strict improvement, but a session read afterwards can show no drain alongside a stale current key for that reason rather than because of this change.
704 lines
30 KiB
Rust
704 lines
30 KiB
Rust
//! Periodic rekey (key rotation) for FMP link sessions.
|
|
//!
|
|
//! Checks all active peers on each tick for:
|
|
//! 1. Rekey trigger (time elapsed or send counter exceeded)
|
|
//! 2. Drain window expiry (clean up previous session after cutover)
|
|
//! 3. Initiator-side cutover (first send after handshake completion)
|
|
|
|
use crate::NodeAddr;
|
|
use crate::node::Node;
|
|
use crate::node::wire::build_msg1;
|
|
use crate::noise::HandshakeState;
|
|
use crate::protocol::{SessionDatagram, SessionSetup};
|
|
use tracing::{debug, info, trace, warn};
|
|
|
|
/// Keep previous session alive for this long after cutover.
|
|
const DRAIN_WINDOW_SECS: u64 = 10;
|
|
|
|
/// Suppress local rekey initiation for this long after receiving
|
|
/// a peer's rekey msg1.
|
|
const REKEY_DAMPENING_SECS: u64 = 30;
|
|
|
|
/// Floor on the absolute ceiling for `previous`-slot retention after a
|
|
/// cutover, in seconds.
|
|
///
|
|
/// The drain deadline is peer-progress-aware: it slides forward on every
|
|
/// inbound frame that authenticates against the old epoch, so a peer that
|
|
/// keeps sealing in that epoch holds the retired key for as long as it
|
|
/// likes. This bounds that. It has to stay longer than the worst-case
|
|
/// recovery of a legitimate peer that lost msg3, which at stock defaults
|
|
/// is the msg3 resend ladder (about 31 s) plus `handshake_timeout_secs`
|
|
/// (30 s) before the responder abandons plus `REKEY_DAMPENING_SECS`
|
|
/// (30 s) before it may re-initiate, so about 90 s. 120 s clears that
|
|
/// with margin and still bounds retention to roughly one
|
|
/// `node.rekey.after_secs` period. `drain_max_retention_ms` takes the
|
|
/// larger of this floor and the budget the running configuration
|
|
/// actually implies, so a shortened handshake timer cannot push the
|
|
/// ceiling under the recovery it has to clear.
|
|
///
|
|
/// Lowering it below that budget cuts off legitimate slow peers: their
|
|
/// frames go silently undecryptable until their own rekey retry
|
|
/// re-converges the epochs, because nothing tears an established session
|
|
/// down on repeated decrypt failure. Raising it lengthens the window in
|
|
/// which a retired key stays resident.
|
|
const DRAIN_MAX_RETENTION_SECS: u64 = DRAIN_WINDOW_SECS * 12;
|
|
|
|
/// Effective ceiling on total `previous`-slot retention, in milliseconds.
|
|
///
|
|
/// The larger of `DRAIN_MAX_RETENTION_SECS` and the msg3 recovery budget
|
|
/// the configured handshake timers imply, so the ceiling always clears
|
|
/// the recovery it is supposed to leave room for.
|
|
pub(in crate::node) fn drain_max_retention_ms(rate_limit: &crate::config::RateLimitConfig) -> u64 {
|
|
let mut ladder_ms: u64 = 0;
|
|
let mut interval = rate_limit.handshake_resend_interval_ms as f64;
|
|
for _ in 0..rate_limit.handshake_max_resends {
|
|
ladder_ms = ladder_ms.saturating_add(interval as u64);
|
|
interval *= rate_limit.handshake_resend_backoff;
|
|
}
|
|
let recovery_budget_ms = ladder_ms
|
|
.saturating_add(rate_limit.handshake_timeout_secs.saturating_mul(1000))
|
|
.saturating_add(REKEY_DAMPENING_SECS * 1000);
|
|
(DRAIN_MAX_RETENTION_SECS * 1000).max(recovery_budget_ms)
|
|
}
|
|
|
|
/// Liveness bound on how long the FSP rekey initiator holds the
|
|
/// `current` + `pending` state before cutting over to the new epoch.
|
|
///
|
|
/// This is NOT safety-critical: overlapping-epoch trial-decrypt covers
|
|
/// any skew between the two endpoints' cutovers. The timer only bounds
|
|
/// how long the initiator advertises the old K-bit. An opportunistic
|
|
/// early cutover also fires if the initiator authenticates a peer frame
|
|
/// against its own `pending` session (the responder cut over first).
|
|
const FSP_CUTOVER_DELAY_MS: u64 = 2000;
|
|
|
|
impl Node {
|
|
/// Periodic rekey check. Called from the tick loop.
|
|
///
|
|
/// For each active peer with a session:
|
|
/// - If the initiator has a pending session, perform K-bit cutover
|
|
/// - If the drain window has expired, clean up the previous session
|
|
/// - If the rekey timer/counter fires, initiate a new handshake
|
|
pub(in crate::node) async fn check_rekey(&mut self) {
|
|
if !self.config().node.rekey.enabled {
|
|
return;
|
|
}
|
|
|
|
let rekey_after_secs = self.config().node.rekey.after_secs;
|
|
let rekey_after_messages = self.config().node.rekey.after_messages;
|
|
|
|
// Collect peers that need action (to avoid borrow conflicts)
|
|
let mut peers_to_cutover: Vec<NodeAddr> = Vec::new();
|
|
let mut peers_to_drain: Vec<NodeAddr> = Vec::new();
|
|
let mut peers_to_rekey: Vec<NodeAddr> = Vec::new();
|
|
|
|
for (node_addr, peer) in &self.peers {
|
|
if !peer.has_session() || !peer.is_healthy() {
|
|
continue;
|
|
}
|
|
|
|
// 1. Initiator-side cutover: we completed a rekey and have
|
|
// a pending session ready. Cut over on the next tick.
|
|
if peer.pending_new_session().is_some() && !peer.rekey_in_progress() {
|
|
peers_to_cutover.push(*node_addr);
|
|
continue;
|
|
}
|
|
|
|
// 2. Drain window expiry
|
|
if peer.is_draining() && peer.drain_expired(DRAIN_WINDOW_SECS) {
|
|
peers_to_drain.push(*node_addr);
|
|
}
|
|
|
|
// 3. Rekey trigger
|
|
if peer.rekey_in_progress() {
|
|
continue;
|
|
}
|
|
if peer.is_rekey_dampened(REKEY_DAMPENING_SECS) {
|
|
continue;
|
|
}
|
|
|
|
let elapsed = peer.session_established_at().elapsed().as_secs();
|
|
let counter = peer
|
|
.noise_session()
|
|
.map(|s| s.current_send_counter())
|
|
.unwrap_or(0);
|
|
|
|
// Apply per-session symmetric jitter to desynchronize
|
|
// dual-initiation in symmetric-start meshes.
|
|
let effective_after_secs =
|
|
rekey_after_secs.saturating_add_signed(peer.rekey_jitter_secs());
|
|
if elapsed >= effective_after_secs || counter >= rekey_after_messages {
|
|
peers_to_rekey.push(*node_addr);
|
|
}
|
|
}
|
|
|
|
// Execute cutover for initiator side
|
|
for node_addr in peers_to_cutover {
|
|
let did_cutover = if let Some(peer) = self.peers.get_mut(&node_addr) {
|
|
if let Some(_old_our_index) = peer.cutover_to_new_session() {
|
|
// New index was pre-registered in peers_by_index
|
|
// during msg2 handling (handshake.rs).
|
|
debug_assert!(
|
|
peer.transport_id().is_some()
|
|
&& peer.our_index().is_some()
|
|
&& self.peers_by_index.contains_key(&(
|
|
peer.transport_id().unwrap(),
|
|
peer.our_index().unwrap().as_u32()
|
|
)),
|
|
"peers_by_index should contain pre-registered new index after cutover"
|
|
);
|
|
debug!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
"Rekey cutover complete (initiator), K-bit flipped"
|
|
);
|
|
true
|
|
} else {
|
|
false
|
|
}
|
|
} else {
|
|
false
|
|
};
|
|
// Re-register the new session with the decrypt worker — the
|
|
// cache_key (transport_id, our_index) just changed, so the
|
|
// old worker entry is stale and every packet on the new
|
|
// session would miss the worker's HashMap lookup.
|
|
#[cfg(unix)]
|
|
if did_cutover {
|
|
self.register_decrypt_worker_session(&node_addr);
|
|
}
|
|
#[cfg(not(unix))]
|
|
let _ = did_cutover;
|
|
}
|
|
|
|
// Execute drain completion
|
|
for node_addr in peers_to_drain {
|
|
// Extract the old index and transport_id under the peer
|
|
// borrow, then drop the borrow so the cache_key cleanup
|
|
// below can take &mut self for unregister_decrypt_worker_session.
|
|
let drained = self
|
|
.peers
|
|
.get_mut(&node_addr)
|
|
.and_then(|peer| peer.complete_drain().map(|idx| (idx, peer.transport_id())));
|
|
if let Some((old_our_index, transport_id)) = drained {
|
|
if let Some(tid) = transport_id {
|
|
let cache_key = (tid, old_our_index.as_u32());
|
|
self.peers_by_index.remove(&cache_key);
|
|
#[cfg(unix)]
|
|
self.unregister_decrypt_worker_session(cache_key);
|
|
}
|
|
let _ = self.index_allocator.free(old_our_index);
|
|
trace!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
old_index = %old_our_index,
|
|
"Drain complete, previous session erased"
|
|
);
|
|
}
|
|
}
|
|
|
|
// Initiate new rekeys
|
|
for node_addr in peers_to_rekey {
|
|
self.initiate_rekey(&node_addr).await;
|
|
}
|
|
}
|
|
|
|
/// Initiate an outbound rekey to a peer.
|
|
///
|
|
/// Creates a new IK handshake as initiator, sends msg1 over the existing
|
|
/// link (same transport, same remote address), and stores the handshake
|
|
/// state on the ActivePeer. No new Link or PeerConnection is created.
|
|
async fn initiate_rekey(&mut self, node_addr: &NodeAddr) {
|
|
let peer = match self.peers.get(node_addr) {
|
|
Some(p) => p,
|
|
None => return,
|
|
};
|
|
|
|
let transport_id = match peer.transport_id() {
|
|
Some(t) => t,
|
|
None => return,
|
|
};
|
|
let remote_addr = match peer.current_addr() {
|
|
Some(a) => a.clone(),
|
|
None => return,
|
|
};
|
|
let link_id = peer.link_id();
|
|
let peer_pubkey = peer.identity().pubkey_full();
|
|
|
|
// Allocate a new session index for the rekey
|
|
let our_index = match self.index_allocator.allocate() {
|
|
Ok(idx) => idx,
|
|
Err(e) => {
|
|
warn!(
|
|
peer = %self.peer_display_name(node_addr),
|
|
error = %e,
|
|
"Failed to allocate index for rekey"
|
|
);
|
|
return;
|
|
}
|
|
};
|
|
|
|
// Create IK initiator handshake directly (no PeerConnection)
|
|
// This frame's own copy of the node's long-term private key; the
|
|
// handshake state keeps its own and clears that on drop.
|
|
let mut our_keypair = self.identity().keypair();
|
|
let mut hs = HandshakeState::new_initiator(our_keypair, peer_pubkey);
|
|
our_keypair.non_secure_erase();
|
|
hs.set_local_epoch(self.startup_epoch());
|
|
|
|
let noise_msg1 = match hs.write_message_1() {
|
|
Ok(msg) => msg,
|
|
Err(e) => {
|
|
warn!(
|
|
peer = %self.peer_display_name(node_addr),
|
|
error = %e,
|
|
"Failed to generate rekey msg1"
|
|
);
|
|
let _ = self.index_allocator.free(our_index);
|
|
return;
|
|
}
|
|
};
|
|
|
|
let wire_msg1 = build_msg1(our_index, &noise_msg1);
|
|
|
|
// Send msg1 on the existing link (same transport + address)
|
|
if let Some(transport) = self.transports.get(&transport_id) {
|
|
match transport.send(&remote_addr, &wire_msg1).await {
|
|
Ok(_) => {
|
|
debug!(
|
|
peer = %self.peer_display_name(node_addr),
|
|
our_index = %our_index,
|
|
"Rekey initiated, sent msg1 on existing link"
|
|
);
|
|
}
|
|
Err(e) => {
|
|
warn!(
|
|
peer = %self.peer_display_name(node_addr),
|
|
error = %e,
|
|
"Failed to send rekey msg1"
|
|
);
|
|
let _ = self.index_allocator.free(our_index);
|
|
return;
|
|
}
|
|
}
|
|
}
|
|
|
|
// Store handshake state on the ActivePeer (not a separate PeerConnection)
|
|
let resend_interval = self.config().node.rate_limit.handshake_resend_interval_ms;
|
|
let now_ms = Self::now_ms();
|
|
if let Some(peer) = self.peers.get_mut(node_addr) {
|
|
peer.set_rekey_state(hs, our_index, wire_msg1, now_ms + resend_interval);
|
|
}
|
|
|
|
// Register in pending_outbound for msg2 dispatch (maps to existing link)
|
|
self.pending_outbound
|
|
.insert((transport_id, our_index.as_u32()), link_id);
|
|
}
|
|
|
|
/// Resend pending rekey msg1s and abandon timed-out rekeys.
|
|
///
|
|
/// Called from the tick loop. Uses the same resend interval and max
|
|
/// resend count as initial handshakes.
|
|
pub(in crate::node) async fn resend_pending_rekeys(&mut self, now_ms: u64) {
|
|
if !self.config().node.rekey.enabled {
|
|
return;
|
|
}
|
|
|
|
let interval_ms = self.config().node.rate_limit.handshake_resend_interval_ms;
|
|
let backoff = self.config().node.rate_limit.handshake_resend_backoff;
|
|
let max_resends = self.config().node.rate_limit.handshake_max_resends;
|
|
|
|
// Collect peers needing action
|
|
let mut to_resend: Vec<(NodeAddr, Vec<u8>)> = Vec::new();
|
|
let mut to_abandon: Vec<NodeAddr> = Vec::new();
|
|
|
|
for (node_addr, peer) in &self.peers {
|
|
if !peer.rekey_in_progress() || peer.rekey_msg1().is_none() {
|
|
continue;
|
|
}
|
|
if peer.rekey_msg1_resend_count() >= max_resends {
|
|
to_abandon.push(*node_addr);
|
|
continue;
|
|
}
|
|
if peer.needs_msg1_resend(now_ms) {
|
|
to_resend.push((*node_addr, peer.rekey_msg1().unwrap().to_vec()));
|
|
}
|
|
}
|
|
|
|
// Abandon rekey cycles that exhausted their retransmission budget.
|
|
for node_addr in to_abandon {
|
|
if let Some(peer) = self.peers.get_mut(&node_addr) {
|
|
peer.abandon_rekey();
|
|
}
|
|
debug!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
"FMP rekey aborted: msg1 unconfirmed after max retransmissions, abandoning cycle"
|
|
);
|
|
}
|
|
|
|
for (node_addr, msg1_bytes) in to_resend {
|
|
let (transport_id, remote_addr) = match self.peers.get(&node_addr) {
|
|
Some(p) => match (p.transport_id(), p.current_addr()) {
|
|
(Some(tid), Some(addr)) => (tid, addr.clone()),
|
|
_ => continue,
|
|
},
|
|
None => continue,
|
|
};
|
|
|
|
let sent = if let Some(transport) = self.transports.get(&transport_id) {
|
|
transport.send(&remote_addr, &msg1_bytes).await.is_ok()
|
|
} else {
|
|
false
|
|
};
|
|
|
|
if sent && let Some(peer) = self.peers.get_mut(&node_addr) {
|
|
let count = peer.rekey_msg1_resend_count() + 1;
|
|
let next = now_ms + (interval_ms as f64 * backoff.powi(count as i32)) as u64;
|
|
peer.record_rekey_msg1_resend(next);
|
|
trace!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
resend = count,
|
|
"Resent rekey msg1"
|
|
);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Retransmit FSP rekey msg3 until the responder is confirmed on the
|
|
/// new epoch.
|
|
///
|
|
/// Called from the tick loop. The rekey initiator retains its msg3
|
|
/// wire payload after the first send (`handle_session_ack`); this
|
|
/// driver resends it on the handshake resend interval (with backoff)
|
|
/// while the payload is still retained.
|
|
///
|
|
/// This is a **liveness-only** mechanism. Overlapping-epoch
|
|
/// trial-decrypt makes the rekey transition safe regardless of
|
|
/// cutover skew; retransmission only guarantees the responder
|
|
/// eventually derives the new session. Its lifetime is tied to the
|
|
/// responder *receiving* msg3 — the retained payload is cleared when
|
|
/// an inbound peer frame authenticates against `pending` or
|
|
/// post-cutover `current` — decoupled from the initiator's own
|
|
/// cutover. The initiator may cut over on its liveness timer while
|
|
/// the responder still lacks the new session; retransmission
|
|
/// continues, and overlapping-epoch decrypt keeps both directions
|
|
/// working meanwhile.
|
|
///
|
|
/// After `handshake_max_resends` attempts with no confirmed progress,
|
|
/// the rekey cycle is abandoned cleanly (`abandon_rekey`): the
|
|
/// pending session is dropped and the next cycle retries fresh. This
|
|
/// is safe — an abandoned cycle never leaves a divergent unsafe
|
|
/// state.
|
|
pub(in crate::node) async fn resend_pending_session_msg3(&mut self, now_ms: u64) {
|
|
if !self.config().node.rekey.enabled || self.sessions.is_empty() {
|
|
return;
|
|
}
|
|
|
|
let interval_ms = self.config().node.rate_limit.handshake_resend_interval_ms;
|
|
let backoff = self.config().node.rate_limit.handshake_resend_backoff;
|
|
let max_resends = self.config().node.rate_limit.handshake_max_resends;
|
|
let ttl = self.config().node.session.default_ttl;
|
|
let my_addr = *self.node_addr();
|
|
|
|
// Collect rekey initiators whose msg3 retransmission is due.
|
|
let mut to_resend: Vec<(NodeAddr, Vec<u8>)> = Vec::new();
|
|
let mut to_abandon: Vec<NodeAddr> = Vec::new();
|
|
|
|
for (node_addr, entry) in &self.sessions {
|
|
// Only the rekey initiator retains a msg3 payload.
|
|
let payload = match entry.rekey_msg3_payload() {
|
|
Some(p) => p,
|
|
None => continue,
|
|
};
|
|
if entry.rekey_msg3_next_resend_ms() == 0 || now_ms < entry.rekey_msg3_next_resend_ms()
|
|
{
|
|
continue;
|
|
}
|
|
if entry.rekey_msg3_resend_count() >= max_resends {
|
|
to_abandon.push(*node_addr);
|
|
continue;
|
|
}
|
|
to_resend.push((*node_addr, payload.to_vec()));
|
|
}
|
|
|
|
// Abandon rekey cycles that exhausted their retransmission budget.
|
|
for node_addr in to_abandon {
|
|
if let Some(entry) = self.sessions.get_mut(&node_addr) {
|
|
entry.abandon_rekey();
|
|
}
|
|
debug!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
"FSP rekey aborted: msg3 unconfirmed after max retransmissions, abandoning cycle"
|
|
);
|
|
}
|
|
|
|
// Retransmit msg3 for cycles still within budget.
|
|
for (node_addr, payload) in to_resend {
|
|
let mut datagram = SessionDatagram::new(my_addr, node_addr, payload).with_ttl(ttl);
|
|
let sent = match self.send_session_datagram(&mut datagram).await {
|
|
Ok(_) => true,
|
|
Err(e) => {
|
|
debug!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
error = %e,
|
|
"FSP rekey msg3 retransmission failed"
|
|
);
|
|
false
|
|
}
|
|
};
|
|
|
|
if sent && let Some(entry) = self.sessions.get_mut(&node_addr) {
|
|
let count = entry.rekey_msg3_resend_count() + 1;
|
|
let next = now_ms + (interval_ms as f64 * backoff.powi(count as i32)) as u64;
|
|
entry.record_rekey_msg3_resend(next);
|
|
trace!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
resend = count,
|
|
"Resent FSP rekey msg3"
|
|
);
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Periodic session (FSP) rekey check. Called from the tick loop.
|
|
///
|
|
/// For each established session:
|
|
/// - If the initiator holds a pending session past the liveness
|
|
/// timer, perform the K-bit cutover (overlapping-epoch decrypt
|
|
/// makes this safe on any schedule — see `FSP_CUTOVER_DELAY_MS`)
|
|
/// - If the drain window has expired, clean up the previous session
|
|
/// - If a responder-side handshake the peer never finished has aged
|
|
/// out, abandon it (the handshake only — a completed rekey session
|
|
/// is never discarded on a timer, see below)
|
|
/// - If the rekey timer/counter fires, initiate a new XK handshake
|
|
/// (this last one only when `node.rekey.enabled`)
|
|
///
|
|
/// msg3 retransmission is handled separately by
|
|
/// `resend_pending_session_msg3`; its lifetime is tied to the
|
|
/// responder receiving msg3, not to this initiator's cutover.
|
|
pub(in crate::node) async fn check_session_rekey(&mut self) {
|
|
// The cutover, drain and abandoned-rekey sweeps run whether or not
|
|
// periodic rekey is enabled: a peer's setup message is answered in
|
|
// either configuration, so both a superseded key epoch and an
|
|
// abandoned handshake can exist with rekey disabled. Only the
|
|
// trigger that starts a rekey of our own is gated.
|
|
let rekey_enabled = self.config().node.rekey.enabled;
|
|
|
|
let rekey_after_secs = self.config().node.rekey.after_secs;
|
|
let rekey_after_messages = self.config().node.rekey.after_messages;
|
|
let now_ms = Self::now_ms();
|
|
let drain_ms = DRAIN_WINDOW_SECS * 1000;
|
|
let drain_max_ms = drain_max_retention_ms(&self.config().node.rate_limit);
|
|
let dampening_ms = REKEY_DAMPENING_SECS * 1000;
|
|
|
|
let mut sessions_to_cutover: Vec<NodeAddr> = Vec::new();
|
|
let mut sessions_to_drain: Vec<NodeAddr> = Vec::new();
|
|
let mut sessions_to_rekey: Vec<NodeAddr> = Vec::new();
|
|
let mut handshakes_to_abandon: Vec<(NodeAddr, u64)> = Vec::new();
|
|
|
|
// Bound for a responder-side handshake the peer armed and never
|
|
// finished. A parked handshake blocks a later genuine setup message
|
|
// through the dual-initiation tie-break, so it clears on the
|
|
// handshake timeout. Nothing is lost with it: an armed handshake
|
|
// holds no key material either side can be using.
|
|
//
|
|
// A *completed* rekey has no such bound, and must not acquire one.
|
|
// A responder-side pending session exists only because a msg3
|
|
// authenticated by the session's own peer key arrived, and the
|
|
// initiator that sent that msg3 promotes the new epoch on an
|
|
// unconditional timer (`FSP_CUTOVER_DELAY_MS`, branch 1 below).
|
|
// The pending slot is therefore the epoch the peer has already
|
|
// moved to, not key material it abandoned, and discarding it on a
|
|
// timer makes every later frame from that peer undecryptable. It is
|
|
// released only by the events that supersede it: the peer's own
|
|
// frame promoting it (`handle_peer_kbit_flip`), a newer completed
|
|
// rekey replacing it, or the session going away.
|
|
let stale_handshake_ms = self.config().node.rate_limit.handshake_timeout_secs * 1000;
|
|
|
|
for (node_addr, entry) in &self.sessions {
|
|
if !entry.is_established() {
|
|
continue;
|
|
}
|
|
|
|
// 1. Initiator-side cutover (option A): completed rekey,
|
|
// pending session ready, liveness timer elapsed. This is
|
|
// an unconditional timer, NOT gated on responder progress —
|
|
// overlapping-epoch trial-decrypt covers the cutover skew,
|
|
// so flipping the K-bit here is always safe. An
|
|
// opportunistic early cutover also happens in
|
|
// `handle_encrypted_session_msg` if the initiator
|
|
// authenticates a peer frame against its own `pending`.
|
|
if entry.pending_new_session().is_some()
|
|
&& !entry.has_rekey_in_progress()
|
|
&& entry.is_rekey_initiator()
|
|
&& now_ms.saturating_sub(entry.rekey_completed_ms()) >= FSP_CUTOVER_DELAY_MS
|
|
{
|
|
sessions_to_cutover.push(*node_addr);
|
|
continue;
|
|
}
|
|
|
|
// 2. Drain window expiry
|
|
if entry.is_draining() && entry.drain_expired(now_ms, drain_ms, drain_max_ms) {
|
|
sessions_to_drain.push(*node_addr);
|
|
}
|
|
|
|
// 3. Abandon a responder-side handshake the peer never finished.
|
|
// Anchored on the peer's last accepted setup message, which is
|
|
// the only stamp this path writes. A pending session alongside
|
|
// it survives: only the handshake is dropped.
|
|
if !entry.is_rekey_initiator()
|
|
&& entry.has_rekey_in_progress()
|
|
&& entry.last_peer_rekey_ms() != 0
|
|
{
|
|
let age = now_ms.saturating_sub(entry.last_peer_rekey_ms());
|
|
if age > stale_handshake_ms {
|
|
handshakes_to_abandon.push((*node_addr, age));
|
|
continue;
|
|
}
|
|
}
|
|
|
|
// 4. Rekey trigger
|
|
if !rekey_enabled {
|
|
continue;
|
|
}
|
|
if entry.has_rekey_in_progress() {
|
|
continue;
|
|
}
|
|
if entry.pending_new_session().is_some() {
|
|
continue; // Pending session present, awaiting cutover
|
|
}
|
|
if entry.rekey_msg3_payload().is_some() {
|
|
// Initiator already cut over on its liveness timer but is
|
|
// still retransmitting msg3 to a responder not yet
|
|
// confirmed on the new epoch. Don't start another rekey
|
|
// until the current cycle's msg3 is delivered or abandoned.
|
|
continue;
|
|
}
|
|
if entry.is_rekey_dampened(now_ms, dampening_ms) {
|
|
continue;
|
|
}
|
|
|
|
let elapsed_secs = now_ms.saturating_sub(entry.session_start_ms()) / 1000;
|
|
let counter = entry.send_counter();
|
|
|
|
// Apply per-session symmetric jitter to desynchronize
|
|
// dual-initiation in symmetric-start meshes.
|
|
let effective_after_secs =
|
|
rekey_after_secs.saturating_add_signed(entry.rekey_jitter_secs());
|
|
if elapsed_secs >= effective_after_secs || counter >= rekey_after_messages {
|
|
sessions_to_rekey.push(*node_addr);
|
|
}
|
|
}
|
|
|
|
// Execute cutover for initiator side
|
|
for node_addr in sessions_to_cutover {
|
|
if let Some(entry) = self.sessions.get_mut(&node_addr)
|
|
&& entry.cutover_to_new_session(now_ms)
|
|
{
|
|
debug!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
"FSP rekey cutover complete (initiator), K-bit flipped"
|
|
);
|
|
}
|
|
}
|
|
|
|
// Execute drain completion
|
|
for node_addr in sessions_to_drain {
|
|
if let Some(entry) = self.sessions.get_mut(&node_addr) {
|
|
entry.complete_drain();
|
|
trace!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
"FSP drain complete, previous session erased"
|
|
);
|
|
}
|
|
}
|
|
|
|
// Abandon a handshake the peer armed and never finished. Cheap: no
|
|
// key material is lost, and the slot was blocking re-establishment.
|
|
for (node_addr, age_ms) in handshakes_to_abandon {
|
|
if let Some(entry) = self.sessions.get_mut(&node_addr) {
|
|
entry.abandon_handshake();
|
|
self.stats_mut().session.rekey_expired += 1;
|
|
info!(
|
|
peer = %self.peer_display_name(&node_addr),
|
|
age_ms,
|
|
"FSP rekey armed by peer expired without msg3, session retained"
|
|
);
|
|
}
|
|
}
|
|
|
|
// Initiate new rekeys
|
|
for node_addr in sessions_to_rekey {
|
|
self.initiate_session_rekey(&node_addr).await;
|
|
}
|
|
}
|
|
|
|
/// Initiate an FSP session rekey.
|
|
///
|
|
/// Creates a new XK handshake as initiator, sends SessionSetup msg1
|
|
/// through the mesh, and stores the handshake state on the existing entry.
|
|
async fn initiate_session_rekey(&mut self, dest_addr: &NodeAddr) {
|
|
// Check route availability before paying crypto cost
|
|
if self.find_next_hop(dest_addr).is_none() {
|
|
trace!(
|
|
peer = %self.peer_display_name(dest_addr),
|
|
"FSP rekey skipped: no route to destination"
|
|
);
|
|
return;
|
|
}
|
|
|
|
let entry = match self.sessions.get(dest_addr) {
|
|
Some(e) => e,
|
|
None => return,
|
|
};
|
|
let dest_pubkey = *entry.remote_pubkey();
|
|
|
|
// Create Noise XK initiator handshake
|
|
// This frame's own copy of the node's long-term private key; the
|
|
// handshake state keeps its own and clears that on drop.
|
|
let mut our_keypair = self.identity().keypair();
|
|
let mut handshake = HandshakeState::new_xk_initiator(our_keypair, dest_pubkey);
|
|
our_keypair.non_secure_erase();
|
|
handshake.set_local_epoch(self.startup_epoch());
|
|
|
|
let msg1 = match handshake.write_xk_message_1() {
|
|
Ok(m) => m,
|
|
Err(e) => {
|
|
warn!(
|
|
peer = %self.peer_display_name(dest_addr),
|
|
error = %e,
|
|
"Failed to generate FSP rekey XK msg1"
|
|
);
|
|
return;
|
|
}
|
|
};
|
|
|
|
// Build SessionSetup with coordinates
|
|
let our_coords = self.tree_state.my_coords().clone();
|
|
let dest_coords = self.get_dest_coords(dest_addr);
|
|
let setup = SessionSetup::new(our_coords, dest_coords).with_handshake(msg1);
|
|
let setup_payload = setup.encode();
|
|
|
|
// Send through the mesh
|
|
let my_addr = *self.node_addr();
|
|
let mut datagram = SessionDatagram::new(my_addr, *dest_addr, setup_payload)
|
|
.with_ttl(self.config().node.session.default_ttl);
|
|
|
|
if let Err(e) = self.send_session_datagram(&mut datagram).await {
|
|
debug!(
|
|
peer = %self.peer_display_name(dest_addr),
|
|
error = %e,
|
|
"Failed to send FSP rekey SessionSetup"
|
|
);
|
|
return;
|
|
}
|
|
|
|
// Store rekey state on the existing session entry
|
|
if let Some(entry) = self.sessions.get_mut(dest_addr) {
|
|
entry.set_rekey_state(handshake, true);
|
|
}
|
|
|
|
debug!(
|
|
peer = %self.peer_display_name(dest_addr),
|
|
"FSP rekey initiated, sent SessionSetup"
|
|
);
|
|
}
|
|
}
|