mirror of
https://github.com/jmcorgan/fips.git
synced 2026-10-05 19:18:25 +00:00
Carries show_peers stale reporting, derived from how long a peer has been silent. Carries the removal of the stored peer connectivity, which on this line also drops the always-true health conjunct from the msg3 cross-connection and rekey-responder gates and from the rekey scan, and drops the send filter from next-hop candidate selection. Carries per-connection ids and draining closes on TCP, Tor and Nym, so a crossed TCP handshake's msg3 is written before the losing connection closes. Carries the reuse flags on an adopted traversal socket, whose node test now drives this line's three-message handshake directly instead of one receive-loop pass per node. Carries the netmon test and the probe's measured syscall cost. The rekey msg2 rollback from maint is not carried here: its read rollback, handler arm, node test and changelog entry are written for IK, so this merge keeps this line's XX rekey msg2 path as it was, and the XX change follows in its own commit. Two tests reached the msg2 resend arm through a health lever production never sets. The node test now reaches it through a rekey declaration naming keys the responder no longer holds, after replacing the responder's keys the way a cross-connection swap does, and the core test declares a mismatched rekey on the peer's own link. A new core test covers a matching declaration with no session, so the rekey-responder gate's session check keeps a test that can fail. The establish characterisation test file and the IK duplicate-msg1 decision test stay out, as before. The conflicts were next's XX establish classification and its rewritten establish tests against master's edits to the IK versions; every region resolves to next's text with only the health and send-state reads removed and their doc comments reworded.
4500 lines
166 KiB
Rust
4500 lines
166 KiB
Rust
//! Integration tests for end-to-end Noise XX handshake scenarios.
|
|
|
|
use super::spanning_tree::{
|
|
cleanup_nodes, drain_all_packets, initiate_handshake, make_test_node_with_profile,
|
|
};
|
|
use super::*;
|
|
|
|
#[tokio::test]
|
|
async fn test_two_node_handshake_udp() {
|
|
use crate::config::UdpConfig;
|
|
use crate::proto::fmp::wire::{
|
|
build_encrypted, build_established_header, build_msg1, prepend_inner_header,
|
|
};
|
|
use crate::transport::udp::UdpTransport;
|
|
use tokio::time::{Duration, timeout};
|
|
|
|
// === Setup: Two nodes with UDP transports on localhost ===
|
|
|
|
let mut node_a = make_node();
|
|
let mut node_b = make_node();
|
|
|
|
let transport_id_a = TransportId::new(1);
|
|
let transport_id_b = TransportId::new(1);
|
|
|
|
let udp_config = UdpConfig {
|
|
bind_addr: Some("127.0.0.1:0".to_string()),
|
|
mtu: Some(1280),
|
|
..Default::default()
|
|
};
|
|
|
|
let (packet_tx_a, mut packet_rx_a) = packet_channel(64);
|
|
let (packet_tx_b, mut packet_rx_b) = packet_channel(64);
|
|
|
|
let mut transport_a = UdpTransport::new(transport_id_a, None, udp_config.clone(), packet_tx_a);
|
|
let mut transport_b = UdpTransport::new(transport_id_b, None, udp_config, packet_tx_b);
|
|
|
|
transport_a.start_async().await.unwrap();
|
|
transport_b.start_async().await.unwrap();
|
|
|
|
let addr_a = transport_a.local_addr().unwrap();
|
|
let addr_b = transport_b.local_addr().unwrap();
|
|
let remote_addr_b = TransportAddr::from_string(&addr_b.to_string());
|
|
let remote_addr_a = TransportAddr::from_string(&addr_a.to_string());
|
|
|
|
node_a
|
|
.transports
|
|
.insert(transport_id_a, TransportHandle::Udp(transport_a));
|
|
node_b
|
|
.transports
|
|
.insert(transport_id_b, TransportHandle::Udp(transport_b));
|
|
|
|
// === Phase 1: Node A initiates handshake to Node B ===
|
|
|
|
// Create peer identity for B (must use full key for ECDH parity)
|
|
let peer_b_identity = PeerIdentity::from_pubkey_full(node_b.identity().pubkey_full());
|
|
let peer_b_node_addr = *peer_b_identity.node_addr();
|
|
|
|
let link_id_a = node_a.allocate_link_id();
|
|
|
|
// Allocate session index for A's outbound
|
|
let our_index_a = node_a.index_allocator.allocate().unwrap();
|
|
|
|
node_a
|
|
.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id_a, peer_b_identity, 1000)
|
|
.with_our_index(our_index_a)
|
|
.with_transport_id(transport_id_a)
|
|
.with_source_addr(remote_addr_b.clone()),
|
|
)
|
|
.unwrap();
|
|
|
|
// Start handshake (generates Noise XX msg1)
|
|
let our_keypair_a = node_a.identity().keypair();
|
|
let startup_epoch_a = node_a.startup_epoch();
|
|
let noise_msg1 = node_a
|
|
.peer_machines
|
|
.get_mut(&link_id_a)
|
|
.unwrap()
|
|
.start_handshake(our_keypair_a, startup_epoch_a, 1000)
|
|
.unwrap();
|
|
|
|
// Build wire msg1 and track in node state
|
|
let wire_msg1 = build_msg1(our_index_a, &noise_msg1);
|
|
|
|
let link_a = Link::connectionless(
|
|
link_id_a,
|
|
transport_id_a,
|
|
remote_addr_b.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node_a.links.insert(link_id_a, link_a);
|
|
node_a
|
|
.pending_outbound
|
|
.insert((transport_id_a, our_index_a.as_u32()), link_id_a);
|
|
|
|
// Send msg1 from A to B over UDP
|
|
let transport = node_a.transports.get(&transport_id_a).unwrap();
|
|
transport
|
|
.send(&remote_addr_b, &wire_msg1)
|
|
.await
|
|
.expect("Failed to send msg1");
|
|
|
|
// === Phase 2: Node B receives msg1, sends msg2 (XX: does NOT promote yet) ===
|
|
|
|
let packet_b = timeout(Duration::from_secs(1), packet_rx_b.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg1")
|
|
.expect("Channel closed");
|
|
|
|
node_b.handle_msg1(packet_b).await;
|
|
|
|
let peer_a_node_addr =
|
|
*PeerIdentity::from_pubkey_full(node_a.identity().pubkey_full()).node_addr();
|
|
|
|
// XX: B has NOT promoted yet (needs msg3)
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
0,
|
|
"Node B should have 0 peers after msg1 (XX awaits msg3)"
|
|
);
|
|
assert_eq!(
|
|
node_b.connection_count(),
|
|
1,
|
|
"Node B should have 1 pending connection awaiting msg3"
|
|
);
|
|
|
|
// === Phase 3: Node A receives msg2, sends msg3, promotes ===
|
|
|
|
let packet_a = timeout(Duration::from_secs(1), packet_rx_a.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg2")
|
|
.expect("Channel closed");
|
|
|
|
node_a.handle_msg2(packet_a).await;
|
|
|
|
// Verify A promoted the outbound connection
|
|
assert_eq!(
|
|
node_a.peer_count(),
|
|
1,
|
|
"Node A should have 1 peer after msg2"
|
|
);
|
|
let peer_b_on_a = node_a
|
|
.get_peer(&peer_b_node_addr)
|
|
.expect("Node A should have peer B");
|
|
assert!(
|
|
peer_b_on_a.has_session(),
|
|
"Peer B on A should have NoiseSession"
|
|
);
|
|
assert_eq!(
|
|
peer_b_on_a.our_index(),
|
|
Some(our_index_a),
|
|
"Peer B on A should have our_index matching what we allocated"
|
|
);
|
|
assert!(
|
|
node_a
|
|
.peers_by_index
|
|
.contains_key(&(transport_id_a, our_index_a.as_u32())),
|
|
"Node A peers_by_index should be populated"
|
|
);
|
|
|
|
// === Phase 4: Node B receives msg3, promotes ===
|
|
|
|
let packet_b_msg3 = timeout(Duration::from_secs(1), packet_rx_b.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg3")
|
|
.expect("Channel closed");
|
|
|
|
node_b.handle_msg3(packet_b_msg3).await;
|
|
|
|
// Verify B promoted after msg3
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
1,
|
|
"Node B should have 1 peer after msg3"
|
|
);
|
|
let peer_a_on_b = node_b
|
|
.get_peer(&peer_a_node_addr)
|
|
.expect("Node B should have peer A");
|
|
assert!(
|
|
peer_a_on_b.has_session(),
|
|
"Peer A on B should have NoiseSession"
|
|
);
|
|
let our_index_b = peer_a_on_b.our_index().expect("B should have our_index");
|
|
assert!(
|
|
node_b
|
|
.peers_by_index
|
|
.contains_key(&(transport_id_b, our_index_b.as_u32())),
|
|
"Node B peers_by_index should be populated"
|
|
);
|
|
|
|
// === Phase 4: Encrypted frame A → B ===
|
|
|
|
// A encrypts a test message and sends to B
|
|
// Prepend inner header (timestamp + msg_type) as the real send path does
|
|
let msg_a = b"\x10test from A"; // msg_type 0x10 (TreeAnnounce) + dummy payload
|
|
let inner_a = prepend_inner_header(0, msg_a);
|
|
let peer_b = node_a.get_peer_mut(&peer_b_node_addr).unwrap();
|
|
let their_index_b = peer_b.their_index().expect("A should know B's index");
|
|
let session_a = peer_b.noise_session_mut().unwrap();
|
|
let counter_a = session_a.current_send_counter();
|
|
let header_a = build_established_header(their_index_b, counter_a, 0, inner_a.len() as u16);
|
|
let ciphertext_a = session_a.encrypt_with_aad(&inner_a, &header_a).unwrap();
|
|
|
|
let wire_encrypted = build_encrypted(&header_a, &ciphertext_a);
|
|
let transport = node_a.transports.get(&transport_id_a).unwrap();
|
|
transport
|
|
.send(&remote_addr_b, &wire_encrypted)
|
|
.await
|
|
.expect("Failed to send encrypted frame");
|
|
|
|
// B receives and decrypts
|
|
let encrypted_packet_b = timeout(Duration::from_secs(1), packet_rx_b.recv())
|
|
.await
|
|
.expect("Timeout waiting for encrypted frame")
|
|
.expect("Channel closed");
|
|
|
|
node_b.handle_encrypted_frame(encrypted_packet_b).await;
|
|
|
|
// === Phase 5: Encrypted frame B → A ===
|
|
|
|
// Prepend inner header (timestamp + msg_type) as the real send path does
|
|
let msg_b = b"\x10test from B"; // msg_type 0x10 (TreeAnnounce) + dummy payload
|
|
let inner_b = prepend_inner_header(0, msg_b);
|
|
let peer_a = node_b.get_peer_mut(&peer_a_node_addr).unwrap();
|
|
let their_index_a = peer_a.their_index().expect("B should know A's index");
|
|
let session_b = peer_a.noise_session_mut().unwrap();
|
|
let counter_b = session_b.current_send_counter();
|
|
let header_b = build_established_header(their_index_a, counter_b, 0, inner_b.len() as u16);
|
|
let ciphertext_b = session_b.encrypt_with_aad(&inner_b, &header_b).unwrap();
|
|
|
|
let wire_encrypted_b = build_encrypted(&header_b, &ciphertext_b);
|
|
let transport = node_b.transports.get(&transport_id_b).unwrap();
|
|
transport
|
|
.send(&remote_addr_a, &wire_encrypted_b)
|
|
.await
|
|
.expect("Failed to send encrypted frame B→A");
|
|
|
|
// A receives and decrypts
|
|
let encrypted_packet_a = timeout(Duration::from_secs(1), packet_rx_a.recv())
|
|
.await
|
|
.expect("Timeout waiting for encrypted frame B→A")
|
|
.expect("Channel closed");
|
|
|
|
node_a.handle_encrypted_frame(encrypted_packet_a).await;
|
|
|
|
// Clean up transports
|
|
for (_, t) in node_a.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
for (_, t) in node_b.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
}
|
|
|
|
/// Integration test: two nodes complete a handshake via run_rx_loop.
|
|
///
|
|
/// Unlike test_two_node_handshake_udp which calls handle_msg1/handle_msg2
|
|
/// directly, this test exercises the full rx loop dispatch path:
|
|
/// UDP socket → packet channel → run_rx_loop → process_packet →
|
|
/// discriminator dispatch → handler.
|
|
#[tokio::test]
|
|
async fn test_run_rx_loop_handshake() {
|
|
use crate::config::UdpConfig;
|
|
use crate::proto::fmp::wire::build_msg1;
|
|
use crate::transport::udp::UdpTransport;
|
|
use tokio::time::Duration;
|
|
|
|
// === Setup: Two nodes with UDP transports on localhost ===
|
|
|
|
let mut node_a = make_node();
|
|
let mut node_b = make_node();
|
|
|
|
let transport_id_a = TransportId::new(1);
|
|
let transport_id_b = TransportId::new(1);
|
|
|
|
let udp_config = UdpConfig {
|
|
bind_addr: Some("127.0.0.1:0".to_string()),
|
|
mtu: Some(1280),
|
|
..Default::default()
|
|
};
|
|
|
|
let (packet_tx_a, packet_rx_a) = packet_channel(64);
|
|
let (packet_tx_b, packet_rx_b) = packet_channel(64);
|
|
|
|
let mut transport_a = UdpTransport::new(transport_id_a, None, udp_config.clone(), packet_tx_a);
|
|
let mut transport_b = UdpTransport::new(transport_id_b, None, udp_config, packet_tx_b);
|
|
|
|
transport_a.start_async().await.unwrap();
|
|
transport_b.start_async().await.unwrap();
|
|
|
|
let addr_b = transport_b.local_addr().unwrap();
|
|
let remote_addr_b = TransportAddr::from_string(&addr_b.to_string());
|
|
|
|
node_a
|
|
.transports
|
|
.insert(transport_id_a, TransportHandle::Udp(transport_a));
|
|
node_b
|
|
.transports
|
|
.insert(transport_id_b, TransportHandle::Udp(transport_b));
|
|
|
|
// Store packet_rx on nodes for run_rx_loop
|
|
node_a.packet_rx = Some(packet_rx_a);
|
|
node_b.packet_rx = Some(packet_rx_b);
|
|
|
|
// Set node state to Running (transports need to be operational)
|
|
node_a.supervisor.state = NodeState::Running;
|
|
node_b.supervisor.state = NodeState::Running;
|
|
|
|
// === Phase 1: Node A initiates handshake to Node B ===
|
|
|
|
let peer_b_identity = PeerIdentity::from_pubkey_full(node_b.identity().pubkey_full());
|
|
let peer_b_node_addr = *peer_b_identity.node_addr();
|
|
|
|
let link_id_a = node_a.allocate_link_id();
|
|
|
|
let our_index_a = node_a.index_allocator.allocate().unwrap();
|
|
node_a
|
|
.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id_a, peer_b_identity, 1000)
|
|
.with_our_index(our_index_a)
|
|
.with_transport_id(transport_id_a)
|
|
.with_source_addr(remote_addr_b.clone()),
|
|
)
|
|
.unwrap();
|
|
let our_keypair_a = node_a.identity().keypair();
|
|
let startup_epoch_a = node_a.startup_epoch();
|
|
let noise_msg1 = node_a
|
|
.peer_machines
|
|
.get_mut(&link_id_a)
|
|
.unwrap()
|
|
.start_handshake(our_keypair_a, startup_epoch_a, 1000)
|
|
.unwrap();
|
|
|
|
let wire_msg1 = build_msg1(our_index_a, &noise_msg1);
|
|
|
|
let link_a = Link::connectionless(
|
|
link_id_a,
|
|
transport_id_a,
|
|
remote_addr_b.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node_a.links.insert(link_id_a, link_a);
|
|
node_a
|
|
.pending_outbound
|
|
.insert((transport_id_a, our_index_a.as_u32()), link_id_a);
|
|
|
|
// Send msg1 from A to B over real UDP
|
|
let transport = node_a.transports.get(&transport_id_a).unwrap();
|
|
transport
|
|
.send(&remote_addr_b, &wire_msg1)
|
|
.await
|
|
.expect("Failed to send msg1");
|
|
|
|
// Small delay to ensure msg1 is received by B's transport
|
|
tokio::time::sleep(Duration::from_millis(50)).await;
|
|
|
|
// === Phase 2: Run Node B's rx loop (processes msg1 and later msg3) ===
|
|
//
|
|
// This is the key difference from test_two_node_handshake_udp:
|
|
// instead of calling handle_msg1() directly, we run the full rx loop
|
|
// which dispatches based on the common prefix phase field.
|
|
//
|
|
// With XX, the rx loop will process msg1 (sending msg2) but NOT
|
|
// promote B yet (needs msg3). We run the rx loop once for msg1,
|
|
// then later use direct handler calls for msg3 (since run_rx_loop
|
|
// takes packet_rx and can't be called twice).
|
|
|
|
tokio::select! {
|
|
result = node_b.run_rx_loop() => {
|
|
panic!("Node B rx loop exited unexpectedly: {:?}", result);
|
|
}
|
|
_ = tokio::time::sleep(Duration::from_millis(500)) => {
|
|
// Timeout: rx loop processed available packets
|
|
}
|
|
}
|
|
|
|
// XX: Node B has NOT promoted yet (needs msg3)
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
0,
|
|
"Node B should have 0 peers after rx loop processed msg1 (XX awaits msg3)"
|
|
);
|
|
assert_eq!(
|
|
node_b.connection_count(),
|
|
1,
|
|
"Node B should have 1 pending connection"
|
|
);
|
|
|
|
// === Phase 3: Run Node A's rx loop (processes msg2, sends msg3) ===
|
|
|
|
tokio::select! {
|
|
result = node_a.run_rx_loop() => {
|
|
panic!("Node A rx loop exited unexpectedly: {:?}", result);
|
|
}
|
|
_ = tokio::time::sleep(Duration::from_millis(500)) => {
|
|
// Timeout: rx loop processed msg2
|
|
}
|
|
}
|
|
|
|
// Verify Node A promoted after processing msg2
|
|
assert_eq!(
|
|
node_a.peer_count(),
|
|
1,
|
|
"Node A should have 1 peer after rx loop processed msg2"
|
|
);
|
|
let peer_b_on_a = node_a
|
|
.get_peer(&peer_b_node_addr)
|
|
.expect("Node A should have peer B");
|
|
assert!(
|
|
peer_b_on_a.has_session(),
|
|
"Peer B on A should have NoiseSession"
|
|
);
|
|
assert_eq!(
|
|
peer_b_on_a.our_index(),
|
|
Some(our_index_a),
|
|
"Peer B on A should have our_index matching what we allocated"
|
|
);
|
|
assert!(
|
|
peer_b_on_a.their_index().is_some(),
|
|
"A should know B's index"
|
|
);
|
|
assert!(
|
|
node_a
|
|
.peers_by_index
|
|
.contains_key(&(transport_id_a, our_index_a.as_u32())),
|
|
"Node A peers_by_index should be populated"
|
|
);
|
|
|
|
// Note: Phase 4 (msg3 → B promotes) cannot be tested via run_rx_loop
|
|
// because it consumes packet_rx on first call. The msg3 dispatch is
|
|
// verified by test_two_node_handshake_udp which uses direct handler calls.
|
|
// This test verifies rx_loop correctly dispatches PHASE_MSG1 (Phase 2)
|
|
// and PHASE_MSG2 (Phase 3). B still has a pending connection awaiting msg3.
|
|
assert_eq!(
|
|
node_b.connection_count(),
|
|
1,
|
|
"Node B should still have pending connection awaiting msg3"
|
|
);
|
|
|
|
// Clean up transports
|
|
for (_, t) in node_a.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
for (_, t) in node_b.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
}
|
|
|
|
/// Integration test: simultaneous cross-connection (both nodes initiate).
|
|
///
|
|
/// Simulates the live scenario where both nodes have auto_connect to each other.
|
|
/// Both send msg1 simultaneously, creating a cross-connection that must be
|
|
/// resolved by the tie-breaker rule. Exercises the addr_to_link fix that allows
|
|
/// inbound msg1 when an outbound link to the same address already exists.
|
|
#[tokio::test]
|
|
async fn test_cross_connection_both_initiate() {
|
|
use crate::config::UdpConfig;
|
|
use crate::proto::fmp::wire::build_msg1;
|
|
use crate::transport::udp::UdpTransport;
|
|
use tokio::time::{Duration, timeout};
|
|
|
|
// === Setup: Two nodes with UDP transports on localhost ===
|
|
|
|
let mut node_a = make_node();
|
|
let mut node_b = make_node();
|
|
|
|
let transport_id_a = TransportId::new(1);
|
|
let transport_id_b = TransportId::new(1);
|
|
|
|
let udp_config = UdpConfig {
|
|
bind_addr: Some("127.0.0.1:0".to_string()),
|
|
mtu: Some(1280),
|
|
..Default::default()
|
|
};
|
|
|
|
let (packet_tx_a, mut packet_rx_a) = packet_channel(64);
|
|
let (packet_tx_b, mut packet_rx_b) = packet_channel(64);
|
|
|
|
let mut transport_a = UdpTransport::new(transport_id_a, None, udp_config.clone(), packet_tx_a);
|
|
let mut transport_b = UdpTransport::new(transport_id_b, None, udp_config, packet_tx_b);
|
|
|
|
transport_a.start_async().await.unwrap();
|
|
transport_b.start_async().await.unwrap();
|
|
|
|
let addr_a = transport_a.local_addr().unwrap();
|
|
let addr_b = transport_b.local_addr().unwrap();
|
|
let remote_addr_b = TransportAddr::from_string(&addr_b.to_string());
|
|
let remote_addr_a = TransportAddr::from_string(&addr_a.to_string());
|
|
|
|
node_a
|
|
.transports
|
|
.insert(transport_id_a, TransportHandle::Udp(transport_a));
|
|
node_b
|
|
.transports
|
|
.insert(transport_id_b, TransportHandle::Udp(transport_b));
|
|
|
|
// Peer identities (must use full key for ECDH parity)
|
|
let peer_b_identity = PeerIdentity::from_pubkey_full(node_b.identity().pubkey_full());
|
|
let peer_b_node_addr = *peer_b_identity.node_addr();
|
|
let peer_a_identity = PeerIdentity::from_pubkey_full(node_a.identity().pubkey_full());
|
|
let peer_a_node_addr = *peer_a_identity.node_addr();
|
|
|
|
// === Phase 1: Both nodes initiate handshakes (simulate auto_connect) ===
|
|
|
|
// Node A initiates to Node B
|
|
let link_id_a_out = node_a.allocate_link_id();
|
|
let our_index_a = node_a.index_allocator.allocate().unwrap();
|
|
node_a
|
|
.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id_a_out, peer_b_identity, 1000)
|
|
.with_our_index(our_index_a)
|
|
.with_transport_id(transport_id_a)
|
|
.with_source_addr(remote_addr_b.clone()),
|
|
)
|
|
.unwrap();
|
|
let our_keypair_a = node_a.identity().keypair();
|
|
let startup_epoch_a = node_a.startup_epoch();
|
|
let noise_msg1_a = node_a
|
|
.peer_machines
|
|
.get_mut(&link_id_a_out)
|
|
.unwrap()
|
|
.start_handshake(our_keypair_a, startup_epoch_a, 1000)
|
|
.unwrap();
|
|
|
|
let wire_msg1_a = build_msg1(our_index_a, &noise_msg1_a);
|
|
|
|
let link_a_out = Link::connectionless(
|
|
link_id_a_out,
|
|
transport_id_a,
|
|
remote_addr_b.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node_a.links.insert(link_id_a_out, link_a_out);
|
|
node_a
|
|
.addr_to_link
|
|
.insert((transport_id_a, remote_addr_b.clone()), link_id_a_out);
|
|
node_a
|
|
.pending_outbound
|
|
.insert((transport_id_a, our_index_a.as_u32()), link_id_a_out);
|
|
|
|
// Node B initiates to Node A
|
|
let link_id_b_out = node_b.allocate_link_id();
|
|
let our_index_b = node_b.index_allocator.allocate().unwrap();
|
|
node_b
|
|
.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id_b_out, peer_a_identity, 1000)
|
|
.with_our_index(our_index_b)
|
|
.with_transport_id(transport_id_b)
|
|
.with_source_addr(remote_addr_a.clone()),
|
|
)
|
|
.unwrap();
|
|
let our_keypair_b = node_b.identity().keypair();
|
|
let startup_epoch_b = node_b.startup_epoch();
|
|
let noise_msg1_b = node_b
|
|
.peer_machines
|
|
.get_mut(&link_id_b_out)
|
|
.unwrap()
|
|
.start_handshake(our_keypair_b, startup_epoch_b, 1000)
|
|
.unwrap();
|
|
|
|
let wire_msg1_b = build_msg1(our_index_b, &noise_msg1_b);
|
|
|
|
let link_b_out = Link::connectionless(
|
|
link_id_b_out,
|
|
transport_id_b,
|
|
remote_addr_a.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node_b.links.insert(link_id_b_out, link_b_out);
|
|
node_b
|
|
.addr_to_link
|
|
.insert((transport_id_b, remote_addr_a.clone()), link_id_b_out);
|
|
node_b
|
|
.pending_outbound
|
|
.insert((transport_id_b, our_index_b.as_u32()), link_id_b_out);
|
|
|
|
// Both send msg1 over UDP
|
|
let transport = node_a.transports.get(&transport_id_a).unwrap();
|
|
transport
|
|
.send(&remote_addr_b, &wire_msg1_a)
|
|
.await
|
|
.expect("A send msg1");
|
|
|
|
let transport = node_b.transports.get(&transport_id_b).unwrap();
|
|
transport
|
|
.send(&remote_addr_a, &wire_msg1_b)
|
|
.await
|
|
.expect("B send msg1");
|
|
|
|
// === Phase 2: Both nodes receive the other's msg1 (XX: no promotion yet) ===
|
|
|
|
// B receives A's msg1
|
|
let packet_at_b = timeout(Duration::from_secs(1), packet_rx_b.recv())
|
|
.await
|
|
.expect("Timeout")
|
|
.expect("Channel closed");
|
|
node_b.handle_msg1(packet_at_b).await;
|
|
|
|
// XX: B has NOT promoted yet (needs msg3 from A)
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
0,
|
|
"Node B should have 0 peers after processing A's msg1 (XX)"
|
|
);
|
|
|
|
// A receives B's msg1
|
|
let packet_at_a = timeout(Duration::from_secs(1), packet_rx_a.recv())
|
|
.await
|
|
.expect("Timeout")
|
|
.expect("Channel closed");
|
|
node_a.handle_msg1(packet_at_a).await;
|
|
|
|
// XX: A has NOT promoted yet (needs msg3 from B)
|
|
assert_eq!(
|
|
node_a.peer_count(),
|
|
0,
|
|
"Node A should have 0 peers after processing B's msg1 (XX)"
|
|
);
|
|
|
|
// === Phase 3: Both nodes receive msg2 + send msg3, initiator side promotes ===
|
|
|
|
// A receives B's msg2 (response to A's original msg1) → A sends msg3, A promotes
|
|
let msg2_at_a = timeout(Duration::from_secs(1), packet_rx_a.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg2 at A")
|
|
.expect("Channel closed");
|
|
node_a.handle_msg2(msg2_at_a).await;
|
|
|
|
// A promoted as initiator
|
|
assert_eq!(
|
|
node_a.peer_count(),
|
|
1,
|
|
"Node A should have 1 peer after processing msg2"
|
|
);
|
|
|
|
// B receives A's msg2 (response to B's original msg1) → B sends msg3, B promotes
|
|
let msg2_at_b = timeout(Duration::from_secs(1), packet_rx_b.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg2 at B")
|
|
.expect("Channel closed");
|
|
node_b.handle_msg2(msg2_at_b).await;
|
|
|
|
// B promoted as initiator
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
1,
|
|
"Node B should have 1 peer after processing msg2"
|
|
);
|
|
|
|
// === Phase 4: Both nodes receive msg3, responder side completes ===
|
|
// Cross-connection resolution happens here (or in Phase 3 promotion).
|
|
|
|
// A receives B's msg3 (B completing A's inbound handshake)
|
|
let msg3_at_a = timeout(Duration::from_secs(1), packet_rx_a.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg3 at A")
|
|
.expect("Channel closed");
|
|
node_a.handle_msg3(msg3_at_a).await;
|
|
|
|
// B receives A's msg3 (A completing B's inbound handshake)
|
|
let msg3_at_b = timeout(Duration::from_secs(1), packet_rx_b.recv())
|
|
.await
|
|
.expect("Timeout waiting for msg3 at B")
|
|
.expect("Channel closed");
|
|
node_b.handle_msg3(msg3_at_b).await;
|
|
|
|
// === Verification ===
|
|
// Both nodes should have exactly 1 peer each after cross-connection resolution
|
|
assert_eq!(
|
|
node_a.peer_count(),
|
|
1,
|
|
"Node A should have exactly 1 peer after cross-connection"
|
|
);
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
1,
|
|
"Node B should have exactly 1 peer after cross-connection"
|
|
);
|
|
|
|
let peer_b_on_a = node_a
|
|
.get_peer(&peer_b_node_addr)
|
|
.expect("A should have peer B");
|
|
let peer_a_on_b = node_b
|
|
.get_peer(&peer_a_node_addr)
|
|
.expect("B should have peer A");
|
|
|
|
assert!(peer_b_on_a.has_session(), "Peer B on A should have session");
|
|
assert!(peer_a_on_b.has_session(), "Peer A on B should have session");
|
|
|
|
// The property the tie-break exists to produce: both ends kept the SAME
|
|
// session, not merely a session each. The index pair is what makes that
|
|
// checkable — A sends to B on the index B receives on, and vice versa.
|
|
//
|
|
// Every per-end predicate above holds when the two ends resolve onto
|
|
// different sessions, because each end does have a healthy session; the
|
|
// link then decrypts one direction and silently drops the other. So the
|
|
// pairing has to be asserted across the two nodes or it is not asserted at
|
|
// all. Kept as a standing guard on the tie-break rather than as a
|
|
// reproduction of any particular defect: it passes on the pre-marker tree
|
|
// as well, where this ordering was already resolved correctly.
|
|
assert_eq!(
|
|
peer_b_on_a.their_index(),
|
|
peer_a_on_b.our_index(),
|
|
"A sends to B on an index B does not receive on: the ends diverged"
|
|
);
|
|
assert_eq!(
|
|
peer_a_on_b.their_index(),
|
|
peer_b_on_a.our_index(),
|
|
"B sends to A on an index A does not receive on: the ends diverged"
|
|
);
|
|
assert!(
|
|
peer_b_on_a.our_index().is_some() && peer_b_on_a.their_index().is_some(),
|
|
"both indices must be set, or the equality above passes on None == None"
|
|
);
|
|
|
|
// Clean up transports
|
|
for (_, t) in node_a.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
for (_, t) in node_b.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
}
|
|
|
|
/// Test that stale handshake connections are cleaned up by check_timeouts().
|
|
///
|
|
/// Simulates the scenario where a node initiates a handshake to a peer that
|
|
/// isn't running. The outbound connection should be cleaned up after the
|
|
/// handshake timeout expires.
|
|
#[tokio::test]
|
|
async fn test_stale_connection_cleanup() {
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
let peer_identity = make_peer_identity();
|
|
let remote_addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
// Create outbound connection with a timestamp far in the past
|
|
let past_time_ms = 1000; // A very early timestamp
|
|
let link_id = node.allocate_link_id();
|
|
|
|
// Allocate session index and set transport info
|
|
let our_index = node.index_allocator.allocate().unwrap();
|
|
node.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id, peer_identity, past_time_ms)
|
|
.with_our_index(our_index)
|
|
.with_transport_id(transport_id)
|
|
.with_source_addr(remote_addr.clone()),
|
|
)
|
|
.unwrap();
|
|
let our_keypair = node.identity().keypair();
|
|
let startup_epoch = node.startup_epoch();
|
|
let _noise_msg1 = node
|
|
.peer_machines
|
|
.get_mut(&link_id)
|
|
.unwrap()
|
|
.start_handshake(our_keypair, startup_epoch, past_time_ms)
|
|
.unwrap();
|
|
|
|
// Set up all the state that initiate_peer_connection would create
|
|
let link = Link::connectionless(
|
|
link_id,
|
|
transport_id,
|
|
remote_addr.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node.links.insert(link_id, link);
|
|
node.addr_to_link
|
|
.insert((transport_id, remote_addr.clone()), link_id);
|
|
node.pending_outbound
|
|
.insert((transport_id, our_index.as_u32()), link_id);
|
|
|
|
// Verify state before timeout check
|
|
assert_eq!(node.connection_count(), 1);
|
|
assert_eq!(node.link_count(), 1);
|
|
assert!(
|
|
node.pending_outbound
|
|
.contains_key(&(transport_id, our_index.as_u32()))
|
|
);
|
|
assert_eq!(node.index_allocator.count(), 1);
|
|
|
|
// Connection was created at time 1000ms. check_timeouts uses SystemTime::now(),
|
|
// which is far beyond the 30s timeout. The connection should be cleaned up.
|
|
node.check_timeouts().await;
|
|
|
|
// Verify everything was cleaned up
|
|
assert_eq!(
|
|
node.connection_count(),
|
|
0,
|
|
"Stale connection should be removed"
|
|
);
|
|
assert_eq!(node.link_count(), 0, "Stale link should be removed");
|
|
assert!(
|
|
!node
|
|
.pending_outbound
|
|
.contains_key(&(transport_id, our_index.as_u32())),
|
|
"pending_outbound should be cleaned up"
|
|
);
|
|
assert_eq!(
|
|
node.index_allocator.count(),
|
|
0,
|
|
"Session index should be freed"
|
|
);
|
|
assert!(
|
|
!node.addr_to_link.contains_key(&(transport_id, remote_addr)),
|
|
"addr_to_link should be cleaned up"
|
|
);
|
|
}
|
|
|
|
/// Test that failed connections are cleaned up by check_timeouts().
|
|
#[tokio::test]
|
|
async fn test_failed_connection_cleanup() {
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
let peer_identity = make_peer_identity();
|
|
let remote_addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
// Create a connection and mark it failed (simulating a send failure)
|
|
let now_ms = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_millis() as u64)
|
|
.unwrap_or(0);
|
|
let link_id = node.allocate_link_id();
|
|
|
|
let our_index = node.index_allocator.allocate().unwrap();
|
|
node.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id, peer_identity, now_ms)
|
|
.with_our_index(our_index)
|
|
.with_transport_id(transport_id)
|
|
.with_source_addr(remote_addr.clone()),
|
|
)
|
|
.unwrap();
|
|
let our_keypair = node.identity().keypair();
|
|
let startup_epoch = node.startup_epoch();
|
|
let _noise_msg1 = node
|
|
.peer_machines
|
|
.get_mut(&link_id)
|
|
.unwrap()
|
|
.start_handshake(our_keypair, startup_epoch, now_ms)
|
|
.unwrap();
|
|
|
|
let link = Link::connectionless(
|
|
link_id,
|
|
transport_id,
|
|
remote_addr.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node.links.insert(link_id, link);
|
|
node.addr_to_link
|
|
.insert((transport_id, remote_addr.clone()), link_id);
|
|
node.pending_outbound
|
|
.insert((transport_id, our_index.as_u32()), link_id);
|
|
|
|
// Simulate a stored-handshake send failure through the control machine —
|
|
// the failure carrier the stale-connection sweep now reads (the leg no
|
|
// longer carries a failed phase of its own).
|
|
{
|
|
let machine = node
|
|
.peer_machines
|
|
.get_mut(&link_id)
|
|
.expect("machine seeded by the handshake seeder");
|
|
let alloc = &mut node.index_allocator;
|
|
let actions = machine.step(
|
|
crate::peer::machine::PeerEvent::HandshakeSendFailed,
|
|
now_ms,
|
|
alloc,
|
|
);
|
|
assert!(actions.is_empty());
|
|
assert!(machine.is_failed());
|
|
}
|
|
|
|
assert_eq!(node.connection_count(), 1);
|
|
|
|
// Failed connections should be cleaned up immediately regardless of age
|
|
node.check_timeouts().await;
|
|
|
|
assert_eq!(
|
|
node.connection_count(),
|
|
0,
|
|
"Failed connection should be removed"
|
|
);
|
|
assert_eq!(node.link_count(), 0, "Failed link should be removed");
|
|
assert_eq!(
|
|
node.index_allocator.count(),
|
|
0,
|
|
"Session index should be freed"
|
|
);
|
|
}
|
|
|
|
/// Test that msg1 bytes are stored on connection for resend.
|
|
#[tokio::test]
|
|
async fn test_msg1_stored_for_resend() {
|
|
use crate::proto::fmp::wire::build_msg1;
|
|
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
let peer_identity = make_peer_identity();
|
|
let remote_addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
let now_ms = std::time::SystemTime::now()
|
|
.duration_since(std::time::UNIX_EPOCH)
|
|
.map(|d| d.as_millis() as u64)
|
|
.unwrap_or(0);
|
|
let link_id = node.allocate_link_id();
|
|
let mut conn = outbound_leg(link_id, peer_identity, now_ms);
|
|
|
|
let our_index = node.index_allocator.allocate().unwrap();
|
|
let our_keypair = node.identity().keypair();
|
|
let noise_msg1 = conn
|
|
.start_handshake(our_keypair, node.startup_epoch(), now_ms)
|
|
.unwrap();
|
|
conn.set_conn_our_index(our_index);
|
|
conn.set_conn_transport_id(transport_id);
|
|
conn.set_conn_source_addr(remote_addr.clone());
|
|
|
|
// Build wire msg1 and store it (as initiate_peer_connection does)
|
|
let wire_msg1 = build_msg1(our_index, &noise_msg1);
|
|
let resend_interval = node.config().node.rate_limit.handshake_resend_interval_ms;
|
|
conn.set_conn_handshake_msg1(wire_msg1.clone(), now_ms + resend_interval);
|
|
|
|
// Verify stored msg1 matches what was built
|
|
assert_eq!(conn.conn_handshake_msg1().unwrap(), &wire_msg1);
|
|
}
|
|
|
|
/// Test that resend scheduling respects max_resends and backoff.
|
|
#[tokio::test]
|
|
async fn test_resend_scheduling() {
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
let peer_identity = make_peer_identity();
|
|
let remote_addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
let now_ms = 100_000u64; // Use a fixed time for predictable testing
|
|
let link_id = node.allocate_link_id();
|
|
let mut conn = outbound_leg(link_id, peer_identity, now_ms);
|
|
|
|
let our_index = node.index_allocator.allocate().unwrap();
|
|
let our_keypair = node.identity().keypair();
|
|
let noise_msg1 = conn
|
|
.start_handshake(our_keypair, node.startup_epoch(), now_ms)
|
|
.unwrap();
|
|
conn.set_conn_source_addr(remote_addr.clone());
|
|
|
|
// Store msg1 with first resend at now + 1000ms
|
|
let wire_msg1 = crate::proto::fmp::wire::build_msg1(our_index, &noise_msg1);
|
|
|
|
let link = Link::connectionless(
|
|
link_id,
|
|
transport_id,
|
|
remote_addr.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node.links.insert(link_id, link);
|
|
node.addr_to_link
|
|
.insert((transport_id, remote_addr.clone()), link_id);
|
|
node.pending_outbound
|
|
.insert((transport_id, our_index.as_u32()), link_id);
|
|
|
|
// The msg1-resend counter and its due timer live on the per-peer machine,
|
|
// which also carries the pending connection. Dial it to `SentMsg1`
|
|
// (connectionless: no connect step) and arm its retransmit timer at
|
|
// now + 1000ms, mirroring what a real dial arms.
|
|
let mut machine =
|
|
crate::peer::machine::PeerMachine::new_outbound(link_id, Some(peer_identity), now_ms);
|
|
let _ = machine.step(
|
|
crate::peer::machine::PeerEvent::Dial {
|
|
transport_id,
|
|
remote_addr: remote_addr.clone(),
|
|
peer_identity,
|
|
connection_oriented: false,
|
|
},
|
|
now_ms,
|
|
&mut node.index_allocator,
|
|
);
|
|
// The msg1 wire lives on the machine's carrier (the retransmit driver's
|
|
// resend source), mirroring `prepare_outbound_msg1`.
|
|
machine.set_conn_handshake_msg1(wire_msg1, now_ms + 1000);
|
|
machine.set_conn_our_index(our_index);
|
|
machine.set_conn_transport_id(transport_id);
|
|
machine.set_leg(conn.take_leg().unwrap());
|
|
node.peer_machines.insert(link_id, machine);
|
|
node.peer_timers.entry(link_id).or_default().insert(
|
|
crate::peer::machine::TimerKind::HandshakeRetransmit,
|
|
now_ms + 1000,
|
|
);
|
|
|
|
// Before the scheduled time the timer isn't due, so nothing fires.
|
|
node.drive_peer_timers(now_ms + 500).await;
|
|
assert_eq!(
|
|
node.connection_resend_count(link_id),
|
|
0,
|
|
"No resend before scheduled time"
|
|
);
|
|
|
|
// At the scheduled time the timer is due, but no transport is registered so
|
|
// the send fails. Record-on-success: the count does NOT advance (and the
|
|
// connection is not marked failed) — a failed resend just retries next tick.
|
|
node.drive_peer_timers(now_ms + 1000).await;
|
|
assert_eq!(
|
|
node.connection_resend_count(link_id),
|
|
0,
|
|
"Failed send records no resend"
|
|
);
|
|
}
|
|
|
|
/// Test that the timer driver reaps an outbound leg whose machine
|
|
/// `HandshakeTimeout` timer has come due (the timeout fold). The reap re-checks
|
|
/// the shell `is_timed_out` predicate, then tears the connection down exactly as
|
|
/// the old `check_timeouts` Teardown path did.
|
|
#[tokio::test]
|
|
async fn test_handshake_timeout_drive() {
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let peer_identity = make_peer_identity();
|
|
let remote_addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
let dial_ms = 1000u64;
|
|
let link_id = node.allocate_link_id();
|
|
let mut conn = outbound_leg(link_id, peer_identity, dial_ms);
|
|
let our_index = node.index_allocator.allocate().unwrap();
|
|
let our_keypair = node.identity().keypair();
|
|
let _ = conn
|
|
.start_handshake(our_keypair, node.startup_epoch(), dial_ms)
|
|
.unwrap();
|
|
conn.set_conn_source_addr(remote_addr.clone());
|
|
|
|
let link = Link::connectionless(
|
|
link_id,
|
|
transport_id,
|
|
remote_addr.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
node.links.insert(link_id, link);
|
|
node.addr_to_link
|
|
.insert((transport_id, remote_addr.clone()), link_id);
|
|
node.pending_outbound
|
|
.insert((transport_id, our_index.as_u32()), link_id);
|
|
|
|
// Machine in SentMsg1, carrying the pending connection, with a
|
|
// HandshakeTimeout timer armed at dial + 30s.
|
|
let mut machine =
|
|
crate::peer::machine::PeerMachine::new_outbound(link_id, Some(peer_identity), dial_ms);
|
|
let _ = machine.step(
|
|
crate::peer::machine::PeerEvent::Dial {
|
|
transport_id,
|
|
remote_addr: remote_addr.clone(),
|
|
peer_identity,
|
|
connection_oriented: false,
|
|
},
|
|
dial_ms,
|
|
&mut node.index_allocator,
|
|
);
|
|
machine.set_conn_our_index(our_index);
|
|
machine.set_conn_transport_id(transport_id);
|
|
machine.set_leg(conn.take_leg().unwrap());
|
|
node.peer_machines.insert(link_id, machine);
|
|
node.peer_timers.entry(link_id).or_default().insert(
|
|
crate::peer::machine::TimerKind::HandshakeTimeout,
|
|
dial_ms + 30_000,
|
|
);
|
|
|
|
assert_eq!(node.connection_count(), 1);
|
|
|
|
// Well past dial + 30s: the timer is due and the leg is idle-timed-out.
|
|
node.drive_peer_timers(dial_ms + 100_000).await;
|
|
|
|
assert_eq!(
|
|
node.connection_count(),
|
|
0,
|
|
"Timed-out leg reaped by the timer drive"
|
|
);
|
|
assert_eq!(node.index_allocator.count(), 0, "Session index freed");
|
|
assert!(
|
|
!node.peer_machines.contains_key(&link_id),
|
|
"Control machine dropped with the reaped connection"
|
|
);
|
|
assert!(
|
|
!node.peer_timers.contains_key(&link_id),
|
|
"Timer store dropped with the reaped connection"
|
|
);
|
|
}
|
|
|
|
/// Test that msg2 is stored on the control machine's carrier for responder resend.
|
|
#[test]
|
|
fn test_msg2_stored_on_connection() {
|
|
let mut machine = crate::peer::machine::PeerMachine::new_inbound(LinkId::new(1), 1000);
|
|
|
|
assert!(machine.conn_handshake_msg2().is_none());
|
|
|
|
let msg2_bytes = vec![0x01, 0x02, 0x03, 0x04];
|
|
machine.set_conn_handshake_msg2(msg2_bytes.clone());
|
|
|
|
assert_eq!(machine.conn_handshake_msg2().unwrap(), &msg2_bytes);
|
|
}
|
|
|
|
/// Test that duplicate msg2 is silently dropped when pending_outbound is already cleared.
|
|
#[tokio::test]
|
|
async fn test_duplicate_msg2_dropped() {
|
|
use crate::proto::fmp::wire::build_msg2;
|
|
use crate::transport::ReceivedPacket;
|
|
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
// No pending_outbound entry — simulate post-promotion state
|
|
let receiver_idx = SessionIndex::new(42);
|
|
let sender_idx = SessionIndex::new(99);
|
|
|
|
// Build a fake msg2 packet (XX msg2 is at least 106 bytes)
|
|
let fake_noise_msg2 = vec![0u8; 106];
|
|
let wire_msg2 = build_msg2(sender_idx, receiver_idx, &fake_noise_msg2);
|
|
|
|
let packet = ReceivedPacket {
|
|
transport_id,
|
|
remote_addr: TransportAddr::from_string("10.0.0.2:2121"),
|
|
data: wire_msg2,
|
|
timestamp_ms: 1000,
|
|
};
|
|
|
|
// Should silently drop — no pending_outbound for this index
|
|
node.handle_msg2(packet).await;
|
|
// No panic, no state change — that's the test
|
|
assert_eq!(node.connection_count(), 0);
|
|
assert_eq!(node.peer_count(), 0);
|
|
}
|
|
|
|
// ===== Profile Rejection Tests =====
|
|
|
|
/// Helper: create two test nodes, set their profiles, attempt a handshake,
|
|
/// and return whether they successfully peered.
|
|
async fn attempt_profile_handshake(
|
|
profile_a: crate::proto::fmp::NodeProfile,
|
|
profile_b: crate::proto::fmp::NodeProfile,
|
|
) -> (usize, usize) {
|
|
let mut nodes = vec![
|
|
make_test_node_with_profile(profile_a).await,
|
|
make_test_node_with_profile(profile_b).await,
|
|
];
|
|
|
|
initiate_handshake(&mut nodes, 0, 1).await;
|
|
drain_all_packets(&mut nodes, false).await;
|
|
|
|
let peers = (nodes[0].node.peer_count(), nodes[1].node.peer_count());
|
|
cleanup_nodes(&mut nodes).await;
|
|
peers
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_nonrouting_nonrouting_rejected() {
|
|
use crate::proto::fmp::NodeProfile;
|
|
let (a, b) = attempt_profile_handshake(NodeProfile::NonRouting, NodeProfile::NonRouting).await;
|
|
assert_eq!(a, 0, "NonRouting↔NonRouting should reject: node A");
|
|
assert_eq!(b, 0, "NonRouting↔NonRouting should reject: node B");
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_leaf_leaf_rejected() {
|
|
use crate::proto::fmp::NodeProfile;
|
|
let (a, b) = attempt_profile_handshake(NodeProfile::Leaf, NodeProfile::Leaf).await;
|
|
assert_eq!(a, 0, "Leaf↔Leaf should reject: node A");
|
|
assert_eq!(b, 0, "Leaf↔Leaf should reject: node B");
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_nonrouting_leaf_rejected() {
|
|
use crate::proto::fmp::NodeProfile;
|
|
let (a, b) = attempt_profile_handshake(NodeProfile::NonRouting, NodeProfile::Leaf).await;
|
|
assert_eq!(a, 0, "NonRouting↔Leaf should reject: node A");
|
|
assert_eq!(b, 0, "NonRouting↔Leaf should reject: node B");
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_leaf_nonrouting_rejected() {
|
|
use crate::proto::fmp::NodeProfile;
|
|
let (a, b) = attempt_profile_handshake(NodeProfile::Leaf, NodeProfile::NonRouting).await;
|
|
assert_eq!(a, 0, "Leaf↔NonRouting should reject: node A");
|
|
assert_eq!(b, 0, "Leaf↔NonRouting should reject: node B");
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_full_nonrouting_accepted() {
|
|
use crate::proto::fmp::NodeProfile;
|
|
let (a, b) = attempt_profile_handshake(NodeProfile::Full, NodeProfile::NonRouting).await;
|
|
assert_eq!(a, 1, "Full↔NonRouting should accept: node A");
|
|
assert_eq!(b, 1, "Full↔NonRouting should accept: node B");
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_full_leaf_accepted() {
|
|
use crate::proto::fmp::NodeProfile;
|
|
let (a, b) = attempt_profile_handshake(NodeProfile::Full, NodeProfile::Leaf).await;
|
|
assert_eq!(a, 1, "Full↔Leaf should accept: node A");
|
|
assert_eq!(b, 1, "Full↔Leaf should accept: node B");
|
|
}
|
|
|
|
// ===== XX Address-Based Dedup Tests =====
|
|
|
|
#[tokio::test]
|
|
async fn test_xx_duplicate_msg1_resends_msg2() {
|
|
use crate::proto::fmp::wire::build_msg1;
|
|
use crate::transport::ReceivedPacket;
|
|
|
|
// Node B with NO transport — msg2 send silently skips (if let Some check),
|
|
// but the pending connection and link are created.
|
|
let mut node_b = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
// Build a valid XX msg1 from an external initiator
|
|
let initiator = Identity::generate();
|
|
let mut hs = crate::noise::HandshakeState::new_initiator(initiator.keypair());
|
|
let noise_msg1 = hs.write_message_1().unwrap();
|
|
let sender_idx = SessionIndex::new(42);
|
|
let wire_msg1 = build_msg1(sender_idx, &noise_msg1);
|
|
|
|
let remote_addr = TransportAddr::from_string("10.0.0.1:2121");
|
|
|
|
// First msg1 → B creates pending inbound connection
|
|
let first_packet = ReceivedPacket {
|
|
transport_id,
|
|
remote_addr: remote_addr.clone(),
|
|
data: wire_msg1.clone(),
|
|
timestamp_ms: 1000,
|
|
};
|
|
node_b.handle_msg1(first_packet).await;
|
|
|
|
assert_eq!(
|
|
node_b.connection_count(),
|
|
1,
|
|
"B: 1 connection after first msg1"
|
|
);
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
0,
|
|
"B: 0 peers (XX, no promotion at msg1)"
|
|
);
|
|
|
|
// Duplicate msg1 from same address → dedup triggers msg2 resend, not new handshake
|
|
let dup_packet = ReceivedPacket {
|
|
transport_id,
|
|
remote_addr: remote_addr.clone(),
|
|
data: wire_msg1.clone(),
|
|
timestamp_ms: 1100,
|
|
};
|
|
node_b.handle_msg1(dup_packet).await;
|
|
|
|
assert_eq!(
|
|
node_b.connection_count(),
|
|
1,
|
|
"B: still 1 connection after duplicate msg1 (dedup, not new handshake)"
|
|
);
|
|
assert_eq!(node_b.peer_count(), 0, "B: still 0 peers");
|
|
}
|
|
|
|
/// `should_admit_msg1` admits when no transport is registered for the id.
|
|
/// (No gate to apply — the caller's other checks decide the outcome.)
|
|
///
|
|
/// This node is also the discriminator for the extraction of
|
|
/// `is_established_link_msg1`: with no transport registered the
|
|
/// `accept_connections` fallback admits, so the two predicates disagree
|
|
/// here and nowhere else. An extraction that dragged the fallback into
|
|
/// `is_established_link_msg1` fails the second assertion.
|
|
#[test]
|
|
fn test_should_admit_msg1_no_transport() {
|
|
let node = make_node();
|
|
let addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
assert!(node.should_admit_msg1(TransportId::new(1), &addr));
|
|
assert!(
|
|
!node.is_established_link_msg1(TransportId::new(1), &addr),
|
|
"the accept_connections fallback must not be part of the \
|
|
established-link predicate"
|
|
);
|
|
}
|
|
|
|
/// `should_admit_msg1` rejects a fresh msg1 (no addr_to_link entry) when
|
|
/// the transport has accept_connections=false. Behavior unchanged from
|
|
/// before the carve-out.
|
|
#[tokio::test]
|
|
async fn test_should_admit_msg1_rejects_fresh_when_accept_off() {
|
|
use crate::config::TcpConfig;
|
|
use crate::transport::tcp::TcpTransport;
|
|
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
// bind_addr=None → accept_connections() == false
|
|
let cfg = TcpConfig {
|
|
bind_addr: None,
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let tcp = TcpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Tcp(tcp));
|
|
|
|
let addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
assert!(!node.should_admit_msg1(transport_id, &addr));
|
|
}
|
|
|
|
/// Regression test: `should_admit_msg1` admits rekey/restart
|
|
/// msg1 from a peer with an existing link even when the transport has
|
|
/// accept_connections=false. Without this, the dual-init tie-breaker
|
|
/// deadlocks (the larger-NodeAddr side drops the winner's rekey msg1).
|
|
#[tokio::test]
|
|
async fn test_should_admit_msg1_admits_rekey_when_accept_off() {
|
|
use crate::config::TcpConfig;
|
|
use crate::transport::tcp::TcpTransport;
|
|
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
let cfg = TcpConfig {
|
|
bind_addr: None,
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let tcp = TcpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Tcp(tcp));
|
|
|
|
let addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
// Pre-populate addr_to_link as if a session were established for this
|
|
// peer on this transport (rekey msg1 will arrive against this entry).
|
|
let link_id = node.allocate_link_id();
|
|
node.addr_to_link
|
|
.insert((transport_id, addr.clone()), link_id);
|
|
|
|
assert!(node.should_admit_msg1(transport_id, &addr));
|
|
}
|
|
|
|
/// Same regression coverage as the TCP test above, but exercising the
|
|
/// UDP transport's new `accept_connections` config field (introduced
|
|
/// alongside the `outbound_only` mode). Proves the Node-level gate's
|
|
/// addr_to_link carve-out is transport-agnostic and that the new UDP
|
|
/// config knob is wired correctly through the Transport trait.
|
|
#[tokio::test]
|
|
async fn test_should_admit_msg1_admits_rekey_when_udp_accept_off() {
|
|
use crate::config::UdpConfig;
|
|
use crate::transport::udp::UdpTransport;
|
|
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
let cfg = UdpConfig {
|
|
bind_addr: Some("127.0.0.1:0".to_string()),
|
|
accept_connections: Some(false),
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let udp = UdpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Udp(udp));
|
|
|
|
let addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
|
|
// Fresh msg1 (no addr_to_link entry) is rejected by the gate when
|
|
// the transport refuses inbound.
|
|
assert!(!node.should_admit_msg1(transport_id, &addr));
|
|
|
|
// Pre-populate addr_to_link as if a session were established. The
|
|
// rekey carve-out admits the msg1 even though the transport still
|
|
// says accept_connections() == false.
|
|
let link_id = node.allocate_link_id();
|
|
node.addr_to_link
|
|
.insert((transport_id, addr.clone()), link_id);
|
|
|
|
assert!(node.should_admit_msg1(transport_id, &addr));
|
|
}
|
|
|
|
/// Regression test for the udp.outbound_only rekey loop observed in
|
|
/// production 2026-04-30 (parallel to the rekey/restart admission case
|
|
/// above).
|
|
///
|
|
/// Production scenario: nomad runs `udp.outbound_only=true` with peer
|
|
/// core-vm configured by hostname (`core-vm.tail65015.ts.net:2121`).
|
|
/// `initiate_connection` populates `addr_to_link` with the literal
|
|
/// hostname-form `TransportAddr`. core-vm's later rekey msg1 arrives at
|
|
/// nomad with a numeric source addr (the kernel always reports
|
|
/// `SocketAddr` in numeric form via `recvfrom`), so the `addr_to_link`
|
|
/// lookup misses, the gate falls through to `accept_connections()`
|
|
/// (false in outbound_only mode), and rejects. Result: dual-init
|
|
/// tie-breaker stalls because the loser side never produces msg2.
|
|
///
|
|
/// The carve-out predicate must also consult peer state by source
|
|
/// address: `current_addr()` is updated from inbound encrypted-frame
|
|
/// source addrs (`dataplane/encrypted.rs`), so an established peer can
|
|
/// be matched even when the addr_to_link key is hostname-form and the
|
|
/// incoming addr is numeric.
|
|
#[tokio::test]
|
|
async fn test_should_admit_msg1_admits_rekey_when_addr_form_differs() {
|
|
use crate::config::UdpConfig;
|
|
use crate::peer::ActivePeer;
|
|
use crate::transport::udp::UdpTransport;
|
|
|
|
let mut node = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
|
|
// outbound_only mode forces accept_connections() to false.
|
|
let cfg = UdpConfig {
|
|
outbound_only: Some(true),
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let udp = UdpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Udp(udp));
|
|
|
|
// Simulate initiate_connection's effect when peer config carries a
|
|
// hostname: addr_to_link is populated with hostname-form, not
|
|
// numeric-form.
|
|
let hostname_addr = TransportAddr::from_string("core-vm.example:2121");
|
|
let link_id = node.allocate_link_id();
|
|
node.addr_to_link
|
|
.insert((transport_id, hostname_addr.clone()), link_id);
|
|
|
|
// Promote a peer at the hostname's resolved numeric form
|
|
// (current_addr is set from the SocketAddr in udp_receive_loop).
|
|
let peer_full = crate::Identity::generate();
|
|
let peer_identity = PeerIdentity::from_pubkey(peer_full.pubkey());
|
|
let peer_node_addr = *peer_identity.node_addr();
|
|
let mut peer = ActivePeer::new(peer_identity, link_id, 1000);
|
|
let numeric_addr = TransportAddr::from_string("100.64.0.5:2121");
|
|
peer.set_current_addr(transport_id, numeric_addr.clone());
|
|
node.peers.insert(peer_node_addr, peer);
|
|
|
|
// Sanity: legacy carve-out still works for the hostname-form lookup.
|
|
assert!(node.should_admit_msg1(transport_id, &hostname_addr));
|
|
|
|
// The bug: incoming rekey msg1 arrives with numeric source addr.
|
|
// Without the additional carve-out, this is rejected (addr_to_link
|
|
// miss → accept_connections() false → drop).
|
|
assert!(
|
|
node.should_admit_msg1(transport_id, &numeric_addr),
|
|
"rekey msg1 from established peer must be admitted even when \
|
|
addr_to_link is keyed by a different addr-form (hostname vs \
|
|
numeric); the carve-out must consult peer current_addr"
|
|
);
|
|
|
|
// Negative: a stranger at a different numeric addr is still rejected
|
|
// (no peer there, no addr_to_link entry, falls to accept_connections).
|
|
let stranger_addr = TransportAddr::from_string("198.51.100.1:2121");
|
|
assert!(
|
|
!node.should_admit_msg1(transport_id, &stranger_addr),
|
|
"fresh msg1 from unknown source must still be rejected"
|
|
);
|
|
|
|
// The same two predicates read directly: both addr-forms of the
|
|
// established peer are established links, the stranger is not.
|
|
assert!(node.is_established_link_msg1(transport_id, &hostname_addr));
|
|
assert!(node.is_established_link_msg1(transport_id, &numeric_addr));
|
|
assert!(!node.is_established_link_msg1(transport_id, &stranger_addr));
|
|
}
|
|
|
|
/// `is_established_link_msg1` and `should_admit_msg1` are deliberately
|
|
/// independent on this branch, and this asserts the three cases where they
|
|
/// must disagree.
|
|
///
|
|
/// The `master` lineage defines `should_admit_msg1` as
|
|
/// `is_established_link_msg1() || accept_connections()`, which is sound at IK
|
|
/// but not at XX: here a bare `addr_to_link` hit also covers a *pending*
|
|
/// inbound connection and an outbound dial still in flight. The gate needs
|
|
/// that breadth (it is what breaks the dual-init deadlock), and the metering
|
|
/// classifier must not have it, or an unpromoted stranger draws on the
|
|
/// established-link bucket.
|
|
///
|
|
/// Every case below registers a transport whose `accept_connections()` is
|
|
/// false. That is not incidental: with no transport registered the gate's
|
|
/// fallback admits unconditionally, and a collapsed `should_admit_msg1` would
|
|
/// still read true, so the disagreement this test exists to pin would vanish.
|
|
#[tokio::test]
|
|
async fn established_predicate_is_independent_of_should_admit_msg1() {
|
|
use crate::config::UdpConfig;
|
|
use crate::peer::ActivePeer;
|
|
use crate::transport::Link;
|
|
use crate::transport::udp::UdpTransport;
|
|
use std::time::Duration;
|
|
|
|
let transport_id = TransportId::new(1);
|
|
|
|
// --- Case 1: no transport registered at all. --------------------------
|
|
// Documentation only, NOT a discriminator: under the collapsed-predicate
|
|
// break this case still passes, because the gate's no-transport fallback
|
|
// admits regardless of what the classifier says.
|
|
{
|
|
let node = make_node();
|
|
let addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
assert!(
|
|
node.should_admit_msg1(transport_id, &addr),
|
|
"no registered transport means no gate to apply"
|
|
);
|
|
assert!(
|
|
!node.is_established_link_msg1(transport_id, &addr),
|
|
"an address with no peer behind it is never established-link"
|
|
);
|
|
}
|
|
|
|
// --- Case 2: pending INBOUND link, no promoted peer. ------------------
|
|
// This is the state `handle_msg1` leaves behind between msg1 and msg3.
|
|
{
|
|
let mut node = make_node();
|
|
let cfg = UdpConfig {
|
|
bind_addr: Some("127.0.0.1:0".to_string()),
|
|
accept_connections: Some(false),
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let udp = UdpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Udp(udp));
|
|
|
|
let addr = TransportAddr::from_string("10.0.0.2:2121");
|
|
let link_id = node.allocate_link_id();
|
|
node.links.insert(
|
|
link_id,
|
|
Link::connectionless(
|
|
link_id,
|
|
transport_id,
|
|
addr.clone(),
|
|
LinkDirection::Inbound,
|
|
Duration::from_millis(100),
|
|
),
|
|
);
|
|
node.addr_to_link
|
|
.insert((transport_id, addr.clone()), link_id);
|
|
|
|
assert!(
|
|
node.peers.is_empty(),
|
|
"precondition: the link is pending, nothing is promoted"
|
|
);
|
|
assert!(
|
|
node.should_admit_msg1(transport_id, &addr),
|
|
"the gate admits on a bare addr_to_link hit"
|
|
);
|
|
assert!(
|
|
!node.is_established_link_msg1(transport_id, &addr),
|
|
"a pending inbound handshake is a stranger's, and must keep \
|
|
drawing on the stranger bucket for its whole lifetime"
|
|
);
|
|
}
|
|
|
|
// --- Case 3: pending OUTBOUND dial, no promoted peer. -----------------
|
|
// This is the dual-init carve-out. The assertion is what stops a later
|
|
// "simplification" collapsing the two predicates.
|
|
{
|
|
let mut node = make_node();
|
|
let cfg = UdpConfig {
|
|
outbound_only: Some(true),
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let udp = UdpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Udp(udp));
|
|
|
|
let addr = TransportAddr::from_string("10.0.0.3:2121");
|
|
let link_id = node.allocate_link_id();
|
|
node.links.insert(
|
|
link_id,
|
|
Link::connectionless(
|
|
link_id,
|
|
transport_id,
|
|
addr.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
),
|
|
);
|
|
node.addr_to_link
|
|
.insert((transport_id, addr.clone()), link_id);
|
|
|
|
assert!(
|
|
node.should_admit_msg1(transport_id, &addr),
|
|
"an outbound dial in flight must admit the peer's inbound msg1, \
|
|
or the dual-init tie-breaker deadlocks"
|
|
);
|
|
assert!(
|
|
!node.is_established_link_msg1(transport_id, &addr),
|
|
"an outbound dial is not a promoted peer"
|
|
);
|
|
}
|
|
|
|
// --- Agreement case: a genuinely promoted peer, both addr forms. ------
|
|
{
|
|
let mut node = make_node();
|
|
let cfg = UdpConfig {
|
|
outbound_only: Some(true),
|
|
..Default::default()
|
|
};
|
|
let (tx, _rx) = packet_channel(64);
|
|
let udp = UdpTransport::new(transport_id, None, cfg, tx);
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Udp(udp));
|
|
|
|
let hostname_addr = TransportAddr::from_string("core-vm.example:2121");
|
|
let link_id = node.allocate_link_id();
|
|
node.addr_to_link
|
|
.insert((transport_id, hostname_addr.clone()), link_id);
|
|
|
|
let peer_full = crate::Identity::generate();
|
|
let peer_identity = PeerIdentity::from_pubkey(peer_full.pubkey());
|
|
let peer_node_addr = *peer_identity.node_addr();
|
|
let mut peer = ActivePeer::new(peer_identity, link_id, 1000);
|
|
let numeric_addr = TransportAddr::from_string("100.64.0.5:2121");
|
|
peer.set_current_addr(transport_id, numeric_addr.clone());
|
|
node.peers.insert(peer_node_addr, peer);
|
|
|
|
// Limb 1: addr_to_link maps the hostname form to a link a peer owns.
|
|
assert!(node.should_admit_msg1(transport_id, &hostname_addr));
|
|
assert!(
|
|
node.is_established_link_msg1(transport_id, &hostname_addr),
|
|
"hostname-keyed promoted peer must class as established-link"
|
|
);
|
|
|
|
// Limb 2: the numeric form matches the peer's current_addr.
|
|
assert!(node.should_admit_msg1(transport_id, &numeric_addr));
|
|
assert!(
|
|
node.is_established_link_msg1(transport_id, &numeric_addr),
|
|
"rekey msg1 arriving in numeric form from a promoted peer must \
|
|
class as established-link even though addr_to_link is keyed on \
|
|
the hostname form"
|
|
);
|
|
|
|
// Negative: a stranger elsewhere is neither admitted nor established.
|
|
let stranger_addr = TransportAddr::from_string("198.51.100.1:2121");
|
|
assert!(!node.should_admit_msg1(transport_id, &stranger_addr));
|
|
assert!(!node.is_established_link_msg1(transport_id, &stranger_addr));
|
|
}
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Regression: `handle_msg3` must return the msg1-allocated session index to
|
|
// the allocator on the two inbound-establish arms that abandon the pending
|
|
// inbound leg without promoting it — the `Reject{DualRekeyWon}` tie-break
|
|
// (dual-init rekey we win) and the `ResendMsg2` duplicate-handshake arm. Both
|
|
// tear the pending connection/link down; neither must orphan the index.
|
|
// ===========================================================================
|
|
|
|
/// A node bundled with its UDP transport, receive channel, and bound address,
|
|
/// used to drive real msg1/msg2/msg3 exchanges below.
|
|
struct HsNode {
|
|
node: Node,
|
|
transport_id: TransportId,
|
|
packet_rx: crate::transport::PacketRx,
|
|
addr: TransportAddr,
|
|
}
|
|
|
|
/// Build an `HsNode` on an ephemeral localhost UDP port from an explicit config.
|
|
async fn make_hs_node(config: Config) -> HsNode {
|
|
use crate::config::UdpConfig;
|
|
use crate::transport::udp::UdpTransport;
|
|
|
|
let mut node = make_node_with(config);
|
|
let transport_id = TransportId::new(1);
|
|
let udp_config = UdpConfig {
|
|
bind_addr: Some("127.0.0.1:0".to_string()),
|
|
mtu: Some(1280),
|
|
..Default::default()
|
|
};
|
|
let (packet_tx, packet_rx) = packet_channel(64);
|
|
let mut transport = UdpTransport::new(transport_id, None, udp_config, packet_tx);
|
|
transport.start_async().await.unwrap();
|
|
let addr = TransportAddr::from_string(&transport.local_addr().unwrap().to_string());
|
|
node.transports
|
|
.insert(transport_id, TransportHandle::Udp(transport));
|
|
|
|
HsNode {
|
|
node,
|
|
transport_id,
|
|
packet_rx,
|
|
addr,
|
|
}
|
|
}
|
|
|
|
async fn stop_hs(n: &mut HsNode) {
|
|
for (_, t) in n.node.transports.iter_mut() {
|
|
t.stop().await.ok();
|
|
}
|
|
}
|
|
|
|
/// Receive the next packet whose handshake phase matches `phase` (the low
|
|
/// nibble of the wire type byte: 1=msg1, 2=msg2, 3=msg3), skipping unrelated
|
|
/// traffic. Post-promotion tree/filter announces (phase 0, encrypted data)
|
|
/// share these channels, so a phase filter keeps the hand-driven exchange in
|
|
/// step.
|
|
async fn recv_phase(rx: &mut crate::transport::PacketRx, phase: u8, what: &str) -> ReceivedPacket {
|
|
use std::time::Duration;
|
|
use tokio::time::timeout;
|
|
|
|
loop {
|
|
let pkt = timeout(Duration::from_secs(1), rx.recv())
|
|
.await
|
|
.unwrap_or_else(|_| panic!("timeout waiting for {}", what))
|
|
.expect("channel closed");
|
|
if pkt.data.first().is_some_and(|b| b & 0x0f == phase) {
|
|
return pkt;
|
|
}
|
|
}
|
|
}
|
|
|
|
/// Drive one initiator -> responder XX exchange (msg1 then msg2) and return the
|
|
/// responder's inbound msg3 packet, left unhandled for the caller. The msg3 is
|
|
/// produced by the initiator's real `handle_msg2`, so it carries a valid Noise
|
|
/// payload.
|
|
async fn drive_to_msg3(
|
|
initiator: &mut HsNode,
|
|
responder: &mut HsNode,
|
|
now_ms: u64,
|
|
) -> ReceivedPacket {
|
|
use crate::proto::fmp::wire::build_msg1;
|
|
use std::time::Duration;
|
|
|
|
let peer_identity = PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full());
|
|
|
|
let link_id = initiator.node.allocate_link_id();
|
|
let our_index = initiator.node.index_allocator.allocate().unwrap();
|
|
// Mirror the production dial path: the seam seeds the identified outbound
|
|
// leg's control machine at dial, and the promote feedback later
|
|
// crystallizes that same machine in place.
|
|
initiator
|
|
.node
|
|
.seed_handshake_machine(
|
|
HandshakeSeed::outbound(link_id, peer_identity, now_ms)
|
|
.with_our_index(our_index)
|
|
.with_transport_id(initiator.transport_id)
|
|
.with_source_addr(responder.addr.clone()),
|
|
)
|
|
.unwrap();
|
|
let our_keypair = initiator.node.identity().keypair();
|
|
let startup_epoch = initiator.node.startup_epoch();
|
|
let noise_msg1 = initiator
|
|
.node
|
|
.peer_machines
|
|
.get_mut(&link_id)
|
|
.unwrap()
|
|
.start_handshake(our_keypair, startup_epoch, now_ms)
|
|
.unwrap();
|
|
|
|
let wire_msg1 = build_msg1(our_index, &noise_msg1);
|
|
let link = Link::connectionless(
|
|
link_id,
|
|
initiator.transport_id,
|
|
responder.addr.clone(),
|
|
LinkDirection::Outbound,
|
|
Duration::from_millis(100),
|
|
);
|
|
initiator.node.links.insert(link_id, link);
|
|
initiator
|
|
.node
|
|
.addr_to_link
|
|
.insert((initiator.transport_id, responder.addr.clone()), link_id);
|
|
initiator
|
|
.node
|
|
.pending_outbound
|
|
.insert((initiator.transport_id, our_index.as_u32()), link_id);
|
|
|
|
initiator
|
|
.node
|
|
.transports
|
|
.get(&initiator.transport_id)
|
|
.unwrap()
|
|
.send(&responder.addr, &wire_msg1)
|
|
.await
|
|
.expect("send msg1");
|
|
|
|
// Responder processes msg1 and emits msg2 back to the initiator.
|
|
let msg1_pkt = recv_phase(&mut responder.packet_rx, 1, "msg1").await;
|
|
responder.node.handle_msg1(msg1_pkt).await;
|
|
|
|
// Initiator processes msg2 and emits msg3 to the responder.
|
|
let msg2_pkt = recv_phase(&mut initiator.packet_rx, 2, "msg2").await;
|
|
initiator.node.handle_msg2(msg2_pkt).await;
|
|
|
|
// Capture the responder's inbound msg3, left unhandled for the caller.
|
|
recv_phase(&mut responder.packet_rx, 3, "msg3").await
|
|
}
|
|
|
|
/// Drive a **real** rekey: the initiator's own `check_rekey` builds the rekey
|
|
/// msg1, and the msg3 that returns carries the rekey marker.
|
|
///
|
|
/// A bare second handshake will not substitute. It declares no rekey — because
|
|
/// it is not one — so the responder reads it as a fresh dial. Reaching the rekey
|
|
/// arm by backdating a session and re-handshaking tested the age proxy that the
|
|
/// marker replaces, not the rekey path.
|
|
///
|
|
/// Both ends' sessions must be aged past the trigger before calling this. Age
|
|
/// them well past it: the trigger is `after_secs + jitter` with jitter drawn from
|
|
/// [-15, +15], so a margin under 15s makes firing depend on the draw and the test
|
|
/// flaky.
|
|
///
|
|
/// The callers configure `after_secs = 30` rather than something smaller. Config
|
|
/// validation rejects any interval at or below the jitter bound, since such a
|
|
/// value saturates to zero on a negative draw and rekeys on sight for roughly
|
|
/// half of sessions, so a test cannot ask for one either — the node would fail to
|
|
/// build. Thirty clears the bound and still leaves the 120s backdate these tests
|
|
/// use a margin of 75s against the worst draw.
|
|
async fn drive_rekey_to_msg3(initiator: &mut HsNode, responder: &mut HsNode) -> ReceivedPacket {
|
|
initiator.node.check_rekey().await;
|
|
|
|
let msg1_pkt = recv_phase(&mut responder.packet_rx, 1, "rekey msg1").await;
|
|
responder.node.handle_msg1(msg1_pkt).await;
|
|
|
|
let msg2_pkt = recv_phase(&mut initiator.packet_rx, 2, "rekey msg2").await;
|
|
initiator.node.handle_msg2(msg2_pkt).await;
|
|
|
|
recv_phase(&mut responder.packet_rx, 3, "rekey msg3").await
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_msg3_dual_rekey_won_frees_index() {
|
|
// Both ends carry the rekey config so each can fire its own trigger; the
|
|
// classifier itself reads no config, only the sender's declaration.
|
|
let make_config = || {
|
|
let mut c = Config::new();
|
|
c.node.rekey.enabled = true;
|
|
c.node.rekey.after_secs = 30;
|
|
c
|
|
};
|
|
|
|
let mut initiator = make_hs_node(make_config()).await;
|
|
// The DualRekeyWon tie-break is won by the numerically smaller node addr,
|
|
// so the responder (whose handle_msg3 we exercise) must be the smaller.
|
|
let mut responder = loop {
|
|
let cand = make_hs_node(make_config()).await;
|
|
if cand.node.node_addr() < initiator.node.node_addr() {
|
|
break cand;
|
|
}
|
|
};
|
|
|
|
// First handshake: the responder promotes the initiator to a healthy active
|
|
// peer holding exactly one allocated session index.
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
let baseline = responder.node.index_allocator.count();
|
|
assert_eq!(baseline, 1, "responder holds exactly the peer's index");
|
|
|
|
// Mark a rekey of our own in progress so the peer's declared rekey meets it
|
|
// as a dual initiation, which the smaller node addr wins. Age both ends so
|
|
// the initiator's own trigger fires and produces a real, marked rekey msg3 —
|
|
// a bare handshake declares no rekey and would never reach this arm.
|
|
let peer_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
{
|
|
let peer = responder.node.get_peer_mut(&peer_addr).unwrap();
|
|
peer.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
peer.set_rekey_in_progress();
|
|
}
|
|
|
|
// The rekey's msg1 allocates a fresh index on the responder, then its msg3
|
|
// lands on the DualRekeyWon reject arm.
|
|
let msg3b = drive_rekey_to_msg3(&mut initiator, &mut responder).await;
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
baseline + 1,
|
|
"second msg1 allocated a fresh index"
|
|
);
|
|
responder.node.handle_msg3(msg3b).await;
|
|
|
|
// The rejected msg3 must return its index and leave the active peer intact.
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
baseline,
|
|
"DualRekeyWon must free the msg1-allocated index"
|
|
);
|
|
assert_eq!(responder.node.peer_count(), 1, "active peer untouched");
|
|
// The rejected leg's msg1-born machine goes with the leg; only the
|
|
// established peer's machine remains.
|
|
let peer_link = responder.node.get_peer(&peer_addr).unwrap().link_id();
|
|
assert_eq!(responder.node.peer_machines.len(), 1);
|
|
assert!(responder.node.peer_machines.contains_key(&peer_link));
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
assert!(
|
|
responder
|
|
.node
|
|
.get_peer(&peer_addr)
|
|
.unwrap()
|
|
.pending_new_session()
|
|
.is_none(),
|
|
"reject arm must not store rekey-responder state"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// Complete a Noise XX handshake between two identities and return the
|
|
/// initiator's session, standing in for the fresh keys a cross-connection swap
|
|
/// installs.
|
|
fn replacement_session(ours: &Identity, theirs: &Identity) -> crate::noise::NoiseSession {
|
|
use crate::noise::HandshakeState;
|
|
|
|
let mut initiator = HandshakeState::new_initiator(ours.keypair());
|
|
let mut responder = HandshakeState::new_responder(theirs.keypair());
|
|
initiator.set_local_epoch([0x11; 8]);
|
|
responder.set_local_epoch([0x22; 8]);
|
|
|
|
let msg1 = initiator.write_message_1().unwrap();
|
|
responder.read_message_1(&msg1).unwrap();
|
|
let msg2 = responder.write_message_2().unwrap();
|
|
initiator.read_message_2(&msg2).unwrap();
|
|
let msg3 = initiator.write_message_3().unwrap();
|
|
responder.read_message_3(&msg3).unwrap();
|
|
|
|
initiator.into_session().unwrap()
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_msg3_resend_msg2_frees_index() {
|
|
// A declared rekey naming keys the responder no longer holds. Production
|
|
// reaches this when a cross-connection swap replaces the responder's keys
|
|
// for the peer while the peer's rekey is in flight: the msg3 marker then
|
|
// resolves to a mismatch, which is unambiguously a rekey and not a crossing
|
|
// dial, and the classifier sends it to the duplicate arm. The shell must
|
|
// then free the msg1-allocated index and leave the existing session alone.
|
|
//
|
|
// A bare second handshake will not reach this arm: it declares no rekey, and
|
|
// an undeclared msg3 on a different link is a cross-connection, on which
|
|
// every assertion below also holds — so routing through it would exercise
|
|
// something else entirely while passing.
|
|
//
|
|
// The responder's own `rekey.enabled` is deliberately NOT the lever here.
|
|
// That flag governs whether this node initiates rekeys and has no say in
|
|
// whether it accepts one; using it to reach this arm is the divergence
|
|
// `test_asymmetric_rekey_config_converges` exists to forbid.
|
|
let mut init_config = Config::new();
|
|
init_config.node.rekey.enabled = true;
|
|
init_config.node.rekey.after_secs = 30;
|
|
let config = Config::new();
|
|
|
|
let mut initiator = make_hs_node(init_config).await;
|
|
let mut responder = make_hs_node(config).await;
|
|
|
|
// First handshake establishes the active peer.
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
let baseline = responder.node.index_allocator.count();
|
|
assert_eq!(baseline, 1, "responder holds exactly the peer's index");
|
|
|
|
let peer_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
// Age the initiator's session past the whole jitter band so its own rekey
|
|
// trigger fires and the msg3 that arrives carries a matching marker.
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
|
|
// The rekey's msg1 allocates a fresh index; its msg3 then frees it on the
|
|
// duplicate arm.
|
|
let msg3b = drive_rekey_to_msg3(&mut initiator, &mut responder).await;
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
baseline + 1,
|
|
"the rekey msg1 allocated a fresh index"
|
|
);
|
|
|
|
// The index the msg3 marker declares: the responder's index as the
|
|
// initiator knows it.
|
|
let declared = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.their_index()
|
|
.expect("initiator holds the responder's index");
|
|
|
|
// While the rekey is in flight, replace the responder's keys for the peer
|
|
// exactly as the outbound cross-connection swap does: a fresh index and
|
|
// session on the peer, the index map moved to the new index, the old index
|
|
// freed, and the peer's control machine told of the swap.
|
|
let fresh_index = responder.node.index_allocator.allocate().unwrap();
|
|
let session = replacement_session(responder.node.identity(), initiator.node.identity());
|
|
let (old_index, their_index, transport_id, link) = {
|
|
let peer = responder.node.get_peer_mut(&peer_addr).unwrap();
|
|
let their_index = peer
|
|
.their_index()
|
|
.expect("responder holds the peer's index");
|
|
let old_index = peer.replace_session(session, fresh_index, their_index);
|
|
(
|
|
old_index.expect("responder held an index before the swap"),
|
|
their_index,
|
|
peer.transport_id().expect("peer has a transport"),
|
|
peer.link_id(),
|
|
)
|
|
};
|
|
responder
|
|
.node
|
|
.peers_by_index
|
|
.remove(&(transport_id, old_index.as_u32()));
|
|
let _ = responder.node.index_allocator.free(old_index);
|
|
responder
|
|
.node
|
|
.peers_by_index
|
|
.insert((transport_id, fresh_index.as_u32()), peer_addr);
|
|
let acts = responder.node.peer_machines.get_mut(&link).unwrap().step(
|
|
crate::peer::machine::PeerEvent::CrossConnResolved {
|
|
outcome: crate::peer::machine::CrossConnOutcome::Swap {
|
|
our_index: fresh_index,
|
|
their_index,
|
|
},
|
|
},
|
|
Node::now_ms(),
|
|
&mut responder.node.index_allocator,
|
|
);
|
|
assert!(
|
|
acts.is_empty(),
|
|
"cross-connection resolution is a pure observation"
|
|
);
|
|
|
|
assert_ne!(
|
|
responder.node.get_peer(&peer_addr).unwrap().our_index(),
|
|
Some(declared),
|
|
"the responder must no longer hold the keys the msg3 declares, or this \
|
|
reaches the rekey-responder arm"
|
|
);
|
|
|
|
let before = responder.node.get_peer(&peer_addr).unwrap();
|
|
let session_before = (before.our_index(), before.their_index());
|
|
|
|
responder.node.handle_msg3(msg3b).await;
|
|
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
baseline,
|
|
"ResendMsg2 must free the msg1-allocated index"
|
|
);
|
|
assert_eq!(responder.node.peer_count(), 1, "active peer untouched");
|
|
let after = responder.node.get_peer(&peer_addr).unwrap();
|
|
assert_eq!(
|
|
(after.our_index(), after.their_index()),
|
|
session_before,
|
|
"the duplicate arm must leave the session alone; changed indices mean \
|
|
the cross-connection arm ran instead"
|
|
);
|
|
// The duplicate leg's msg1-born machine goes with the leg; only the
|
|
// established peer's machine remains.
|
|
let peer_link = responder.node.get_peer(&peer_addr).unwrap().link_id();
|
|
assert_eq!(responder.node.peer_machines.len(), 1);
|
|
assert!(responder.node.peer_machines.contains_key(&peer_link));
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
assert!(
|
|
responder
|
|
.node
|
|
.get_peer(&peer_addr)
|
|
.unwrap()
|
|
.pending_new_session()
|
|
.is_none(),
|
|
"duplicate-handshake arm must not store rekey-responder state"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Inbound machine lifecycle: every window leg carries a persistent machine
|
|
// from msg1 — parked `SentMsg2`, crystallized in place on promote, disposed
|
|
// with the leg on every terminating msg3 arm.
|
|
// ===========================================================================
|
|
|
|
#[tokio::test]
|
|
async fn test_inbound_machine_born_at_msg1_and_crystallized_at_promote() {
|
|
use crate::peer::machine::{HandshakePhase, PeerState};
|
|
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
|
|
// After msg1 the responder's window leg carries a machine parked at
|
|
// `SentMsg2`, seeded with the leg's msg1-allocated index.
|
|
assert_eq!(responder.node.connection_count(), 1);
|
|
let leg_link = responder.node.connections().next().unwrap().1.link_id();
|
|
let leg_index = responder
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.unwrap()
|
|
.our_index();
|
|
assert!(leg_index.is_some(), "msg1 allocated the leg index");
|
|
{
|
|
let machine = responder
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.expect("window leg carries a machine from msg1");
|
|
assert!(matches!(
|
|
machine.state(),
|
|
PeerState::Handshaking {
|
|
phase: HandshakePhase::SentMsg2,
|
|
..
|
|
}
|
|
));
|
|
assert_eq!(machine.our_index(), leg_index);
|
|
}
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
|
|
// msg3 promotes; the SAME machine survives and crystallizes in place.
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
let peer_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let peer = responder.node.get_peer(&peer_addr).unwrap();
|
|
assert_eq!(peer.link_id(), leg_link, "promote keeps the leg's link");
|
|
let peer_index = peer.our_index();
|
|
assert_eq!(peer_index, leg_index, "promote keeps the msg1 index");
|
|
let machine = responder
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.expect("machine survives promotion");
|
|
assert_eq!(machine.state(), PeerState::Established { addr: peer_addr });
|
|
assert_eq!(machine.our_index(), peer_index);
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
|
|
// The initiator's dial-persisted machine crystallized in place too.
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
let init_link = initiator.node.get_peer(&responder_addr).unwrap().link_id();
|
|
let init_machine = initiator
|
|
.node
|
|
.peer_machines
|
|
.get(&init_link)
|
|
.expect("dial machine survives promotion");
|
|
assert_eq!(
|
|
init_machine.state(),
|
|
PeerState::Established {
|
|
addr: responder_addr
|
|
}
|
|
);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_msg3_crypto_fail_disposes_leg_machine() {
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
|
|
let mut msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
assert_eq!(responder.node.peer_machines.len(), 1, "msg1-born machine");
|
|
assert_eq!(responder.node.index_allocator.count(), 1);
|
|
|
|
// Corrupt the Noise payload so `complete_handshake_msg3` fails.
|
|
let last = msg3.data.len() - 1;
|
|
msg3.data[last] ^= 0xFF;
|
|
responder.node.handle_msg3(msg3).await;
|
|
|
|
assert_eq!(responder.node.peer_count(), 0, "no promotion");
|
|
assert!(
|
|
responder.node.connections().next().is_none(),
|
|
"leg torn down"
|
|
);
|
|
assert!(
|
|
responder.node.peer_machines.is_empty(),
|
|
"crypto-fail teardown disposes the leg's machine"
|
|
);
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
0,
|
|
"msg1-allocated index returned"
|
|
);
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_msg3_rekey_respond_disposes_leg_machine() {
|
|
// A REAL rekey, driven from the initiator's own trigger so its msg3 carries
|
|
// the rekey marker. The initiator needs the rekey config to fire the
|
|
// trigger; the responder's copy only lets it run its own cadence, since the
|
|
// classifier reads no config.
|
|
let mut init_config = Config::new();
|
|
init_config.node.rekey.enabled = true;
|
|
init_config.node.rekey.after_secs = 30;
|
|
let mut config = Config::new();
|
|
config.node.rekey.enabled = true;
|
|
config.node.rekey.after_secs = 30;
|
|
|
|
let mut initiator = make_hs_node(init_config).await;
|
|
let mut responder = make_hs_node(config).await;
|
|
|
|
// First handshake establishes the active peer (and its machine).
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
|
|
let peer_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
// Age BOTH ends: the initiator so its rekey trigger fires, the responder so
|
|
// its own state matches what the rekey replaces.
|
|
for (node, addr) in [
|
|
(&mut initiator.node, responder_addr),
|
|
(&mut responder.node, peer_addr),
|
|
] {
|
|
node.get_peer_mut(&addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
}
|
|
|
|
// The rekey's msg3 lands on the rekey-responder arm: the pending session
|
|
// moves onto the established peer; the window leg and its msg1-born
|
|
// machine are consumed.
|
|
let msg3b = drive_rekey_to_msg3(&mut initiator, &mut responder).await;
|
|
responder.node.handle_msg3(msg3b).await;
|
|
|
|
let peer = responder.node.get_peer(&peer_addr).unwrap();
|
|
assert!(
|
|
peer.pending_new_session().is_some(),
|
|
"rekey-responder arm stores the pending session"
|
|
);
|
|
let peer_link = peer.link_id();
|
|
assert_eq!(
|
|
responder.node.peer_machines.len(),
|
|
1,
|
|
"the rekey window leg's machine is disposed with the leg"
|
|
);
|
|
assert!(responder.node.peer_machines.contains_key(&peer_link));
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// A rekey triggered by message count on a *young* session is still a rekey,
|
|
/// and both ends must come out of it holding the same session.
|
|
///
|
|
/// This is the defect the marker exists to remove. The rekey trigger is
|
|
/// `elapsed >= after_secs || counter >= after_messages`, so the count disjunct
|
|
/// fires on a session of any age — while the discriminator it replaced asked
|
|
/// only how old the session was, and read anything young as a fresh
|
|
/// cross-connection. The two ends then resolved onto different sessions, and a
|
|
/// link whose ends disagree decrypts one direction and drops the other.
|
|
///
|
|
/// Driven entirely off the count disjunct: `after_secs` is set far out of reach,
|
|
/// so nothing here depends on the rekey jitter draw.
|
|
#[tokio::test]
|
|
async fn test_count_triggered_rekey_on_young_session_converges() {
|
|
let make_config = || {
|
|
let mut c = Config::new();
|
|
c.node.rekey.enabled = true;
|
|
// Unreachable by age, so only the message counter can fire the trigger.
|
|
c.node.rekey.after_secs = 3600;
|
|
c.node.rekey.after_messages = 4;
|
|
c
|
|
};
|
|
|
|
let mut initiator = make_hs_node(make_config()).await;
|
|
let mut responder = make_hs_node(make_config()).await;
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
|
|
let initiator_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
// Carry the session past the message threshold, leaving its age at zero.
|
|
// Encrypting on the session is what the trigger actually counts, so this
|
|
// moves the real counter rather than a stand-in for it.
|
|
for _ in 0..5 {
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.noise_session_mut()
|
|
.unwrap()
|
|
.encrypt(b"traffic")
|
|
.expect("encrypt under the established session");
|
|
}
|
|
|
|
let rekey_msg3 = drive_rekey_to_msg3(&mut initiator, &mut responder).await;
|
|
responder.node.handle_msg3(rekey_msg3).await;
|
|
|
|
assert!(
|
|
responder
|
|
.node
|
|
.get_peer(&initiator_addr)
|
|
.unwrap()
|
|
.pending_new_session()
|
|
.is_some(),
|
|
"the responder read a real rekey as something else: no pending session"
|
|
);
|
|
|
|
// Both ends cut over to the session they just negotiated.
|
|
initiator.node.check_rekey().await;
|
|
responder.node.check_rekey().await;
|
|
|
|
let on_initiator = initiator.node.get_peer(&responder_addr).unwrap();
|
|
let on_responder = responder.node.get_peer(&initiator_addr).unwrap();
|
|
assert_eq!(
|
|
on_initiator.their_index(),
|
|
on_responder.our_index(),
|
|
"after the rekey the ends hold different sessions: traffic drops one way"
|
|
);
|
|
assert_eq!(
|
|
on_responder.their_index(),
|
|
on_initiator.our_index(),
|
|
"after the rekey the ends hold different sessions: traffic drops one way"
|
|
);
|
|
assert!(
|
|
on_initiator.our_index().is_some() && on_initiator.their_index().is_some(),
|
|
"both indices must be set, or the equality above passes on None == None"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// An asymmetric `rekey.enabled` must not split the link: the two ends still
|
|
/// come out of the peer's rekey holding the same session.
|
|
///
|
|
/// `rekey.enabled` says whether this node INITIATES rekeys. It says nothing
|
|
/// about whether it accepts one, and it cannot: by the time a rekey msg3
|
|
/// arrives the sender has already committed to the new session, and the
|
|
/// declaration on the wire is the only authoritative signal about what the msg3
|
|
/// is. A responder that read its own trigger config as permission to decline
|
|
/// would keep the old session while the peer sends on the new one, and the link
|
|
/// would carry nothing in either direction until the link-dead timer.
|
|
///
|
|
/// Every assertion here spans BOTH nodes, comparing each end's send index
|
|
/// against the index the other end receives on. The divergent outcome leaves
|
|
/// each node internally consistent, so a one-sided check cannot see it — which
|
|
/// is exactly how the defect survived: `test_msg3_resend_msg2_frees_index`
|
|
/// drove a declared rekey into a rekey-disabled responder and asserted only the
|
|
/// responder's own state, so it passed on full divergence.
|
|
#[tokio::test]
|
|
async fn test_asymmetric_rekey_config_converges() {
|
|
let mut init_config = Config::new();
|
|
init_config.node.rekey.enabled = true;
|
|
init_config.node.rekey.after_secs = 30;
|
|
// The responder never initiates a rekey of its own. It must still accept the
|
|
// peer's, and must still cut over to it.
|
|
let mut resp_config = Config::new();
|
|
resp_config.node.rekey.enabled = false;
|
|
|
|
let mut initiator = make_hs_node(init_config).await;
|
|
let mut responder = make_hs_node(resp_config).await;
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
|
|
let initiator_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
// Age the initiator's session well past the jitter band ([-15, +15] around
|
|
// `after_secs`) so its own trigger fires and the msg3 that arrives carries a
|
|
// matching marker.
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
|
|
let rekey_msg3 = drive_rekey_to_msg3(&mut initiator, &mut responder).await;
|
|
responder.node.handle_msg3(rekey_msg3).await;
|
|
|
|
assert!(
|
|
responder
|
|
.node
|
|
.get_peer(&initiator_addr)
|
|
.unwrap()
|
|
.pending_new_session()
|
|
.is_some(),
|
|
"the responder declined the peer's rekey on account of its own trigger \
|
|
config: no pending session"
|
|
);
|
|
|
|
// The initiator cuts over on its own cadence and starts sending on the new
|
|
// session. Cross-node, before either cutover reaches the responder: the
|
|
// session the initiator now sends on has to be the one the responder is
|
|
// holding ready, and vice versa.
|
|
initiator.node.check_rekey().await;
|
|
{
|
|
let on_initiator = initiator.node.get_peer(&responder_addr).unwrap();
|
|
let on_responder = responder.node.get_peer(&initiator_addr).unwrap();
|
|
assert_eq!(
|
|
on_initiator.their_index(),
|
|
on_responder.pending_our_index(),
|
|
"the initiator sends on an index the responder is not standing by to \
|
|
receive on"
|
|
);
|
|
assert_eq!(
|
|
on_responder.pending_their_index(),
|
|
on_initiator.our_index(),
|
|
"the responder is standing by to send on an index the initiator does \
|
|
not receive on"
|
|
);
|
|
assert!(
|
|
on_initiator.our_index().is_some() && on_initiator.their_index().is_some(),
|
|
"both indices must be set, or the equalities above pass on None == None"
|
|
);
|
|
}
|
|
|
|
// The responder runs no rekey cadence at all here, so it cuts over the way a
|
|
// responder always does: on the peer's first frame carrying the flipped
|
|
// K-bit, which the data plane turns into exactly this call.
|
|
responder
|
|
.node
|
|
.get_peer_mut(&initiator_addr)
|
|
.unwrap()
|
|
.handle_peer_kbit_flip();
|
|
|
|
let on_initiator = initiator.node.get_peer(&responder_addr).unwrap();
|
|
let on_responder = responder.node.get_peer(&initiator_addr).unwrap();
|
|
assert_eq!(
|
|
on_initiator.their_index(),
|
|
on_responder.our_index(),
|
|
"after the rekey the ends hold different sessions: traffic drops one way"
|
|
);
|
|
assert_eq!(
|
|
on_responder.their_index(),
|
|
on_initiator.our_index(),
|
|
"after the rekey the ends hold different sessions: traffic drops one way"
|
|
);
|
|
assert!(
|
|
on_responder.our_index().is_some() && on_responder.their_index().is_some(),
|
|
"both indices must be set, or the equalities above pass on None == None"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// A node that never initiates a rekey still finishes the drain of one it
|
|
/// accepted: the demoted session and its index come back.
|
|
///
|
|
/// Accepting a peer's rekey demotes our live session into the drain slot and
|
|
/// keeps its index registered, so the frames still in flight on it decrypt. The
|
|
/// drain arm of the rekey cadence is the only thing that ever releases either.
|
|
/// Skipping the whole cadence on a node with `rekey.enabled = false` therefore
|
|
/// pins one session and one index per accepted rekey, for the life of the peer —
|
|
/// which is a leak that grows with every rekey the peer initiates.
|
|
///
|
|
/// Asserted against the peer's real state and the real allocator, not a log
|
|
/// line: the previous session, the drain flag, the index-to-peer map and the
|
|
/// allocator's live count all have to come back to where they started.
|
|
#[tokio::test]
|
|
async fn test_non_initiating_responder_completes_rekey_drain() {
|
|
let mut init_config = Config::new();
|
|
init_config.node.rekey.enabled = true;
|
|
init_config.node.rekey.after_secs = 30;
|
|
let mut resp_config = Config::new();
|
|
resp_config.node.rekey.enabled = false;
|
|
|
|
let mut initiator = make_hs_node(init_config).await;
|
|
let mut responder = make_hs_node(resp_config).await;
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
let baseline = responder.node.index_allocator.count();
|
|
assert_eq!(baseline, 1, "responder holds exactly the peer's index");
|
|
|
|
let initiator_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
|
|
// The peer's rekey completes and the responder accepts it, exactly as
|
|
// `test_asymmetric_rekey_config_converges` establishes it must.
|
|
let rekey_msg3 = drive_rekey_to_msg3(&mut initiator, &mut responder).await;
|
|
responder.node.handle_msg3(rekey_msg3).await;
|
|
|
|
let (old_index, transport) = {
|
|
let peer = responder.node.get_peer(&initiator_addr).unwrap();
|
|
assert!(
|
|
peer.pending_new_session().is_some(),
|
|
"the responder did not accept the rekey, so there is no drain to test"
|
|
);
|
|
(peer.our_index().unwrap(), peer.transport_id().unwrap())
|
|
};
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
baseline + 1,
|
|
"the accepted rekey holds a second index"
|
|
);
|
|
|
|
// The responder cuts over on the peer's first frame on the new epoch, which
|
|
// demotes the session it was using into the drain slot.
|
|
responder
|
|
.node
|
|
.get_peer_mut(&initiator_addr)
|
|
.unwrap()
|
|
.handle_peer_kbit_flip();
|
|
|
|
{
|
|
let peer = responder.node.get_peer(&initiator_addr).unwrap();
|
|
assert!(peer.is_draining(), "the cutover must start a drain window");
|
|
assert!(
|
|
peer.previous_session().is_some(),
|
|
"the demoted session is kept for in-flight frames"
|
|
);
|
|
assert_eq!(peer.previous_our_index(), Some(old_index));
|
|
assert_ne!(
|
|
peer.our_index(),
|
|
Some(old_index),
|
|
"the live session must have moved off the drained index"
|
|
);
|
|
}
|
|
assert!(
|
|
responder
|
|
.node
|
|
.peers_by_index
|
|
.contains_key(&(transport, old_index.as_u32())),
|
|
"the drained index stays routable while the window is open"
|
|
);
|
|
|
|
// Mid-window the cadence must leave it alone, or the assertion below would
|
|
// pass on an arm that fires unconditionally.
|
|
responder.node.check_rekey().await;
|
|
assert!(
|
|
responder
|
|
.node
|
|
.get_peer(&initiator_addr)
|
|
.unwrap()
|
|
.previous_session()
|
|
.is_some(),
|
|
"the drain completed before its window expired"
|
|
);
|
|
|
|
// Window expires. This is the only path that releases the demoted session,
|
|
// and a node that does not initiate rekeys still has to run it.
|
|
//
|
|
// Backdate by three times the 10s drain window rather than by minutes: the
|
|
// backdate cannot reach past the monotonic clock's epoch, so a margin larger
|
|
// than the machine's uptime is unrepresentable, and a freshly booted CI
|
|
// runner is exactly that machine.
|
|
responder
|
|
.node
|
|
.get_peer_mut(&initiator_addr)
|
|
.unwrap()
|
|
.test_backdate_drain_start(std::time::Duration::from_secs(30));
|
|
responder.node.check_rekey().await;
|
|
|
|
let peer = responder.node.get_peer(&initiator_addr).unwrap();
|
|
assert!(
|
|
peer.previous_session().is_none(),
|
|
"the demoted session is pinned forever on a node that never initiates"
|
|
);
|
|
assert!(!peer.is_draining(), "the drain window must be closed");
|
|
assert_eq!(peer.previous_our_index(), None);
|
|
assert!(
|
|
!peer.rekey_in_progress(),
|
|
"a node that does not initiate must not have started a rekey of its own"
|
|
);
|
|
assert!(
|
|
peer.has_session() && peer.our_index().is_some(),
|
|
"the live session must survive the drain"
|
|
);
|
|
assert!(
|
|
!responder
|
|
.node
|
|
.peers_by_index
|
|
.contains_key(&(transport, old_index.as_u32())),
|
|
"the drained index is still routable, so it was never released"
|
|
);
|
|
assert_eq!(
|
|
responder.node.index_allocator.count(),
|
|
baseline,
|
|
"the drained index never came back to the allocator"
|
|
);
|
|
assert_eq!(responder.node.peer_count(), 1, "active peer untouched");
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// A fresh dial crossing our own in-flight rekey leaves BOTH ends on the same
|
|
/// session, and leaks nothing on the way.
|
|
///
|
|
/// The two events cross: we send a rekey msg1 and, before it completes, the peer
|
|
/// dials us anew on a different link. Its msg3 declares no rekey, correctly, so
|
|
/// it is a genuine cross-connection and resolves by the address tie-break.
|
|
///
|
|
/// The earlier version of this test asserted only the responder's state and so
|
|
/// passed on an outcome where the two ends had diverged onto four distinct
|
|
/// indices — the initiator's half of the tie-break cannot see our rekey and
|
|
/// swaps regardless, so a responder that declines desynchronizes the pair. Any
|
|
/// assertion here has to span both nodes; a one-sided check is the failure mode
|
|
/// this whole change exists to remove.
|
|
#[tokio::test]
|
|
async fn test_fresh_dial_crossing_our_rekey_converges() {
|
|
let make_config = || {
|
|
let mut c = Config::new();
|
|
c.node.rekey.enabled = true;
|
|
c.node.rekey.after_secs = 30;
|
|
c
|
|
};
|
|
|
|
let mut initiator = make_hs_node(make_config()).await;
|
|
// Pin the responder as the LARGER node addr: that is the ordering in which
|
|
// its inbound wins the tie-break and it performs the swap, which is the side
|
|
// that has to clean up the rekey it is abandoning. With a smaller responder
|
|
// it keeps its outbound and the interesting path is never taken.
|
|
let mut responder = loop {
|
|
let cand = make_hs_node(make_config()).await;
|
|
if cand.node.node_addr() > initiator.node.node_addr() {
|
|
break cand;
|
|
}
|
|
};
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
|
|
let peer_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
// Start the responder's OWN rekey and leave it in flight. Driven from the
|
|
// real trigger rather than poking the flag, so the peer carries whatever
|
|
// state a live rekey actually leaves behind. Backdate well past the jitter
|
|
// band ([-15, +15] around `after_secs`) or firing depends on the draw.
|
|
responder
|
|
.node
|
|
.get_peer_mut(&peer_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
responder.node.check_rekey().await;
|
|
|
|
assert!(
|
|
responder
|
|
.node
|
|
.get_peer(&peer_addr)
|
|
.unwrap()
|
|
.rekey_in_progress(),
|
|
"the responder's own rekey must be in flight, or this tests nothing"
|
|
);
|
|
|
|
// The peer now dials fresh on a NEW link, unaware of our rekey. A bare
|
|
// handshake declares no rekey, because it is not one.
|
|
let fresh_msg3 = drive_to_msg3(&mut initiator, &mut responder, 5000).await;
|
|
responder.node.handle_msg3(fresh_msg3).await;
|
|
|
|
let on_initiator = initiator.node.get_peer(&responder_addr).unwrap();
|
|
let on_responder = responder.node.get_peer(&peer_addr).unwrap();
|
|
assert_eq!(
|
|
on_initiator.their_index(),
|
|
on_responder.our_index(),
|
|
"ends diverged: the initiator sends on an index the responder does not receive on"
|
|
);
|
|
assert_eq!(
|
|
on_responder.their_index(),
|
|
on_initiator.our_index(),
|
|
"ends diverged: the responder sends on an index the initiator does not receive on"
|
|
);
|
|
assert!(
|
|
on_responder.our_index().is_some() && on_responder.their_index().is_some(),
|
|
"both indices must be set, or the equalities above pass on None == None"
|
|
);
|
|
|
|
// The rekey the swap displaced is gone rather than left dangling at a
|
|
// session nobody holds, and its index went back to the allocator with it.
|
|
assert!(
|
|
!on_responder.rekey_in_progress(),
|
|
"the displaced rekey must be abandoned by the swap, not left in flight"
|
|
);
|
|
assert!(
|
|
on_responder.pending_new_session().is_none(),
|
|
"the displaced rekey must leave no pending session behind"
|
|
);
|
|
assert!(on_responder.has_session());
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Anonymous-discovery outbound lifecycle: the leg's persistent machine is
|
|
// born identity-less at leg birth inside `start_handshake`, learns its
|
|
// identity from XX msg2 (crystallization), survives the promote, and is
|
|
// disposed with the leg when the dial turns out to target ourselves.
|
|
// ===========================================================================
|
|
|
|
#[tokio::test]
|
|
async fn test_anonymous_dial_births_identityless_machine_at_leg_birth() {
|
|
use crate::peer::machine::PeerState;
|
|
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let responder = make_hs_node(Config::new()).await;
|
|
|
|
// Anonymous dial (no peer identity): the connectionless path runs the
|
|
// inline handshake, which creates the leg and its machine together.
|
|
initiator
|
|
.node
|
|
.initiate_connection(initiator.transport_id, responder.addr.clone(), None)
|
|
.await
|
|
.expect("anonymous dial");
|
|
|
|
assert_eq!(initiator.node.connection_count(), 1);
|
|
let leg_link = initiator.node.connections().next().unwrap().1.link_id();
|
|
let machine = initiator
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.expect("anonymous leg carries a machine from leg birth");
|
|
assert!(
|
|
machine.identity().is_none(),
|
|
"anonymous machine is born without an identity"
|
|
);
|
|
assert_eq!(
|
|
machine.state(),
|
|
PeerState::Discovered,
|
|
"no event is dispatched on the inline dial path"
|
|
);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
let mut initiator = initiator;
|
|
let mut responder = responder;
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
#[tokio::test]
|
|
async fn test_anonymous_msg2_crystallizes_identity_and_promotes() {
|
|
use crate::peer::machine::PeerState;
|
|
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
|
|
initiator
|
|
.node
|
|
.initiate_connection(initiator.transport_id, responder.addr.clone(), None)
|
|
.await
|
|
.expect("anonymous dial");
|
|
let leg_link = initiator.node.connections().next().unwrap().1.link_id();
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
// Responder answers msg1 with msg2; the initiator's msg2 processing learns
|
|
// who answered, crystallizes the identity onto the leg-born machine, and
|
|
// promotes through it.
|
|
let msg1_pkt = recv_phase(&mut responder.packet_rx, 1, "msg1").await;
|
|
responder.node.handle_msg1(msg1_pkt).await;
|
|
let msg2_pkt = recv_phase(&mut initiator.packet_rx, 2, "msg2").await;
|
|
initiator.node.handle_msg2(msg2_pkt).await;
|
|
|
|
let responder_identity =
|
|
PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full());
|
|
let responder_addr = *responder_identity.node_addr();
|
|
assert_eq!(initiator.node.peer_count(), 1);
|
|
let peer = initiator.node.get_peer(&responder_addr).expect("promoted");
|
|
assert_eq!(peer.link_id(), leg_link, "promote keeps the leg's link");
|
|
|
|
// The SAME machine survived the promote, with the learned identity and
|
|
// the established state crystallized in place.
|
|
let machine = initiator
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.expect("machine survives the anonymous promote");
|
|
assert_eq!(
|
|
machine.identity().map(|id| *id.node_addr()),
|
|
Some(responder_addr),
|
|
"msg2 crystallized the learned identity onto the machine"
|
|
);
|
|
assert_eq!(
|
|
machine.state(),
|
|
PeerState::Established {
|
|
addr: responder_addr
|
|
}
|
|
);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
// Complete the exchange so the responder promotes too, and both sides
|
|
// stay coherent across the full anonymous establish path.
|
|
let msg3_pkt = recv_phase(&mut responder.packet_rx, 3, "msg3").await;
|
|
responder.node.handle_msg3(msg3_pkt).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// The msg2 self-connect drop must release all three things the outbound leg
|
|
/// holds: its session index, its link, and its `pending_outbound` entry.
|
|
///
|
|
/// The three assertions are independent by construction — each names a
|
|
/// different registry — so deleting any one of the three releases from the arm
|
|
/// reds exactly one of them.
|
|
///
|
|
/// The node handshakes with itself, so it holds TWO legs here: the outbound one
|
|
/// being dropped and the inbound one its own `handle_msg1` created. That is why
|
|
/// the allocator goes to `baseline - 1` rather than to zero, and why the
|
|
/// inbound leg's index staying allocated is a useful control that the right
|
|
/// index was freed.
|
|
#[tokio::test]
|
|
async fn test_anonymous_self_connect_drop_disposes_machine() {
|
|
let mut node = make_hs_node(Config::new()).await;
|
|
let self_addr = node.addr.clone();
|
|
let transport_id = node.transport_id;
|
|
|
|
// Anonymously dial our own bound address (a shared-media beacon can echo
|
|
// ourselves back at us).
|
|
node.node
|
|
.initiate_connection(node.transport_id, self_addr.clone(), None)
|
|
.await
|
|
.expect("anonymous self dial");
|
|
let leg_link = node.node.connections().next().unwrap().1.link_id();
|
|
assert_eq!(node.node.peer_machines.len(), 1);
|
|
let outbound_idx = node
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("the dial allocated an index for the outbound leg");
|
|
assert!(
|
|
node.node
|
|
.pending_outbound
|
|
.contains_key(&(transport_id, outbound_idx.as_u32())),
|
|
"control: the dial registered the leg for msg2 dispatch"
|
|
);
|
|
|
|
// We answer our own msg1, then our msg2 processing discovers the learned
|
|
// identity is our own and drops the leg — machine included.
|
|
let msg1_pkt = recv_phase(&mut node.packet_rx, 1, "msg1").await;
|
|
node.node.handle_msg1(msg1_pkt).await;
|
|
|
|
// The inbound leg msg1 just parked: the one machine that is not the dial's.
|
|
let inbound_link = *node
|
|
.node
|
|
.peer_machines
|
|
.keys()
|
|
.find(|k| **k != leg_link)
|
|
.expect("msg1 parks an inbound leg of its own");
|
|
let inbound_idx = node
|
|
.node
|
|
.peer_machines
|
|
.get(&inbound_link)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("msg1 allocates an index for the inbound leg");
|
|
let baseline = node.node.index_allocator.count();
|
|
assert_eq!(baseline, 2, "one index per leg, and there are two legs");
|
|
assert!(
|
|
node.node.index_allocator.is_allocated(outbound_idx),
|
|
"control: the outbound leg's index is live before the drop"
|
|
);
|
|
|
|
let msg2_pkt = recv_phase(&mut node.packet_rx, 2, "msg2").await;
|
|
node.node.handle_msg2(msg2_pkt).await;
|
|
|
|
assert_eq!(node.node.peer_count(), 0, "no promotion");
|
|
assert!(
|
|
!node.node.has_pending_leg(&leg_link),
|
|
"self-connect drop removes the outbound leg"
|
|
);
|
|
assert!(
|
|
!node.node.peer_machines.contains_key(&leg_link),
|
|
"self-connect drop disposes the outbound leg's machine"
|
|
);
|
|
|
|
// Leak 1: the session index.
|
|
assert!(
|
|
!node.node.index_allocator.is_allocated(outbound_idx),
|
|
"the self-connect drop must return the outbound leg's index"
|
|
);
|
|
assert_eq!(
|
|
node.node.index_allocator.count(),
|
|
baseline - 1,
|
|
"and must free exactly one"
|
|
);
|
|
assert!(
|
|
node.node.index_allocator.is_allocated(inbound_idx),
|
|
"the surviving inbound leg keeps its own index"
|
|
);
|
|
|
|
// Leak 2: the link.
|
|
assert!(
|
|
!node.node.links.contains_key(&leg_link),
|
|
"the self-connect drop must remove the outbound leg's link"
|
|
);
|
|
|
|
// Leak 3: the pending_outbound entry.
|
|
assert!(
|
|
!node
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(transport_id, outbound_idx.as_u32())),
|
|
"the self-connect drop must clear the leg's msg2 dispatch entry"
|
|
);
|
|
|
|
// Not a leak assertion: a guard against the wrong fix. `handle_msg1`
|
|
// overwrote the reverse mapping for this address with the INBOUND leg's
|
|
// link, and `remove_link` declines to clear an entry that no longer points
|
|
// at the link being removed. An implementer replacing `remove_link` with a
|
|
// bare `links.remove` plus `addr_to_link.remove` would take down the live
|
|
// inbound leg's routing, and this is where that shows up.
|
|
assert_eq!(
|
|
node.node.addr_to_link.get(&(transport_id, self_addr)),
|
|
Some(&inbound_link),
|
|
"the surviving inbound leg keeps the address mapping; the dropped \
|
|
outbound leg must not take it down"
|
|
);
|
|
|
|
node.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut node).await;
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Initiator rekey static-key continuity
|
|
//
|
|
// The rekey msg2 is dispatched to its peer by the session index the initiator
|
|
// itself allocated, and that index travels in the CLEARTEXT rekey msg1 header.
|
|
// Under XX the responder's static arrives in msg2 rather than being pinned at
|
|
// dial (as IK pinned it), so an on-path party that beats the real peer to the
|
|
// reply produces a perfectly valid handshake under its own static. The
|
|
// continuity gate is what stops that session from taking the peer's slot.
|
|
// ===========================================================================
|
|
|
|
/// Establish initiator↔responder, start a real rekey on the initiator, then
|
|
/// let a third node answer the rekey msg1 with a valid XX msg2 built from its
|
|
/// OWN static. The initiator must reject it and keep the established session
|
|
/// live and usable.
|
|
#[tokio::test]
|
|
async fn test_rekey_msg2_foreign_static_rejected() {
|
|
let mut rekey_config = Config::new();
|
|
rekey_config.node.rekey.enabled = true;
|
|
rekey_config.node.rekey.after_secs = 30;
|
|
|
|
let mut initiator = make_hs_node(rekey_config).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
let mut attacker = make_hs_node(Config::new()).await;
|
|
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
let initiator_addr =
|
|
*PeerIdentity::from_pubkey_full(initiator.node.identity().pubkey_full()).node_addr();
|
|
let attacker_addr =
|
|
*PeerIdentity::from_pubkey_full(attacker.node.identity().pubkey_full()).node_addr();
|
|
|
|
// Establish the link both ways.
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(initiator.node.peer_count(), 1);
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
|
|
// Record what the established session must still look like afterwards.
|
|
let session_hash = *initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.noise_session()
|
|
.unwrap()
|
|
.handshake_hash();
|
|
let peer_link = initiator.node.get_peer(&responder_addr).unwrap().link_id();
|
|
|
|
// Age the session past the (jittered) rekey threshold and let the real
|
|
// cadence fire, so the rekey msg1 and its index are produced exactly as in
|
|
// production.
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
let baseline = initiator.node.index_allocator.count();
|
|
initiator.node.check_rekey().await;
|
|
let rekey_index = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_our_index()
|
|
.expect("cadence started a rekey");
|
|
assert_eq!(
|
|
initiator.node.index_allocator.count(),
|
|
baseline + 1,
|
|
"rekey allocated its own index"
|
|
);
|
|
|
|
// The attacker observes the cleartext rekey msg1 on path and answers it
|
|
// first, under its own static. The real responder never sees it.
|
|
let rekey_msg1 = recv_phase(&mut responder.packet_rx, 1, "rekey msg1").await;
|
|
attacker.node.handle_msg1(rekey_msg1).await;
|
|
let forged_msg2 = recv_phase(&mut initiator.packet_rx, 2, "forged rekey msg2").await;
|
|
initiator.node.handle_msg2(forged_msg2).await;
|
|
|
|
// The impostor never becomes (or displaces) a peer.
|
|
assert_eq!(initiator.node.peer_count(), 1, "peer set unchanged");
|
|
assert!(
|
|
initiator.node.get_peer(&attacker_addr).is_none(),
|
|
"impostor must not enter the peer set"
|
|
);
|
|
let peer = initiator.node.get_peer(&responder_addr).expect("kept");
|
|
assert!(
|
|
peer.pending_new_session().is_none(),
|
|
"a foreign static must not be installed as the pending session"
|
|
);
|
|
assert!(
|
|
!peer.rekey_in_progress(),
|
|
"the rejected rekey cycle is abandoned"
|
|
);
|
|
assert_eq!(peer.link_id(), peer_link, "the peer keeps its link");
|
|
|
|
// The established session is byte-for-byte the one we started with, still
|
|
// bound to the real responder.
|
|
assert_eq!(
|
|
peer.noise_session().unwrap().handshake_hash(),
|
|
&session_hash,
|
|
"the established session was not replaced"
|
|
);
|
|
assert_eq!(
|
|
peer.noise_session().unwrap().remote_static_xonly(),
|
|
responder.node.identity().pubkey(),
|
|
"the established session stays bound to the real peer"
|
|
);
|
|
|
|
// The rekey index is returned and its msg2 dispatch entry is gone, so a
|
|
// late (or replayed) msg2 on that index cannot re-enter the dead cycle.
|
|
assert_eq!(
|
|
initiator.node.index_allocator.count(),
|
|
baseline,
|
|
"the rejected rekey must free its index"
|
|
);
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(initiator.transport_id, rekey_index.as_u32())),
|
|
"the rejected rekey's dispatch entry must not survive"
|
|
);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
// ...and it is still usable: the initiator encrypts under the surviving
|
|
// session and the real responder decrypts it.
|
|
let probe = b"link still live after the rejected rekey";
|
|
let counter = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.noise_session()
|
|
.unwrap()
|
|
.current_send_counter();
|
|
let ciphertext = initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.noise_session_mut()
|
|
.unwrap()
|
|
.encrypt(probe)
|
|
.expect("encrypt under the surviving session");
|
|
let plaintext = responder
|
|
.node
|
|
.get_peer_mut(&initiator_addr)
|
|
.unwrap()
|
|
.noise_session_mut()
|
|
.unwrap()
|
|
.decrypt_with_replay_check(&ciphertext, counter)
|
|
.expect("the peer still decrypts under the original session");
|
|
assert_eq!(plaintext, probe);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
stop_hs(&mut attacker).await;
|
|
}
|
|
|
|
/// The forged rekey msg2 is charged to `rekey_static_mismatch`, not to the
|
|
/// undifferentiated `bad_state` bucket.
|
|
///
|
|
/// This counter is the only machine-readable indicator that the continuity
|
|
/// gate is firing. `bad_state` is shared with header parse failures, ACL
|
|
/// denials, allocator pressure and admission drops, so an operator alerting on
|
|
/// it cannot tell an on-path attacker forging rekey msg2 from routine
|
|
/// handshake noise. The `bad_state == 0` limb is what makes this test
|
|
/// discriminating: recording the old catch-all variant from the reject arm
|
|
/// fails both assertions, not neither.
|
|
#[tokio::test]
|
|
async fn test_forged_rekey_msg2_is_counted_as_rekey_static_mismatch_not_bad_state() {
|
|
let mut rekey_config = Config::new();
|
|
rekey_config.node.rekey.enabled = true;
|
|
rekey_config.node.rekey.after_secs = 30;
|
|
|
|
let mut initiator = make_hs_node(rekey_config).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
let mut attacker = make_hs_node(Config::new()).await;
|
|
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(initiator.node.peer_count(), 1);
|
|
|
|
// Counter identity: the established handshake must not itself have
|
|
// charged either counter, or the post-attack readings prove nothing.
|
|
assert_eq!(
|
|
initiator.node.stats().handshake.rekey_static_mismatch,
|
|
0,
|
|
"the clean handshake charges no mismatch"
|
|
);
|
|
assert_eq!(
|
|
initiator.node.stats().handshake.bad_state,
|
|
0,
|
|
"the clean handshake charges no catch-all reject"
|
|
);
|
|
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
initiator.node.check_rekey().await;
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_our_index()
|
|
.is_some(),
|
|
"the cadence started a rekey, so the reject arm is reachable"
|
|
);
|
|
|
|
// The attacker answers the cleartext rekey msg1 under its own static.
|
|
let rekey_msg1 = recv_phase(&mut responder.packet_rx, 1, "rekey msg1").await;
|
|
attacker.node.handle_msg1(rekey_msg1).await;
|
|
let forged_msg2 = recv_phase(&mut initiator.packet_rx, 2, "forged rekey msg2").await;
|
|
initiator.node.handle_msg2(forged_msg2).await;
|
|
|
|
// Arm identity: the reject really happened, rather than the msg2 being
|
|
// dropped earlier for some unrelated reason.
|
|
assert_eq!(initiator.node.peer_count(), 1, "peer set unchanged");
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_in_progress(),
|
|
"the rejected rekey cycle is abandoned"
|
|
);
|
|
|
|
assert_eq!(
|
|
initiator.node.stats().handshake.rekey_static_mismatch,
|
|
1,
|
|
"the forged rekey msg2 is charged to its own counter"
|
|
);
|
|
assert_eq!(
|
|
initiator.node.stats().handshake.bad_state,
|
|
0,
|
|
"and is no longer hidden in the undifferentiated bucket"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
stop_hs(&mut attacker).await;
|
|
}
|
|
|
|
/// The same cadence-driven rekey, answered by the REAL peer, still installs the
|
|
/// pending session — the gate must be invisible on the legitimate path.
|
|
#[tokio::test]
|
|
async fn test_rekey_msg2_matching_static_installs() {
|
|
let mut rekey_config = Config::new();
|
|
rekey_config.node.rekey.enabled = true;
|
|
rekey_config.node.rekey.after_secs = 30;
|
|
|
|
let mut initiator = make_hs_node(rekey_config).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(initiator.node.peer_count(), 1);
|
|
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
initiator.node.check_rekey().await;
|
|
let rekey_index = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_our_index()
|
|
.expect("cadence started a rekey");
|
|
|
|
// The real peer answers its own rekey msg1.
|
|
let rekey_msg1 = recv_phase(&mut responder.packet_rx, 1, "rekey msg1").await;
|
|
responder.node.handle_msg1(rekey_msg1).await;
|
|
let rekey_msg2 = recv_phase(&mut initiator.packet_rx, 2, "rekey msg2").await;
|
|
initiator.node.handle_msg2(rekey_msg2).await;
|
|
|
|
let peer = initiator.node.get_peer(&responder_addr).expect("kept");
|
|
assert!(
|
|
peer.pending_new_session().is_some(),
|
|
"a matching static installs the pending session"
|
|
);
|
|
assert_eq!(
|
|
peer.pending_new_session().unwrap().remote_static_xonly(),
|
|
responder.node.identity().pubkey(),
|
|
"the pending session is bound to the real peer"
|
|
);
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.peers_by_index
|
|
.contains_key(&(initiator.transport_id, rekey_index.as_u32())),
|
|
"the rekey index maps to the peer, awaiting K-bit cutover"
|
|
);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Initiator dial-identity pinning (initial outbound handshake)
|
|
//
|
|
// The rekey gate above protects a link that is already established; this one
|
|
// protects the link being formed, and needs no rekey to reach. Under XX the
|
|
// responder's static arrives in msg2 rather than being pinned at dial (as IK
|
|
// pinned it), so an on-path party that observes our msg1 and answers it first
|
|
// produces a perfectly valid handshake under its own static. The dial-identity
|
|
// gate is what stops that leg from being promoted as the peer we dialed.
|
|
//
|
|
// The three cases below are the whole decision surface: a named dial answered
|
|
// by a stranger (reject), a named dial answered by its peer (promote), and an
|
|
// anonymous dial, which names nobody and so must still promote whoever answers
|
|
// - the carve-out that keeps shared-media discovery working.
|
|
// ===========================================================================
|
|
|
|
/// A named dial answered by a foreign static must not promote, and must leave
|
|
/// no residue behind: no machine, no leg, no link, no `pending_outbound`
|
|
/// dispatch entry, and no orphaned session index.
|
|
#[tokio::test]
|
|
async fn test_dial_msg2_foreign_static_rejected() {
|
|
use std::time::Duration;
|
|
use tokio::time::timeout;
|
|
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let mut intended = make_hs_node(Config::new()).await;
|
|
let mut attacker = make_hs_node(Config::new()).await;
|
|
|
|
let intended_identity = PeerIdentity::from_pubkey_full(intended.node.identity().pubkey_full());
|
|
let intended_addr = *intended_identity.node_addr();
|
|
let attacker_addr =
|
|
*PeerIdentity::from_pubkey_full(attacker.node.identity().pubkey_full()).node_addr();
|
|
assert_ne!(intended_addr, attacker_addr);
|
|
|
|
let baseline = initiator.node.index_allocator.count();
|
|
|
|
// A named dial whose msg1 reaches the attacker instead of the peer. From
|
|
// the initiator's side this is indistinguishable from an on-path party
|
|
// racing the real responder's msg2, and it is the same thing the code sees.
|
|
initiator
|
|
.node
|
|
.initiate_connection(
|
|
initiator.transport_id,
|
|
attacker.addr.clone(),
|
|
Some(intended_identity),
|
|
)
|
|
.await
|
|
.expect("named dial");
|
|
|
|
let leg_link = initiator.node.connections().next().unwrap().1.link_id();
|
|
let leg_index = initiator
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("msg1 preparation allocated our index");
|
|
assert_eq!(
|
|
initiator.node.index_allocator.count(),
|
|
baseline + 1,
|
|
"the dial allocated its own index"
|
|
);
|
|
|
|
// The attacker answers the dial with a valid XX msg2 under its own static.
|
|
let msg1 = recv_phase(&mut attacker.packet_rx, 1, "msg1").await;
|
|
attacker.node.handle_msg1(msg1).await;
|
|
let forged_msg2 = recv_phase(&mut initiator.packet_rx, 2, "forged msg2").await;
|
|
initiator.node.handle_msg2(forged_msg2).await;
|
|
|
|
// Nothing is promoted - not the impostor, and not the peer we dialed
|
|
// (whose identity never authenticated anything here).
|
|
assert_eq!(initiator.node.peer_count(), 0, "no promotion");
|
|
assert!(
|
|
initiator.node.get_peer(&attacker_addr).is_none(),
|
|
"the impostor must not enter the peer set"
|
|
);
|
|
assert!(
|
|
initiator.node.get_peer(&intended_addr).is_none(),
|
|
"the dialed peer must not be credited with a handshake it never ran"
|
|
);
|
|
|
|
// No registry residue: the leg, its machine, its link, its dispatch entry
|
|
// and its index are all gone.
|
|
assert!(
|
|
!initiator.node.has_pending_leg(&leg_link),
|
|
"the rejected leg is torn down"
|
|
);
|
|
assert!(
|
|
!initiator.node.peer_machines.contains_key(&leg_link),
|
|
"the rejected leg's machine is disposed"
|
|
);
|
|
assert!(
|
|
!initiator.node.links.contains_key(&leg_link),
|
|
"the rejected leg's link is removed"
|
|
);
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(initiator.transport_id, leg_index.as_u32())),
|
|
"the rejected leg's dispatch entry must not survive, or a replayed \
|
|
msg2 could re-enter the dead leg"
|
|
);
|
|
assert_eq!(
|
|
initiator.node.index_allocator.count(),
|
|
baseline,
|
|
"the rejected dial must free the index it allocated"
|
|
);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
// The gate sits ahead of the msg3 send, so the impostor's handshake is
|
|
// never completed: it is left waiting on a msg3 that never comes.
|
|
assert!(
|
|
timeout(Duration::from_millis(250), attacker.packet_rx.recv())
|
|
.await
|
|
.is_err(),
|
|
"no msg3 may be sent to a responder that substituted its identity"
|
|
);
|
|
assert_eq!(
|
|
attacker.node.peer_count(),
|
|
0,
|
|
"the impostor never completes its own side either"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut intended).await;
|
|
stop_hs(&mut attacker).await;
|
|
}
|
|
|
|
/// The rejected dial must stay on the dial schedule. Disposing the leg takes it
|
|
/// out of both reapers, so the handshake-timeout sweep that normally reschedules
|
|
/// a stuck outbound dial never sees it; the reject arm has to fire that reflex
|
|
/// itself. Without it a configured peer is dialed once at startup and, after one
|
|
/// substituted msg2, never again for the life of the process - a persistent
|
|
/// outbound blackhole costing the attacker a single packet. The retry must also
|
|
/// name the peer we dialed, not the static that answered.
|
|
#[tokio::test]
|
|
async fn test_dial_msg2_foreign_static_reschedules_dial() {
|
|
let intended_local = Identity::generate();
|
|
let intended_identity =
|
|
PeerIdentity::from_npub(&intended_local.npub()).expect("generated npub parses");
|
|
let intended_addr = *intended_identity.node_addr();
|
|
|
|
let mut attacker = make_hs_node(Config::new()).await;
|
|
let attacker_addr =
|
|
*PeerIdentity::from_pubkey_full(attacker.node.identity().pubkey_full()).node_addr();
|
|
assert_ne!(intended_addr, attacker_addr);
|
|
|
|
// The dialed peer is a configured auto-connect peer: that is the only
|
|
// shape the retry machinery will seed a schedule entry for, and it is the
|
|
// shape the blackhole strands.
|
|
let mut config = Config::new();
|
|
config.peers.push(crate::config::PeerConfig::new(
|
|
intended_local.npub(),
|
|
"udp",
|
|
"10.0.0.2:2121",
|
|
));
|
|
let mut initiator = make_hs_node(config).await;
|
|
|
|
assert!(
|
|
initiator.node.peering.reconciler.retry_pending.is_empty(),
|
|
"nothing is scheduled before the dial"
|
|
);
|
|
|
|
initiator
|
|
.node
|
|
.initiate_connection(
|
|
initiator.transport_id,
|
|
attacker.addr.clone(),
|
|
Some(intended_identity),
|
|
)
|
|
.await
|
|
.expect("named dial");
|
|
|
|
let msg1 = recv_phase(&mut attacker.packet_rx, 1, "msg1").await;
|
|
attacker.node.handle_msg1(msg1).await;
|
|
let forged_msg2 = recv_phase(&mut initiator.packet_rx, 2, "forged msg2").await;
|
|
initiator.node.handle_msg2(forged_msg2).await;
|
|
|
|
assert_eq!(initiator.node.peer_count(), 0, "no promotion");
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.peering
|
|
.reconciler
|
|
.retry_pending
|
|
.contains_key(&intended_addr),
|
|
"the rejected dial must leave the peer we dialed scheduled for retry, \
|
|
or the configured peer is never dialed again"
|
|
);
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.peering
|
|
.reconciler
|
|
.retry_pending
|
|
.contains_key(&attacker_addr),
|
|
"the retry must name the peer we dialed, never the static that answered"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut attacker).await;
|
|
}
|
|
|
|
/// The same named dial, answered by the peer it named, still promotes - the
|
|
/// gate must be invisible on the legitimate path.
|
|
#[tokio::test]
|
|
async fn test_dial_msg2_matching_static_promotes() {
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
|
|
let responder_identity =
|
|
PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full());
|
|
let responder_addr = *responder_identity.node_addr();
|
|
|
|
initiator
|
|
.node
|
|
.initiate_connection(
|
|
initiator.transport_id,
|
|
responder.addr.clone(),
|
|
Some(responder_identity),
|
|
)
|
|
.await
|
|
.expect("named dial");
|
|
let leg_link = initiator.node.connections().next().unwrap().1.link_id();
|
|
|
|
let msg1 = recv_phase(&mut responder.packet_rx, 1, "msg1").await;
|
|
responder.node.handle_msg1(msg1).await;
|
|
let msg2 = recv_phase(&mut initiator.packet_rx, 2, "msg2").await;
|
|
initiator.node.handle_msg2(msg2).await;
|
|
|
|
assert_eq!(initiator.node.peer_count(), 1);
|
|
let peer = initiator.node.get_peer(&responder_addr).expect("promoted");
|
|
assert_eq!(peer.link_id(), leg_link, "promote keeps the leg's link");
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
// The responder completes its own side from the msg3 the gate let through.
|
|
let msg3 = recv_phase(&mut responder.packet_rx, 3, "msg3").await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
responder.node.debug_assert_peer_maps_coherent();
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// The anonymous carve-out. A shared-media dial names nobody, so the msg2
|
|
/// static is the ONLY identity the leg will ever have and there is no intent
|
|
/// for it to contradict - it must promote exactly as before. A gate that
|
|
/// pinned the wrong thing (or pinned unconditionally) stops anonymous
|
|
/// discovery promoting anything at all, and this is where that shows up.
|
|
///
|
|
/// Whether a leg is anonymous is settled here, at construction, from what the
|
|
/// caller passed - never from anything on the wire - so no responder can steer
|
|
/// a named dial onto this path.
|
|
#[tokio::test]
|
|
async fn test_anonymous_dial_msg2_promotes_whoever_answers() {
|
|
let mut initiator = make_hs_node(Config::new()).await;
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
initiator
|
|
.node
|
|
.initiate_connection(initiator.transport_id, responder.addr.clone(), None)
|
|
.await
|
|
.expect("anonymous dial");
|
|
let leg_link = initiator.node.connections().next().unwrap().1.link_id();
|
|
assert_eq!(
|
|
initiator
|
|
.node
|
|
.peer_machines
|
|
.get(&leg_link)
|
|
.unwrap()
|
|
.conn_dialed_identity(),
|
|
None,
|
|
"an anonymous dial records no dial intent, which is what selects the \
|
|
no-comparison branch"
|
|
);
|
|
|
|
let msg1 = recv_phase(&mut responder.packet_rx, 1, "msg1").await;
|
|
responder.node.handle_msg1(msg1).await;
|
|
let msg2 = recv_phase(&mut initiator.packet_rx, 2, "msg2").await;
|
|
initiator.node.handle_msg2(msg2).await;
|
|
|
|
assert_eq!(
|
|
initiator.node.peer_count(),
|
|
1,
|
|
"an anonymous dial promotes whoever answered it"
|
|
);
|
|
let peer = initiator.node.get_peer(&responder_addr).expect("promoted");
|
|
assert_eq!(peer.link_id(), leg_link);
|
|
initiator.node.debug_assert_peer_maps_coherent();
|
|
|
|
let msg3 = recv_phase(&mut responder.packet_rx, 3, "msg3").await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(responder.node.peer_count(), 1);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
// ===========================================================================
|
|
// Reject-arm resource release on the XX handshake path
|
|
//
|
|
// Every arm below disposes a leg and must return what that leg allocated. The
|
|
// discriminating assertion is always allocator-level and names the index:
|
|
// `is_allocated(idx)` before, `!is_allocated(idx)` after, plus a `count()`
|
|
// limb that catches a fix which frees an extra index. A `peers_by_index`
|
|
// assertion is worthless at all of these sites — nothing inserts there ahead
|
|
// of the msg3 gates, so its pass value and its failure value coincide.
|
|
// ===========================================================================
|
|
|
|
/// Hand-carry one XX exchange up to the responder's parked msg3 wait, with no
|
|
/// transports involved. Returns the responder's link, its msg1-allocated index,
|
|
/// and the initiator leg (which still owes a msg3).
|
|
async fn park_inbound_leg(
|
|
node_b: &mut Node,
|
|
initiator_keypair: secp256k1::Keypair,
|
|
initiator_epoch: [u8; 8],
|
|
transport_id: TransportId,
|
|
remote_addr: &TransportAddr,
|
|
sender_idx: SessionIndex,
|
|
) -> (LinkId, SessionIndex, PeerMachine) {
|
|
use crate::proto::fmp::wire::build_msg1;
|
|
|
|
let peer_b_identity = PeerIdentity::from_pubkey_full(node_b.identity().pubkey_full());
|
|
let mut conn_a = outbound_leg(LinkId::new(1), peer_b_identity, 1000);
|
|
let noise_msg1 = conn_a
|
|
.start_handshake(initiator_keypair, initiator_epoch, 1000)
|
|
.unwrap();
|
|
|
|
node_b
|
|
.handle_msg1(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr.clone(),
|
|
build_msg1(sender_idx, &noise_msg1),
|
|
1000,
|
|
))
|
|
.await;
|
|
|
|
let link_id_b = *node_b
|
|
.peer_machines
|
|
.keys()
|
|
.next()
|
|
.expect("msg1 parks a machine on the responder");
|
|
let our_index_b = node_b
|
|
.peer_machines
|
|
.get(&link_id_b)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("msg1 allocates the responder's session index");
|
|
(link_id_b, our_index_b, conn_a)
|
|
}
|
|
|
|
/// Finish the initiator's half: read the framed msg2 off the responder's
|
|
/// carrier and produce the noise msg3, optionally carrying a negotiation
|
|
/// payload.
|
|
fn answer_with_msg3(
|
|
node_b: &Node,
|
|
link_id_b: LinkId,
|
|
conn_a: &mut PeerMachine,
|
|
negotiation: Option<&[u8]>,
|
|
) -> Vec<u8> {
|
|
use crate::proto::fmp::wire::Msg2Header;
|
|
|
|
let wire_msg2 = node_b
|
|
.peer_machines
|
|
.get(&link_id_b)
|
|
.unwrap()
|
|
.conn_handshake_msg2()
|
|
.expect("msg1 stores the framed msg2 on the carrier")
|
|
.to_vec();
|
|
let msg2_header = Msg2Header::parse(&wire_msg2).unwrap();
|
|
let (noise_msg3, _) = conn_a
|
|
.complete_handshake(msg2_header.noise_msg2(&wire_msg2), negotiation, 1100)
|
|
.unwrap();
|
|
noise_msg3
|
|
}
|
|
|
|
/// S3/T1 — the orphaned pending-inbound arm frees the index it can only read
|
|
/// off the wire, and disposes the link with it.
|
|
///
|
|
/// The planted state is a `pending_inbound` entry whose link has lost its
|
|
/// machine. It is produced through `remove_peer_machine`, the disposal
|
|
/// choke-point, so the peer's timer store goes with it exactly as a real
|
|
/// teardown would leave things.
|
|
///
|
|
/// `unknown_connection` is this arm's own counter, which separates it from
|
|
/// every `bad_state` arm. It does NOT separate it from the
|
|
/// `pending_inbound`-miss arm above, which records the same counter — the
|
|
/// `is_allocated` transition is what does, since that arm has no index to free.
|
|
#[tokio::test]
|
|
async fn test_msg3_orphaned_pending_inbound_frees_index() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let node_a = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let (link_id_b, our_index_b, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
node_a.identity().keypair(),
|
|
node_a.startup_epoch(),
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
let noise_msg3 = answer_with_msg3(&node_b, link_id_b, &mut conn_a, None);
|
|
|
|
// The unaudited disposal this arm exists to survive: the machine goes, the
|
|
// pending-inbound entry does not.
|
|
node_b.remove_peer_machine(link_id_b);
|
|
assert!(
|
|
node_b
|
|
.pending_inbound
|
|
.contains_key(&(transport_id, our_index_b.as_u32())),
|
|
"control: the pending-inbound entry outlived its machine"
|
|
);
|
|
|
|
let baseline = node_b.index_allocator.count();
|
|
assert!(
|
|
node_b.index_allocator.is_allocated(our_index_b),
|
|
"control: the leg allocated an index"
|
|
);
|
|
assert!(
|
|
node_b.links.contains_key(&link_id_b),
|
|
"control: the orphaned leg still holds a link"
|
|
);
|
|
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, our_index_b, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
|
|
assert!(
|
|
!node_b.index_allocator.is_allocated(our_index_b),
|
|
"the orphaned pending-inbound arm must return the index it allocated"
|
|
);
|
|
assert_eq!(
|
|
node_b.index_allocator.count(),
|
|
baseline - 1,
|
|
"and must free exactly one"
|
|
);
|
|
assert!(
|
|
!node_b.links.contains_key(&link_id_b),
|
|
"the link goes with the orphaned entry"
|
|
);
|
|
assert_eq!(
|
|
node_b.stats().handshake.unknown_connection,
|
|
1,
|
|
"this arm records its own reject counter"
|
|
);
|
|
}
|
|
|
|
/// S3/T2a — a hostile msg3 naming a PROMOTED peer's index must not free it.
|
|
///
|
|
/// `receiver_idx` is chosen by the sender. Freeing it the way the other arms
|
|
/// free would turn a slow leak into a remote session-teardown primitive: name
|
|
/// an index belonging to an unrelated live session and the node hands it back
|
|
/// to the allocator. The ownership check is what stops that.
|
|
#[tokio::test]
|
|
async fn test_msg3_orphaned_entry_naming_live_peer_index_does_not_free() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let node_a = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let (link_id_b, our_index_b, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
node_a.identity().keypair(),
|
|
node_a.startup_epoch(),
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
let noise_msg3 = answer_with_msg3(&node_b, link_id_b, &mut conn_a, None);
|
|
|
|
// First msg3 promotes: the responder now holds a live session index.
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr.clone(),
|
|
build_msg3(sender_idx, our_index_b, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
assert_eq!(node_b.peer_count(), 1, "control: the first msg3 promoted");
|
|
let peer_addr = *PeerIdentity::from_pubkey_full(node_a.identity().pubkey_full()).node_addr();
|
|
let victim_idx = node_b
|
|
.get_peer(&peer_addr)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("a promoted peer holds a session index");
|
|
let victim_link = node_b.get_peer(&peer_addr).unwrap().link_id();
|
|
assert!(
|
|
node_b
|
|
.peers_by_index
|
|
.contains_key(&(transport_id, victim_idx.as_u32())),
|
|
"control: the live session index is registered for dispatch"
|
|
);
|
|
|
|
// The aliasing state: a stale pending-inbound entry naming the LIVE index,
|
|
// pointing at a link that has no machine.
|
|
let dead_link = LinkId::new(4242);
|
|
assert!(!node_b.peer_machines.contains_key(&dead_link));
|
|
node_b
|
|
.pending_inbound
|
|
.insert((transport_id, victim_idx.as_u32()), dead_link);
|
|
|
|
let baseline = node_b.index_allocator.count();
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, victim_idx, &noise_msg3),
|
|
1300,
|
|
))
|
|
.await;
|
|
|
|
assert!(
|
|
node_b.index_allocator.is_allocated(victim_idx),
|
|
"a live peer's session index must never be freed by a wire-named \
|
|
orphan cleanup"
|
|
);
|
|
assert_eq!(
|
|
node_b.index_allocator.count(),
|
|
baseline,
|
|
"nothing at all was freed"
|
|
);
|
|
assert!(
|
|
node_b
|
|
.peers_by_index
|
|
.contains_key(&(transport_id, victim_idx.as_u32())),
|
|
"the victim's dispatch entry survives"
|
|
);
|
|
assert_eq!(node_b.peer_count(), 1, "the victim peer survives");
|
|
assert!(node_b.links.contains_key(&victim_link));
|
|
assert!(node_b.peer_machines.contains_key(&victim_link));
|
|
}
|
|
|
|
/// S3/T2b — the same hostile shape against a victim `peers_by_index` cannot
|
|
/// see: a live PENDING INBOUND leg, whose index exists only on its machine.
|
|
///
|
|
/// T2a alone proves only the predicate's first limb, since a promoted peer
|
|
/// short-circuits it. This is the population that motivates the rest of the
|
|
/// scan: deleting the `peer_machines` limb must red this test and leave T2a
|
|
/// green.
|
|
#[tokio::test]
|
|
async fn test_msg3_orphaned_entry_naming_pending_leg_index_does_not_free() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let node_a = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let (victim_link, victim_idx, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
node_a.identity().keypair(),
|
|
node_a.startup_epoch(),
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
let noise_msg3 = answer_with_msg3(&node_b, victim_link, &mut conn_a, None);
|
|
|
|
// The victim's index is held by a parked machine and by nothing else: no
|
|
// peer exists yet, and the msg3 gate consumes the pending-inbound key
|
|
// before the ownership check runs.
|
|
assert_eq!(node_b.peer_count(), 0);
|
|
assert!(node_b.peers_by_index.is_empty());
|
|
assert!(node_b.pending_outbound.is_empty());
|
|
|
|
// Overwrite the victim's own pending-inbound entry with a machine-less
|
|
// link. The point is a link with no machine, which is what selects the arm.
|
|
let dead_link = LinkId::new(4242);
|
|
assert!(!node_b.peer_machines.contains_key(&dead_link));
|
|
node_b
|
|
.pending_inbound
|
|
.insert((transport_id, victim_idx.as_u32()), dead_link);
|
|
|
|
let baseline = node_b.index_allocator.count();
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, victim_idx, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
|
|
assert!(
|
|
node_b.index_allocator.is_allocated(victim_idx),
|
|
"an index held by a live pending leg must never be freed by a \
|
|
wire-named orphan cleanup"
|
|
);
|
|
assert_eq!(node_b.index_allocator.count(), baseline);
|
|
assert!(
|
|
node_b.peer_machines.contains_key(&victim_link),
|
|
"the victim leg's machine survives"
|
|
);
|
|
assert!(node_b.links.contains_key(&victim_link));
|
|
}
|
|
|
|
/// S3/T2c — the same hostile shape delivered on a DIFFERENT transport from the
|
|
/// one the victim's index is registered under.
|
|
///
|
|
/// This is the discriminating test for the one regression the ownership check
|
|
/// has: re-scoping `session_index_is_claimed` to the transport the msg3 arrived
|
|
/// on. `index_allocator` is a single node-global `HashSet<u32>` while every
|
|
/// registry it is checked against is keyed `(TransportId, u32)`, so a
|
|
/// transport-filtered scan reports "unclaimed" for an index that is very much
|
|
/// claimed, on another transport — and the arm hands a live session's index
|
|
/// back to the allocator, which is exactly the teardown primitive the check
|
|
/// exists to prevent.
|
|
///
|
|
/// The victim has to be chosen with care, and T2a's victim will not do. A
|
|
/// promoted peer's `our_index` also sits on that peer's surviving `PeerMachine`,
|
|
/// and the `peer_machines` limb has no transport in its key to filter by, so it
|
|
/// masks the mutation and the test reports green either way. An in-flight
|
|
/// REKEY index is the one live index held by no machine: it lives in
|
|
/// `pending_outbound` under the peer's transport and on the `ActivePeer`, both
|
|
/// of which a transport-scoped scan filters out. The two controls below pin
|
|
/// that.
|
|
#[tokio::test]
|
|
async fn test_msg3_orphan_on_other_transport_does_not_free_live_index() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let make_config = || {
|
|
let mut c = Config::new();
|
|
c.node.rekey.enabled = true;
|
|
c.node.rekey.after_secs = 30;
|
|
c
|
|
};
|
|
|
|
let mut initiator = make_hs_node(make_config()).await;
|
|
let mut responder = make_hs_node(make_config()).await;
|
|
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(initiator.node.peer_count(), 1, "control: the session is up");
|
|
|
|
// Start a real rekey so the victim index is produced exactly as production
|
|
// produces it.
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
initiator.node.check_rekey().await;
|
|
|
|
let victim_idx = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_our_index()
|
|
.expect("check_rekey started a cycle and allocated its index");
|
|
|
|
// Control 1: the victim is registered under the peer's OWN transport only.
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(initiator.transport_id, victim_idx.as_u32())),
|
|
"control: the rekey index is registered under the peer's transport"
|
|
);
|
|
// Control 2: no machine holds it. This is what makes the test able to see a
|
|
// transport-scoped scan at all — the `peer_machines` limb is unfiltered, so
|
|
// a machine-held victim would mask the regression.
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.peer_machines
|
|
.values()
|
|
.any(|m| m.our_index() == Some(victim_idx)),
|
|
"control: no PeerMachine holds the rekey index, so only the \
|
|
transport-keyed limbs can answer for it"
|
|
);
|
|
|
|
// The hostile msg3: a different transport, an orphaned pending-inbound entry
|
|
// naming the live rekey index, and a link with no machine.
|
|
let other_transport = TransportId::new(99);
|
|
assert_ne!(other_transport, initiator.transport_id);
|
|
let dead_link = LinkId::new(4242);
|
|
assert!(!initiator.node.peer_machines.contains_key(&dead_link));
|
|
initiator
|
|
.node
|
|
.pending_inbound
|
|
.insert((other_transport, victim_idx.as_u32()), dead_link);
|
|
|
|
let baseline = initiator.node.index_allocator.count();
|
|
initiator
|
|
.node
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
other_transport,
|
|
TransportAddr::from_string("127.0.0.1:5999"),
|
|
build_msg3(
|
|
SessionIndex::new(7),
|
|
victim_idx,
|
|
&[0u8; crate::noise::HANDSHAKE_MSG3_SIZE],
|
|
),
|
|
1300,
|
|
))
|
|
.await;
|
|
|
|
assert!(
|
|
initiator.node.index_allocator.is_allocated(victim_idx),
|
|
"an index claimed on ANOTHER transport must not be freed: the \
|
|
allocator is node-global, so the ownership scan must be too"
|
|
);
|
|
assert_eq!(
|
|
initiator.node.index_allocator.count(),
|
|
baseline,
|
|
"nothing at all was freed"
|
|
);
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(initiator.transport_id, victim_idx.as_u32())),
|
|
"the in-flight rekey's dispatch entry survives"
|
|
);
|
|
assert_eq!(initiator.node.peer_count(), 1, "the victim peer survives");
|
|
}
|
|
|
|
/// S3/T2d — the `ActivePeer` limb of the ownership scan, pinned directly.
|
|
///
|
|
/// **The state below is fabricated and is not reachable in production.** Every
|
|
/// `self.peers` insertion goes through `ActivePeer::with_session` with a
|
|
/// concrete `TransportId` and an unconditional `peers_by_index` insert beside
|
|
/// it, so today a promoted peer's `our_index` is always answerable from the
|
|
/// maps and from its surviving machine. T2a and T2b ride those two limbs, and
|
|
/// deleting the `peers` scan entirely leaves the whole suite green — which is
|
|
/// the reason this test exists rather than a reason it should not.
|
|
///
|
|
/// The limb is defence-in-depth against holders the transport-keyed maps cannot
|
|
/// answer for: `remove_active_peer` gates its index frees on `if let Some(tid)`,
|
|
/// and the `peers_by_index` insert for a rekeying peer's `pending_our_index` is
|
|
/// gated the same way, so an index can in principle be held by an `ActivePeer`
|
|
/// and by nothing else. This calls the predicate directly rather than dressing
|
|
/// that state up as an arrival, because pretending it is reachable would be the
|
|
/// worse dishonesty.
|
|
#[tokio::test]
|
|
async fn test_session_index_is_claimed_answers_from_the_peer_when_maps_cannot() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let node_a = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let (link_id_b, our_index_b, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
node_a.identity().keypair(),
|
|
node_a.startup_epoch(),
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
let noise_msg3 = answer_with_msg3(&node_b, link_id_b, &mut conn_a, None);
|
|
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, our_index_b, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
assert_eq!(node_b.peer_count(), 1, "control: the peer promoted");
|
|
|
|
let peer_addr = *PeerIdentity::from_pubkey_full(node_a.identity().pubkey_full()).node_addr();
|
|
let victim_idx = node_b
|
|
.get_peer(&peer_addr)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("a promoted peer holds a session index");
|
|
let victim_link = node_b.get_peer(&peer_addr).unwrap().link_id();
|
|
|
|
assert!(
|
|
node_b.session_index_is_claimed(victim_idx),
|
|
"control: with every registry intact the index reads as claimed"
|
|
);
|
|
|
|
// Strip the two limbs that answer today, leaving the `ActivePeer` as the
|
|
// only holder. Fabricated, per the doc comment above.
|
|
node_b
|
|
.peers_by_index
|
|
.remove(&(transport_id, victim_idx.as_u32()));
|
|
node_b.remove_peer_machine(victim_link);
|
|
assert!(
|
|
!node_b
|
|
.peers_by_index
|
|
.keys()
|
|
.any(|(_, i)| *i == victim_idx.as_u32()),
|
|
"control: no transport-keyed registry names the index any more"
|
|
);
|
|
assert!(
|
|
!node_b
|
|
.peer_machines
|
|
.values()
|
|
.any(|m| m.our_index() == Some(victim_idx)),
|
|
"control: no machine names it either"
|
|
);
|
|
|
|
assert!(
|
|
node_b.session_index_is_claimed(victim_idx),
|
|
"the peer itself still holds the index, so the orphan arm must not \
|
|
free it"
|
|
);
|
|
}
|
|
|
|
/// S4 — the msg3 FMP-negotiation reject arm must free the msg1-allocated index.
|
|
///
|
|
/// The payload must be non-empty or the responder's split yields `None` and the
|
|
/// negotiation block is skipped entirely. Three bytes of `0xff` decrypt fine
|
|
/// and then fail `NegotiationPayload::decode`'s length gate, which is the arm
|
|
/// this test is aimed at.
|
|
///
|
|
/// This arm sits ahead of the inbound ACL gate, so any peer able to complete a
|
|
/// Noise msg3 reaches it without being authorized.
|
|
///
|
|
/// The counters alone do not identify the arm: the msg3-processing-failure arm
|
|
/// 25 lines above records the identical `bad_state` + empty registries AND
|
|
/// already frees the index, so a harness slip that makes the msg3 fail to
|
|
/// DECRYPT rather than fail to DECODE would pass every assertion here with or
|
|
/// without the fix. `test_msg3_negotiation_wellformed_promotes` is the control
|
|
/// for that: it runs the identical construction with a well-formed payload and
|
|
/// requires promotion, which can only happen if the appended bytes decrypt.
|
|
#[tokio::test]
|
|
async fn test_msg3_negotiation_failure_frees_index() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let node_a = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let (link_id_b, our_index_b, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
node_a.identity().keypair(),
|
|
node_a.startup_epoch(),
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
|
|
let bad_neg = vec![0xffu8; 3];
|
|
let noise_msg3 = answer_with_msg3(&node_b, link_id_b, &mut conn_a, Some(&bad_neg));
|
|
|
|
let baseline = node_b.index_allocator.count();
|
|
assert!(
|
|
node_b.index_allocator.is_allocated(our_index_b),
|
|
"control: msg1 allocated the responder's index"
|
|
);
|
|
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, our_index_b, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
|
|
assert!(
|
|
!node_b.index_allocator.is_allocated(our_index_b),
|
|
"the negotiation-failure arm must return the msg1-allocated index"
|
|
);
|
|
assert_eq!(
|
|
node_b.index_allocator.count(),
|
|
baseline - 1,
|
|
"and must free exactly one"
|
|
);
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
0,
|
|
"a failed negotiation never promotes"
|
|
);
|
|
assert!(node_b.peer_machines.is_empty());
|
|
assert_eq!(node_b.link_count(), 0);
|
|
assert_eq!(node_b.stats().handshake.bad_state, 1);
|
|
}
|
|
|
|
/// The positive sibling of `test_msg3_negotiation_failure_frees_index`, and the
|
|
/// only thing that proves the appended payload reaches the responder's
|
|
/// negotiation step at all rather than failing earlier in the Noise decrypt.
|
|
#[tokio::test]
|
|
async fn test_msg3_negotiation_wellformed_promotes() {
|
|
use crate::proto::fmp::NegotiationPayload;
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let node_a = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let (link_id_b, our_index_b, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
node_a.identity().keypair(),
|
|
node_a.startup_epoch(),
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
|
|
let good_neg = NegotiationPayload::fmp(1, 1, node_a.node_profile()).encode();
|
|
let noise_msg3 = answer_with_msg3(&node_b, link_id_b, &mut conn_a, Some(&good_neg));
|
|
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, our_index_b, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
|
|
assert_eq!(
|
|
node_b.peer_count(),
|
|
1,
|
|
"a well-formed negotiation payload on the same construction promotes, \
|
|
which is what proves the appended bytes decrypt on the responder"
|
|
);
|
|
assert_eq!(node_b.stats().handshake.bad_state, 0);
|
|
}
|
|
|
|
/// S7 — the msg3 self-connect drop must free the msg1-allocated index.
|
|
///
|
|
/// A real self-dial cannot reach this arm: the msg2 self-connect drop takes the
|
|
/// leg down first. It is constructed directly by handing the initiator leg
|
|
/// node_b's OWN keypair, so the static in msg3 is node_b's own. The inbound ACL
|
|
/// gate runs first and default-allows our own identity, so the arm is reached.
|
|
///
|
|
/// Limitation, stated rather than fixed: the assertions do not by themselves
|
|
/// identify this arm. The msg3-processing-failure arm above records the same
|
|
/// `bad_state == 1` with the same empty registries and already frees the index,
|
|
/// so a harness slip that made the hand-built msg3 fail to DECRYPT would leave
|
|
/// every assertion here green with or without S7's fix. That the test lands on
|
|
/// S7 today was established by mutation — deleting S7's free reds it — not by
|
|
/// anything the assertions can distinguish. Unlike S4, no positive sibling
|
|
/// control is available here: a well-formed self-connect msg3 cannot promote,
|
|
/// which is the point of the arm.
|
|
#[tokio::test]
|
|
async fn test_msg3_self_connect_frees_index() {
|
|
use crate::proto::fmp::wire::build_msg3;
|
|
|
|
let mut node_b = make_node();
|
|
let transport_id = TransportId::new(1);
|
|
let remote_addr = TransportAddr::from_string("127.0.0.1:5000");
|
|
let sender_idx = SessionIndex::new(7);
|
|
|
|
let own_keypair = node_b.identity().keypair();
|
|
let own_epoch = node_b.startup_epoch();
|
|
let (link_id_b, our_index_b, mut conn_a) = park_inbound_leg(
|
|
&mut node_b,
|
|
own_keypair,
|
|
own_epoch,
|
|
transport_id,
|
|
&remote_addr,
|
|
sender_idx,
|
|
)
|
|
.await;
|
|
let noise_msg3 = answer_with_msg3(&node_b, link_id_b, &mut conn_a, None);
|
|
|
|
let baseline = node_b.index_allocator.count();
|
|
assert!(
|
|
node_b.index_allocator.is_allocated(our_index_b),
|
|
"control: msg1 allocated the responder's index"
|
|
);
|
|
|
|
node_b
|
|
.handle_msg3(ReceivedPacket::with_timestamp(
|
|
transport_id,
|
|
remote_addr,
|
|
build_msg3(sender_idx, our_index_b, &noise_msg3),
|
|
1200,
|
|
))
|
|
.await;
|
|
|
|
assert!(
|
|
!node_b.index_allocator.is_allocated(our_index_b),
|
|
"the msg3 self-connect drop must return the msg1-allocated index"
|
|
);
|
|
assert_eq!(
|
|
node_b.index_allocator.count(),
|
|
baseline - 1,
|
|
"and must free exactly one"
|
|
);
|
|
assert_eq!(node_b.peer_count(), 0, "we never promote ourselves");
|
|
assert_eq!(node_b.link_count(), 0);
|
|
assert_eq!(node_b.stats().handshake.bad_state, 1);
|
|
}
|
|
|
|
/// S1b — the outbound ACL reject at msg2 must put the dial back on the retry
|
|
/// schedule, or a configured peer is dialed once at startup and never again.
|
|
///
|
|
/// Ordering matters and is a vacuous-pass trap: `initiate_connection` consults
|
|
/// the ACL itself and returns `AccessDenied`, so denying BEFORE the dial means
|
|
/// no msg1 is ever sent, the arm is never reached, and the `retry_pending`
|
|
/// assertion passes for the wrong reason. The deny is written and reloaded
|
|
/// AFTER the dial, which is also the runtime-reload scenario that motivates
|
|
/// the fix.
|
|
///
|
|
/// The emptiness control is anchored twice — after the dial and again after the
|
|
/// reload — because three sites insert into `retry_pending`, so a single
|
|
/// pre-dial check would not attribute the entry to this arm.
|
|
///
|
|
/// Not asserted here: that the reschedule names the DIALED peer rather than the
|
|
/// learned static. In this scenario the two are the same node, so the limb
|
|
/// cannot fail. The sibling gate's own test covers that with two distinct
|
|
/// identities.
|
|
#[tokio::test]
|
|
async fn test_outbound_msg2_acl_reject_reschedules_dial() {
|
|
use crate::node::acl::PeerAclReloader;
|
|
|
|
let mut responder = make_hs_node(Config::new()).await;
|
|
let responder_identity =
|
|
PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full());
|
|
let responder_addr = *responder_identity.node_addr();
|
|
|
|
// The dialed peer is a configured auto-connect peer: the only shape the
|
|
// retry machinery seeds a schedule entry for, and the shape the missing
|
|
// reschedule strands.
|
|
let mut config = Config::new();
|
|
config.peers.push(crate::config::PeerConfig::new(
|
|
responder.node.npub(),
|
|
"udp",
|
|
"10.0.0.2:2121",
|
|
));
|
|
let mut initiator = make_hs_node(config).await;
|
|
|
|
let dir = tempfile::tempdir().unwrap();
|
|
initiator.node.peer_acl = PeerAclReloader::with_paths(
|
|
dir.path().join("peers.allow"),
|
|
dir.path().join("peers.deny"),
|
|
);
|
|
|
|
initiator
|
|
.node
|
|
.initiate_connection(
|
|
initiator.transport_id,
|
|
responder.addr.clone(),
|
|
Some(responder_identity),
|
|
)
|
|
.await
|
|
.expect("named dial");
|
|
assert!(
|
|
initiator.node.peering.reconciler.retry_pending.is_empty(),
|
|
"nothing is scheduled by the dial itself"
|
|
);
|
|
|
|
// Deny only now, so the dial really happened and the msg2 arm is the gate
|
|
// that turns the peer away.
|
|
std::fs::write(
|
|
dir.path().join("peers.deny"),
|
|
format!("{}\n", responder.node.npub()),
|
|
)
|
|
.unwrap();
|
|
assert!(initiator.node.reload_peer_acl().await);
|
|
assert!(
|
|
initiator.node.peering.reconciler.retry_pending.is_empty(),
|
|
"the reload itself schedules nothing"
|
|
);
|
|
|
|
let msg1 = recv_phase(&mut responder.packet_rx, 1, "msg1").await;
|
|
responder.node.handle_msg1(msg1).await;
|
|
let msg2 = recv_phase(&mut initiator.packet_rx, 2, "msg2").await;
|
|
initiator.node.handle_msg2(msg2).await;
|
|
|
|
// Arm identity: without these, an ACL that failed to load produces a
|
|
// promoted peer and an empty schedule, and the assertion below would be
|
|
// testing nothing.
|
|
assert_eq!(
|
|
initiator.node.peer_count(),
|
|
0,
|
|
"the denied peer never promotes"
|
|
);
|
|
assert_eq!(
|
|
initiator.node.stats().handshake.bad_state,
|
|
1,
|
|
"the denial is attributed to the handshake state-machine counter"
|
|
);
|
|
assert!(
|
|
initiator.node.peer_machines.is_empty(),
|
|
"the leg is disposed"
|
|
);
|
|
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.peering
|
|
.reconciler
|
|
.retry_pending
|
|
.contains_key(&responder_addr),
|
|
"the ACL-rejected dial must leave the configured peer scheduled for \
|
|
retry, or it is never dialed again for the life of the process"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|
|
|
|
/// S8 — abandoning a rekey cycle must return the rekey index AND clear the
|
|
/// `pending_outbound` entry seeded with it.
|
|
///
|
|
/// `handshake_max_resends = 0` fires the abandon on the first
|
|
/// `resend_pending_rekeys` call, because the classifier tests
|
|
/// `resend_count >= max_resends` BEFORE it consults the resend-due predicate.
|
|
/// That takes the wall clock out of the path entirely: the rekey deadline is
|
|
/// seeded from `SystemTime::now()`, so a synthetic `now_ms` can never make the
|
|
/// msg1 look due and the budget would never be spent.
|
|
///
|
|
/// The rekey msg2 is deliberately never delivered — that is what makes this the
|
|
/// abandon path.
|
|
///
|
|
/// No assertion is made on `peers_by_index` for the rekey index: nothing
|
|
/// inserts there until the rekey msg2 install, so it would be inert here.
|
|
#[tokio::test]
|
|
async fn test_abandon_rekey_frees_index_and_pending_outbound() {
|
|
let make_config = || {
|
|
let mut c = Config::new();
|
|
c.node.rekey.enabled = true;
|
|
c.node.rekey.after_secs = 30;
|
|
c.node.rate_limit.handshake_max_resends = 0;
|
|
c
|
|
};
|
|
|
|
let mut initiator = make_hs_node(make_config()).await;
|
|
let mut responder = make_hs_node(make_config()).await;
|
|
|
|
let responder_addr =
|
|
*PeerIdentity::from_pubkey_full(responder.node.identity().pubkey_full()).node_addr();
|
|
|
|
let msg3 = drive_to_msg3(&mut initiator, &mut responder, 1000).await;
|
|
responder.node.handle_msg3(msg3).await;
|
|
assert_eq!(initiator.node.peer_count(), 1);
|
|
|
|
let session_idx = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.our_index()
|
|
.expect("the established peer holds a session index");
|
|
|
|
// Age the session past the (jittered) trigger and let the real cadence fire
|
|
// so the rekey index and its pending_outbound entry are produced exactly as
|
|
// in production.
|
|
initiator
|
|
.node
|
|
.get_peer_mut(&responder_addr)
|
|
.unwrap()
|
|
.test_backdate_session_established(std::time::Duration::from_secs(120));
|
|
initiator.node.check_rekey().await;
|
|
|
|
let rekey_idx = initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_our_index()
|
|
.expect("check_rekey started a cycle and allocated its index");
|
|
let baseline = initiator.node.index_allocator.count();
|
|
assert!(
|
|
initiator.node.index_allocator.is_allocated(rekey_idx),
|
|
"control: the rekey allocated an index"
|
|
);
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(initiator.transport_id, rekey_idx.as_u32())),
|
|
"control: the rekey registered its index for msg2 dispatch"
|
|
);
|
|
|
|
// The rekey msg2 never arrives; the first poll spends the (zero) budget.
|
|
initiator.node.resend_pending_rekeys(2000).await;
|
|
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_in_progress(),
|
|
"control: the abandon actually fired"
|
|
);
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.get_peer(&responder_addr)
|
|
.unwrap()
|
|
.rekey_our_index()
|
|
.is_none()
|
|
);
|
|
|
|
assert!(
|
|
!initiator.node.index_allocator.is_allocated(rekey_idx),
|
|
"the abandoned cycle must return its index"
|
|
);
|
|
assert_eq!(
|
|
initiator.node.index_allocator.count(),
|
|
baseline - 1,
|
|
"and must free exactly one"
|
|
);
|
|
assert!(
|
|
!initiator
|
|
.node
|
|
.pending_outbound
|
|
.contains_key(&(initiator.transport_id, rekey_idx.as_u32())),
|
|
"the abandoned cycle must clear its msg2 dispatch entry"
|
|
);
|
|
|
|
// The limb that catches a fix binding the wrong index: the live session must
|
|
// be untouched.
|
|
assert_eq!(initiator.node.peer_count(), 1, "the session survives");
|
|
assert!(
|
|
initiator.node.index_allocator.is_allocated(session_idx),
|
|
"the established session's own index must not be freed"
|
|
);
|
|
assert!(
|
|
initiator
|
|
.node
|
|
.peers_by_index
|
|
.contains_key(&(initiator.transport_id, session_idx.as_u32())),
|
|
"the established session stays registered for dispatch"
|
|
);
|
|
|
|
stop_hs(&mut initiator).await;
|
|
stop_hs(&mut responder).await;
|
|
}
|