mirror of
https://relay.ngit.dev/npub15qydau2hjma6ngxkl2cyar74wzyjshvl65za5k5rl69264ar2exs5cyejr/ngit-grasp.git
synced 2026-10-05 15:08:24 +00:00
Outbound sync answers NIP-42 challenges with the relay owner key on
every connection. When no owner key is available, the previous code
panicked at registration (`.expect`), and the retry machinery would
still have reserved a one-shot authentication retry that nothing could
ever fulfil.
Approach: `RelayConnection::new{,_with_database}` now take
`Option<Keys>` and only attach the SDK authenticator when present,
exposing `answers_auth_challenges()`. Without an authenticator an
auth-required CLOSED is terminal like any other CLOSED: the terminal
listener retires the subscription, the data lane releases its live
permit, and `handle_subscription_closed` skips the one-retry
reservation and goes straight to retirement plus the
AuthenticationRequired policy refusal (24h probe). `register_relay`
degrades gracefully to an unauthenticated connection with a warning
instead of panicking.
Correctness assumptions: rust-nostr only retains auth-refused
subscriptions for post-authentication resubscription when an
authenticator is configured, so every has_authenticator branch mirrors
an SDK behavior split; a reserved retry without an authenticator would
dangle until disconnect cleanup.
Test infrastructure: new AuthGatingRelay helper - a NIP-42 gate in
front of a backend relay that serves a plain NIP-11 document,
challenges every session, refuses queries pre-auth, marks negentropy
unsupported, and either bridges (Admit) or answers `restricted:`
(Restricted) after a valid AUTH, recording authenticated pubkeys and
REQ counts.
Validation: integration tests prove (1) a public instance
authenticates to a gated ordinary relay with its owner key and the
retained subscription is answered after AUTH (announcement reaches
purgatory through the gate, the instance's only event source), and
(2) a restricted refusal after successful authentication parks the
work - the gate's REQ count holds still for a full 2s observation
window. cargo test --lib (789 passed) and --test sync
sync::outbound_auth pass.
10903 lines
440 KiB
Rust
10903 lines
440 KiB
Rust
//! Proactive Sync Module - GRASP-02 v4 Implementation
|
||
//!
|
||
//! This module implements proactive synchronization of repository data from external
|
||
//! relays based on relay URLs listed in 30617 repository announcements.
|
||
//!
|
||
//! ## Architecture
|
||
//!
|
||
//! The sync system uses three index structures:
|
||
//! - `RepoSyncIndex` - What we WANT to sync (source of truth from self-subscription)
|
||
//! - `RelaySyncIndex` - What we have CONFIRMED syncing + connection state
|
||
//! - `PendingSyncIndex` - In-flight batches awaiting EOSE confirmation
|
||
//!
|
||
//! See `docs/explanation/grasp-02-proactive-sync.md` for full design details.
|
||
|
||
pub mod algorithms;
|
||
pub mod discovery;
|
||
pub mod filters;
|
||
pub mod health;
|
||
pub mod metrics;
|
||
pub mod missing_events;
|
||
pub mod naughty_list;
|
||
pub mod rejected_index;
|
||
pub mod relay_connection;
|
||
pub mod self_subscriber;
|
||
|
||
// Re-export core algorithm types
|
||
pub use algorithms::{AddFilters, RelaySyncNeeds};
|
||
|
||
// Re-export metrics types
|
||
pub use metrics::SyncMetrics;
|
||
|
||
// Re-export rejected index types
|
||
pub use rejected_index::{EventType, RejectionReason};
|
||
|
||
// Re-export relay connection types
|
||
pub use relay_connection::{
|
||
NegentropySyncResult, RelayConnection, RelayEvent, TransientRequestClass,
|
||
};
|
||
|
||
// Re-export self-subscriber types
|
||
pub use self_subscriber::SelfSubscriber;
|
||
|
||
// Re-export health tracking types
|
||
pub use health::RelayHealthTracker;
|
||
use std::collections::{HashMap, HashSet, VecDeque};
|
||
use std::path::{Path, PathBuf};
|
||
use std::sync::Arc;
|
||
use std::time::{Duration, Instant};
|
||
|
||
use futures_util::future::join_all;
|
||
use nostr_sdk::prelude::*;
|
||
use tokio::sync::{broadcast, Mutex, RwLock, Semaphore};
|
||
|
||
use crate::config::Config;
|
||
use crate::nostr::builder::Nip34WritePolicy;
|
||
use crate::nostr::SharedDatabase;
|
||
use crate::outbound::{
|
||
url_matches_service_domain, OutboundTargetKind, OutboundTargetPolicy, RelayTargetSource,
|
||
};
|
||
use crate::private::PrivateAccess;
|
||
use nostr_sdk::prelude::LocalRelay;
|
||
|
||
const MAX_PURGATORY_DEPENDENCY_EVENTS_PER_TICK: usize = 32;
|
||
const MAX_PURGATORY_FILTER_ACTIONS_PER_TICK: usize = 1;
|
||
const MAX_PURGATORY_DEPENDENCY_IDS_PER_QUERY: usize = 100;
|
||
const SEMANTIC_FALLBACK_MIN_REQUESTED_EVENTS: usize = 20;
|
||
const SEMANTIC_FALLBACK_MAX_DELIVERED_PERCENT: usize = 10;
|
||
const DESCENDANT_FALLBACK_OVERLAP_SECS: u64 = 15 * 60;
|
||
/// Maximum number of locally known parent/child generations expanded into a
|
||
/// relay's related-event query frontier.
|
||
const MAX_DESCENDANT_FRONTIER_DEPTH: usize = 8;
|
||
|
||
fn mailbox_probe_refresh_interval() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_secs(2)
|
||
} else {
|
||
Duration::from_secs(24 * 60 * 60)
|
||
}
|
||
}
|
||
|
||
fn mailbox_probe_retry_interval() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_millis(500)
|
||
} else {
|
||
Duration::from_secs(5 * 60)
|
||
}
|
||
}
|
||
|
||
fn due_mailbox_relays(
|
||
mailbox_roots: &HashMap<String, HashSet<EventId>>,
|
||
next_at: &HashMap<String, Instant>,
|
||
now: Instant,
|
||
) -> Vec<String> {
|
||
let mut relays: Vec<String> = mailbox_roots
|
||
.keys()
|
||
.filter(|relay| next_at.get(*relay).is_none_or(|due| *due <= now))
|
||
.cloned()
|
||
.collect();
|
||
relays.sort_by(|left, right| {
|
||
next_at
|
||
.get(left)
|
||
.copied()
|
||
.unwrap_or(now)
|
||
.cmp(&next_at.get(right).copied().unwrap_or(now))
|
||
.then_with(|| mailbox_roots[right].len().cmp(&mailbox_roots[left].len()))
|
||
.then_with(|| left.cmp(right))
|
||
});
|
||
relays
|
||
}
|
||
|
||
fn mailbox_probe_connection_ready(
|
||
connection_status: Option<ConnectionStatus>,
|
||
socket_connected: bool,
|
||
) -> bool {
|
||
socket_connected && connection_status.is_some_and(|status| status.is_live_sync_active())
|
||
}
|
||
|
||
fn select_due_mailbox_relay(
|
||
due_relays: &[String],
|
||
active_relays: &HashSet<String>,
|
||
) -> Option<String> {
|
||
due_relays
|
||
.iter()
|
||
.find(|relay| active_relays.contains(*relay))
|
||
.or_else(|| due_relays.first())
|
||
.cloned()
|
||
}
|
||
|
||
fn mailbox_probe_completion(
|
||
filter_index: usize,
|
||
filter_count: usize,
|
||
succeeded: bool,
|
||
) -> (usize, bool, Duration) {
|
||
let next_filter = (filter_index + 1) % filter_count.max(1);
|
||
let completed_cycle = next_filter == 0;
|
||
let next_probe_in = if succeeded && completed_cycle {
|
||
mailbox_probe_refresh_interval()
|
||
} else if succeeded {
|
||
Duration::ZERO
|
||
} else {
|
||
mailbox_probe_retry_interval()
|
||
};
|
||
(next_filter, completed_cycle, next_probe_in)
|
||
}
|
||
|
||
async fn fetch_mailbox_filter(
|
||
connection: RelayConnection,
|
||
filter: Filter,
|
||
) -> Result<Vec<Event>, String> {
|
||
let mut pagination = PaginationState::new(vec![filter]);
|
||
let mut session = RelayPaginationSession::default();
|
||
let mut events = HashMap::<EventId, Event>::new();
|
||
|
||
loop {
|
||
let filter = pagination
|
||
.filters()
|
||
.into_iter()
|
||
.next()
|
||
.expect("mailbox pagination always owns one filter");
|
||
let page = connection
|
||
.fetch_events(filter, Duration::from_secs(30))
|
||
.await?;
|
||
let previous_count = events.len();
|
||
for event in page {
|
||
pagination.record_event(&event);
|
||
events.insert(event.id, event);
|
||
}
|
||
// Inclusive `until` cursors can repeat a relay's oldest timestamp.
|
||
// Stop when a page contributes nothing instead of retaining the only
|
||
// mailbox worker forever on a deterministic repeated page.
|
||
if events.len() == previous_count {
|
||
break;
|
||
}
|
||
let Some(next) = pagination.next_page(&mut session) else {
|
||
break;
|
||
};
|
||
pagination = next;
|
||
}
|
||
|
||
let mut events: Vec<_> = events.into_values().collect();
|
||
events.sort_by(|left, right| {
|
||
left.created_at
|
||
.cmp(&right.created_at)
|
||
.then_with(|| left.id.cmp(&right.id))
|
||
});
|
||
Ok(events)
|
||
}
|
||
|
||
fn should_use_semantic_fallback(requested_count: usize, received_count: usize) -> bool {
|
||
requested_count >= SEMANTIC_FALLBACK_MIN_REQUESTED_EVENTS
|
||
&& received_count.saturating_mul(100)
|
||
<= requested_count.saturating_mul(SEMANTIC_FALLBACK_MAX_DELIVERED_PERCENT)
|
||
}
|
||
|
||
fn purgatory_dependency_retry_after() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_secs(2)
|
||
} else {
|
||
Duration::from_secs(30)
|
||
}
|
||
}
|
||
|
||
fn dependency_relay_retention() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_secs(10)
|
||
} else {
|
||
Duration::from_secs(60)
|
||
}
|
||
}
|
||
|
||
fn effective_private_members(
|
||
configured: &HashSet<PublicKey>,
|
||
accepted_relays: &HashSet<String>,
|
||
relay_owners: &HashMap<String, PublicKey>,
|
||
) -> HashSet<PublicKey> {
|
||
configured
|
||
.iter()
|
||
.copied()
|
||
.chain(
|
||
relay_owners
|
||
.iter()
|
||
.filter_map(|(relay, owner)| accepted_relays.contains(relay).then_some(*owner)),
|
||
)
|
||
.collect()
|
||
}
|
||
|
||
fn byte_limited_catchup_interval() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_secs(2)
|
||
} else {
|
||
Duration::from_secs(5 * 60)
|
||
}
|
||
}
|
||
|
||
fn select_purgatory_dependency_events(
|
||
mut events: Vec<Event>,
|
||
attempts: &mut HashMap<EventId, Instant>,
|
||
now: Instant,
|
||
retry_after: Duration,
|
||
limit: usize,
|
||
) -> Vec<Event> {
|
||
let current_ids: HashSet<EventId> = events.iter().map(|event| event.id).collect();
|
||
attempts.retain(|event_id, _| current_ids.contains(event_id));
|
||
|
||
events.sort_by(|left, right| {
|
||
let left_is_new = !attempts.contains_key(&left.id);
|
||
let right_is_new = !attempts.contains_key(&right.id);
|
||
right_is_new
|
||
.cmp(&left_is_new)
|
||
.then_with(|| right.created_at.cmp(&left.created_at))
|
||
.then_with(|| right.id.cmp(&left.id))
|
||
});
|
||
|
||
let selected: Vec<Event> = events
|
||
.into_iter()
|
||
.filter(|event| {
|
||
attempts
|
||
.get(&event.id)
|
||
.is_none_or(|attempted_at| now.duration_since(*attempted_at) >= retry_after)
|
||
})
|
||
.take(limit)
|
||
.collect();
|
||
|
||
for event in &selected {
|
||
attempts.insert(event.id, now);
|
||
}
|
||
|
||
selected
|
||
}
|
||
|
||
#[derive(Default)]
|
||
struct DependencyRelayBatch {
|
||
event_ids: HashSet<EventId>,
|
||
identifiers: HashSet<String>,
|
||
}
|
||
|
||
/// Return one stable identity for a relay URL throughout all sync indexes.
|
||
///
|
||
/// URL parsers represent a root path as `/`, so announcements that alternate
|
||
/// between `wss://relay.example` and `wss://relay.example/` must not create
|
||
/// separate connections and duplicate subscription work. Non-root paths remain
|
||
/// distinct relay endpoints.
|
||
pub(crate) fn canonical_relay_key(relay_url: &str) -> Result<String, String> {
|
||
let normalized = if relay_url.starts_with("wss://") || relay_url.starts_with("ws://") {
|
||
relay_url.to_string()
|
||
} else {
|
||
format!("wss://{relay_url}")
|
||
};
|
||
let relay = RelayUrl::parse(&normalized).map_err(|error| error.to_string())?;
|
||
let mut canonical = relay.to_string();
|
||
let after_scheme = canonical
|
||
.split_once("://")
|
||
.map(|(_, value)| value)
|
||
.unwrap_or(canonical.as_str());
|
||
if after_scheme.ends_with('/') && after_scheme.matches('/').count() == 1 {
|
||
canonical.pop();
|
||
}
|
||
Ok(canonical)
|
||
}
|
||
|
||
fn is_own_sync_target(relay_url: &str, service_domain: &str) -> bool {
|
||
url_matches_service_domain(relay_url, service_domain)
|
||
}
|
||
|
||
#[cfg(test)]
|
||
fn connections_for_relay_urls<T: Clone>(
|
||
connections: &HashMap<String, T>,
|
||
relay_urls: &[String],
|
||
) -> Vec<(String, T)> {
|
||
let mut seen = HashSet::new();
|
||
relay_urls
|
||
.iter()
|
||
.filter_map(|relay_url| canonical_relay_key(relay_url).ok())
|
||
.filter(|relay_url| seen.insert(relay_url.clone()))
|
||
.filter_map(|relay_url| {
|
||
connections
|
||
.get(&relay_url)
|
||
.cloned()
|
||
.map(|connection| (relay_url, connection))
|
||
})
|
||
.collect()
|
||
}
|
||
|
||
// =============================================================================
|
||
// Type Aliases for Index Structures
|
||
// =============================================================================
|
||
|
||
/// What we WANT to sync - derived from events received via self-subscription.
|
||
/// Updated immediately when self-subscriber batch fires.
|
||
/// Key: repo addressable ref - 30617:pubkey:identifier
|
||
pub type RepoSyncIndex = Arc<RwLock<HashMap<String, RepoSyncNeeds>>>;
|
||
|
||
/// Compact metadata for roots observed by the self-subscription. Candidates
|
||
/// remain inert while their repository is StateOnly and become eligible for
|
||
/// NIP-65 discovery as soon as RepoSyncIndex promotes it to Full.
|
||
pub type RootCandidateIndex = Arc<RwLock<HashMap<EventId, discovery::AcceptedRoot>>>;
|
||
|
||
/// What we have CONFIRMED syncing - includes connection state for integrated lifecycle.
|
||
/// Key: relay URL
|
||
pub type RelaySyncIndex = Arc<RwLock<HashMap<String, RelayState>>>;
|
||
|
||
/// Tracks batches of subscriptions that are in-flight, awaiting EOSE.
|
||
/// Each batch has its own ID and can confirm independently.
|
||
/// Key: relay URL
|
||
pub type PendingSyncIndex = Arc<RwLock<HashMap<String, Vec<PendingBatch>>>>;
|
||
|
||
/// Tracks EventIds of announcement events (30617/30618) that were rejected during sync.
|
||
/// These events are excluded from negentropy sync and skipped during REQ+EOSE processing
|
||
/// to avoid repeatedly fetching and rejecting the same events.
|
||
///
|
||
/// Uses the two-tier RejectedEventsIndex from rejected_index.rs:
|
||
/// - Hot cache: Full events for 2 minutes (enables immediate re-processing)
|
||
/// - Cold index: Metadata for 7 days (prevents repeated downloads)
|
||
use rejected_index::RejectedEventsIndex;
|
||
|
||
// =============================================================================
|
||
// Supporting Data Structures
|
||
// =============================================================================
|
||
|
||
/// Level of sync needed for a repository
|
||
///
|
||
/// Purgatory announcements only need state events synced (to validate git data).
|
||
/// Promoted repos need full L2/L3 sync (patches, issues, PRs, etc.).
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||
pub enum SyncLevel {
|
||
/// Full L2 + L3 sync (promoted repos with git data)
|
||
#[default]
|
||
Full,
|
||
/// Only state events (kind 30618) - for purgatory announcements
|
||
StateOnly,
|
||
}
|
||
|
||
/// What repos and root events need to be synced
|
||
#[derive(Debug, Clone, Default)]
|
||
pub struct RepoSyncNeeds {
|
||
/// Relay URLs listed in this repo's 30617 announcement
|
||
pub relays: HashSet<String>,
|
||
/// Root event IDs - 1617/1618/1621 - that reference this repo
|
||
pub root_events: HashSet<EventId>,
|
||
/// Sync level - StateOnly for purgatory, Full for promoted repos
|
||
pub sync_level: SyncLevel,
|
||
}
|
||
|
||
/// Connection status for a relay
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
|
||
pub enum ConnectionStatus {
|
||
/// Not currently connected
|
||
#[default]
|
||
Disconnected,
|
||
/// Connection attempt in progress
|
||
Connecting,
|
||
/// Successfully connected, historic sync in progress
|
||
Syncing,
|
||
/// Successfully connected, historic sync completed
|
||
Connected,
|
||
/// Successfully connected, historic sync had failures but live sync active
|
||
ConnectedHistoricSyncFailures,
|
||
/// Disconnection initiated, waiting for event loop to terminate
|
||
/// State is retained to process remaining queued events
|
||
Disconnecting,
|
||
}
|
||
|
||
impl ConnectionStatus {
|
||
/// Returns true if live sync is active (can accept new filters)
|
||
pub fn is_live_sync_active(&self) -> bool {
|
||
matches!(
|
||
self,
|
||
ConnectionStatus::Syncing
|
||
| ConnectionStatus::Connected
|
||
| ConnectionStatus::ConnectedHistoricSyncFailures
|
||
)
|
||
}
|
||
}
|
||
|
||
/// Complete state for a single relay - combines sync needs with connection lifecycle
|
||
#[derive(Debug)]
|
||
pub struct RelayState {
|
||
/// Repos we have confirmed full L2/L3 syncing from this relay
|
||
pub repos: HashSet<String>,
|
||
/// Repos we have confirmed state-only syncing from this relay
|
||
pub state_only_repos: HashSet<String>,
|
||
/// Root events we have confirmed tracking
|
||
pub root_events: HashSet<EventId>,
|
||
/// If true, never disconnect this relay
|
||
pub is_bootstrap: bool,
|
||
/// Current connection status
|
||
pub connection_status: ConnectionStatus,
|
||
/// When we last successfully connected - used for since filter on reconnect
|
||
pub last_connected: Option<Timestamp>,
|
||
/// When we disconnected - for 15-minute state retention rule
|
||
pub disconnected_at: Option<Timestamp>,
|
||
/// Whether announcement filter historic sync has completed for this relay
|
||
/// Used to determine if we can use `since` filter on reconnect for Layer 1
|
||
pub announcements_synced: bool,
|
||
/// Whether initial historic sync has fully completed (all layers)
|
||
/// Used to transition from Syncing -> Connected status
|
||
pub historic_sync_completed: bool,
|
||
/// When historic sync completed (None if never completed or cleared on fresh_start)
|
||
pub historic_sync_completed_at: Option<Timestamp>,
|
||
/// Whether any batch failed during historic sync
|
||
/// Set to true when retry protection triggers or other failures occur
|
||
/// Used to transition to ConnectedDegraded instead of Connected
|
||
pub historic_sync_had_failures: bool,
|
||
}
|
||
|
||
impl Default for RelayState {
|
||
fn default() -> Self {
|
||
Self {
|
||
repos: HashSet::new(),
|
||
state_only_repos: HashSet::new(),
|
||
root_events: HashSet::new(),
|
||
is_bootstrap: false,
|
||
connection_status: ConnectionStatus::Disconnected,
|
||
last_connected: None,
|
||
disconnected_at: None,
|
||
announcements_synced: false,
|
||
historic_sync_completed: false,
|
||
historic_sync_completed_at: None,
|
||
historic_sync_had_failures: false,
|
||
}
|
||
}
|
||
}
|
||
|
||
impl RelayState {
|
||
/// Whether this relay has no remaining sync work and can be disconnected.
|
||
///
|
||
/// A newly connected relay has no *confirmed* repos until its historic
|
||
/// subscriptions complete. Treating that temporary state as empty races
|
||
/// the two-second disconnect checker against the historic sync (whose
|
||
/// completion is deliberately delayed to cover the subscriber batch
|
||
/// window). Pending batches and an active incomplete historic sync must
|
||
/// therefore keep the relay connected.
|
||
fn is_disconnect_candidate(&self, has_pending_batches: bool, has_desired_work: bool) -> bool {
|
||
if self.is_bootstrap
|
||
|| self.connection_status == ConnectionStatus::Connecting
|
||
|| self.connection_status == ConnectionStatus::Disconnecting
|
||
|| !self.repos.is_empty()
|
||
|| !self.state_only_repos.is_empty()
|
||
|| !self.root_events.is_empty()
|
||
|| has_pending_batches
|
||
|| has_desired_work
|
||
{
|
||
return false;
|
||
}
|
||
|
||
self.connection_status == ConnectionStatus::Disconnected || self.historic_sync_completed
|
||
}
|
||
|
||
/// Check if state should be cleared based on 15-minute rule
|
||
pub fn should_clear_state(&self) -> bool {
|
||
match self.disconnected_at {
|
||
Some(disconnected) => {
|
||
let now = Timestamp::now();
|
||
now.as_secs().saturating_sub(disconnected.as_secs()) > 900 // 15 minutes
|
||
}
|
||
None => false, // Still connected or never connected
|
||
}
|
||
}
|
||
|
||
/// Clear repos and root_events - called when reconnect takes > 15 minutes
|
||
pub fn clear_sync_state(&mut self) {
|
||
self.repos.clear();
|
||
self.state_only_repos.clear();
|
||
self.root_events.clear();
|
||
self.announcements_synced = false;
|
||
self.historic_sync_completed = false;
|
||
self.historic_sync_completed_at = None;
|
||
self.historic_sync_had_failures = false;
|
||
}
|
||
}
|
||
|
||
fn reconcile_purgatory_relay_ownership(
|
||
index: &mut HashMap<String, RepoSyncNeeds>,
|
||
announcements: &[(String, HashSet<String>)],
|
||
) -> HashSet<String> {
|
||
let active: HashMap<&str, &HashSet<String>> = announcements
|
||
.iter()
|
||
.map(|(repo_id, relays)| (repo_id.as_str(), relays))
|
||
.collect();
|
||
let mut dirty_relays = HashSet::new();
|
||
|
||
index.retain(|repo_id, needs| {
|
||
if needs.sync_level == SyncLevel::Full {
|
||
return true;
|
||
}
|
||
if active.contains_key(repo_id.as_str()) {
|
||
return true;
|
||
}
|
||
dirty_relays.extend(needs.relays.iter().cloned());
|
||
false
|
||
});
|
||
|
||
for (repo_id, relays) in announcements {
|
||
let entry = index
|
||
.entry(repo_id.clone())
|
||
.or_insert_with(|| RepoSyncNeeds {
|
||
relays: HashSet::new(),
|
||
root_events: HashSet::new(),
|
||
sync_level: SyncLevel::StateOnly,
|
||
});
|
||
if entry.sync_level == SyncLevel::StateOnly && entry.relays != *relays {
|
||
dirty_relays.extend(entry.relays.iter().cloned());
|
||
dirty_relays.extend(relays.iter().cloned());
|
||
entry.relays = relays.clone();
|
||
}
|
||
}
|
||
|
||
dirty_relays
|
||
}
|
||
|
||
/// Method used for synchronization
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
pub enum SyncMethod {
|
||
/// Traditional REQ+EOSE flow - waits for EOSE on subscriptions
|
||
ReqEose,
|
||
/// NIP-77 negentropy sync - confirms immediately after sync completes
|
||
Negentropy,
|
||
}
|
||
|
||
/// Result of processing an event from sync
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
pub enum ProcessResult {
|
||
/// Event was new and saved to database
|
||
Saved,
|
||
/// Event already existed in database
|
||
Duplicate,
|
||
/// Event added to Purgatory
|
||
Purgatory,
|
||
/// Event is absent by a valid persisted deletion or vanish request.
|
||
Tombstoned,
|
||
/// Event rejected by write policy.
|
||
Rejected(PolicyRejection),
|
||
/// The database could not be read or an accepted event could not be saved.
|
||
PersistenceError,
|
||
}
|
||
|
||
/// Bounded admission-rejection classes used by hydration accounting.
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
pub enum PolicyRejection {
|
||
Blocked,
|
||
Invalid,
|
||
Restricted,
|
||
Error,
|
||
Other,
|
||
PreviouslyRejected,
|
||
}
|
||
|
||
impl ProcessResult {
|
||
fn hydration_outcome(self) -> &'static str {
|
||
match self {
|
||
Self::Saved => "saved",
|
||
Self::Duplicate => "duplicate",
|
||
Self::Purgatory => "purgatory",
|
||
Self::Tombstoned => "tombstoned",
|
||
Self::Rejected(PolicyRejection::Blocked) => "rejected_blocked",
|
||
Self::Rejected(PolicyRejection::Invalid) => "rejected_invalid",
|
||
Self::Rejected(PolicyRejection::Restricted) => "rejected_restricted",
|
||
Self::Rejected(PolicyRejection::Error) => "rejected_error",
|
||
Self::Rejected(PolicyRejection::Other) => "rejected_other",
|
||
Self::Rejected(PolicyRejection::PreviouslyRejected) => "rejected_cached",
|
||
Self::PersistenceError => "persistence_error",
|
||
}
|
||
}
|
||
|
||
fn is_rejected(self) -> bool {
|
||
matches!(self, Self::Rejected(_))
|
||
}
|
||
|
||
/// Whether this result gives a durable explanation for the requested ID.
|
||
///
|
||
/// Restricted, server-error, and unknown policy failures can become valid
|
||
/// after dependencies arrive or a transient fault clears, so exact-ID
|
||
/// recovery keeps them pending. Invalid/blocked results are permanent for
|
||
/// the event, while purgatory and tombstones are explicit terminal states.
|
||
fn is_terminally_accounted(self) -> bool {
|
||
matches!(
|
||
self,
|
||
Self::Saved
|
||
| Self::Duplicate
|
||
| Self::Purgatory
|
||
| Self::Tombstoned
|
||
| Self::Rejected(PolicyRejection::Blocked)
|
||
| Self::Rejected(PolicyRejection::Invalid)
|
||
| Self::Rejected(PolicyRejection::PreviouslyRejected)
|
||
)
|
||
}
|
||
}
|
||
|
||
/// A low-volume summary of the per-relay data lane.
|
||
///
|
||
/// Sync bursts can contain tens of thousands of duplicates, so per-event logs
|
||
/// obscure whether time is spent waiting for the bounded channel or applying
|
||
/// policy. A periodic aggregate keeps production diagnosis cheap enough to
|
||
/// leave enabled while preserving both parts of that distinction.
|
||
#[derive(Debug)]
|
||
struct EventPipelineWindow {
|
||
started_at: std::time::Instant,
|
||
delivered: u64,
|
||
saved: u64,
|
||
duplicate: u64,
|
||
purgatory: u64,
|
||
tombstoned: u64,
|
||
rejected: u64,
|
||
persistence_error: u64,
|
||
queue_delay: std::time::Duration,
|
||
max_queue_delay: std::time::Duration,
|
||
processing_time: std::time::Duration,
|
||
max_processing_time: std::time::Duration,
|
||
}
|
||
|
||
impl Default for EventPipelineWindow {
|
||
fn default() -> Self {
|
||
Self {
|
||
started_at: std::time::Instant::now(),
|
||
delivered: 0,
|
||
saved: 0,
|
||
duplicate: 0,
|
||
purgatory: 0,
|
||
tombstoned: 0,
|
||
rejected: 0,
|
||
persistence_error: 0,
|
||
queue_delay: std::time::Duration::ZERO,
|
||
max_queue_delay: std::time::Duration::ZERO,
|
||
processing_time: std::time::Duration::ZERO,
|
||
max_processing_time: std::time::Duration::ZERO,
|
||
}
|
||
}
|
||
}
|
||
|
||
impl EventPipelineWindow {
|
||
const REPORT_INTERVAL: std::time::Duration = std::time::Duration::from_secs(30);
|
||
|
||
fn record(
|
||
&mut self,
|
||
result: ProcessResult,
|
||
queue_delay: std::time::Duration,
|
||
processing_time: std::time::Duration,
|
||
) {
|
||
self.delivered += 1;
|
||
match result {
|
||
ProcessResult::Saved => self.saved += 1,
|
||
ProcessResult::Duplicate => self.duplicate += 1,
|
||
ProcessResult::Purgatory => self.purgatory += 1,
|
||
ProcessResult::Tombstoned => self.tombstoned += 1,
|
||
ProcessResult::Rejected(_) => self.rejected += 1,
|
||
ProcessResult::PersistenceError => self.persistence_error += 1,
|
||
}
|
||
self.queue_delay += queue_delay;
|
||
self.max_queue_delay = self.max_queue_delay.max(queue_delay);
|
||
self.processing_time += processing_time;
|
||
self.max_processing_time = self.max_processing_time.max(processing_time);
|
||
}
|
||
|
||
fn report_if_due(&mut self, relay: &str, queue_depth: usize) {
|
||
let elapsed = self.started_at.elapsed();
|
||
if elapsed < Self::REPORT_INTERVAL || self.delivered == 0 {
|
||
return;
|
||
}
|
||
|
||
let delivered = self.delivered as f64;
|
||
tracing::info!(
|
||
relay,
|
||
window_seconds = elapsed.as_secs_f64(),
|
||
delivered = self.delivered,
|
||
saved = self.saved,
|
||
duplicate = self.duplicate,
|
||
purgatory = self.purgatory,
|
||
tombstoned = self.tombstoned,
|
||
rejected = self.rejected,
|
||
persistence_error = self.persistence_error,
|
||
events_per_second = delivered / elapsed.as_secs_f64(),
|
||
average_queue_delay_ms = self.queue_delay.as_secs_f64() * 1000.0 / delivered,
|
||
max_queue_delay_ms = self.max_queue_delay.as_secs_f64() * 1000.0,
|
||
average_processing_ms = self.processing_time.as_secs_f64() * 1000.0 / delivered,
|
||
max_processing_ms = self.max_processing_time.as_secs_f64() * 1000.0,
|
||
queue_depth,
|
||
queue_capacity = relay_connection::RELAY_EVENT_BUFFER_CAPACITY,
|
||
"Relay sync event-pipeline window"
|
||
);
|
||
*self = Self::default();
|
||
}
|
||
}
|
||
|
||
/// Statistics from re-processing events from hot cache
|
||
#[derive(Debug, Clone, Default)]
|
||
pub struct ReprocessingStats {
|
||
/// Number of events successfully saved
|
||
pub saved: usize,
|
||
/// Number of events that were duplicates
|
||
pub duplicate: usize,
|
||
/// Number of events added to purgatory
|
||
pub purgatory: usize,
|
||
/// Number of events still rejected
|
||
pub rejected: usize,
|
||
}
|
||
|
||
/// Pagination state for a subscription in non-Negentropy historic sync
|
||
#[derive(Debug, Clone)]
|
||
pub struct FilterPaginationState {
|
||
/// Number of events received for this filter
|
||
pub event_count: usize,
|
||
/// Smallest created_at timestamp seen for this filter
|
||
pub min_created_at: Option<Timestamp>,
|
||
/// Original filter to reconstruct for next page
|
||
pub original_filter: Filter,
|
||
/// IDs delivered on this page, retained only long enough to verify a NIP-11 hint.
|
||
page_event_ids: HashSet<EventId>,
|
||
/// IDs from the page which caused a one-page NIP-11 hint verification.
|
||
verification_baseline: Option<HashSet<EventId>>,
|
||
/// Whether the verification page delivered an ID absent from its triggering page.
|
||
verification_productive: bool,
|
||
}
|
||
|
||
/// Pagination state for every OR filter carried by one subscription.
|
||
#[derive(Debug, Clone)]
|
||
pub struct PaginationState {
|
||
pub filters: Vec<FilterPaginationState>,
|
||
}
|
||
|
||
impl PaginationState {
|
||
fn new(filters: Vec<Filter>) -> Self {
|
||
Self {
|
||
filters: filters
|
||
.into_iter()
|
||
.map(|original_filter| FilterPaginationState {
|
||
event_count: 0,
|
||
min_created_at: None,
|
||
original_filter,
|
||
page_event_ids: HashSet::new(),
|
||
verification_baseline: None,
|
||
verification_productive: false,
|
||
})
|
||
.collect(),
|
||
}
|
||
}
|
||
|
||
fn record_event(&mut self, event: &Event) {
|
||
for state in &mut self.filters {
|
||
if state
|
||
.original_filter
|
||
.match_event(event, MatchEventOptions::new())
|
||
{
|
||
state.event_count += 1;
|
||
state.page_event_ids.insert(event.id);
|
||
if state
|
||
.verification_baseline
|
||
.as_ref()
|
||
.is_some_and(|baseline| !baseline.contains(&event.id))
|
||
{
|
||
state.verification_productive = true;
|
||
}
|
||
match state.min_created_at {
|
||
None => state.min_created_at = Some(event.created_at),
|
||
Some(min) if event.created_at < min => {
|
||
state.min_created_at = Some(event.created_at);
|
||
}
|
||
_ => {}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
fn filters(&self) -> Vec<Filter> {
|
||
self.filters
|
||
.iter()
|
||
.map(|state| state.original_filter.clone())
|
||
.collect()
|
||
}
|
||
|
||
fn request_class(&self) -> TransientRequestClass {
|
||
if self
|
||
.filters
|
||
.iter()
|
||
.any(|state| state.verification_baseline.is_some())
|
||
{
|
||
TransientRequestClass::PaginationVerification
|
||
} else {
|
||
TransientRequestClass::PaginationPage
|
||
}
|
||
}
|
||
|
||
fn next_page(mut self, session: &mut RelayPaginationSession) -> Option<Self> {
|
||
let mut next = Vec::new();
|
||
for mut state in self.filters.drain(..) {
|
||
session.observe_page(state.event_count);
|
||
let completed_verification = state.verification_baseline.is_some();
|
||
|
||
let continue_page = if completed_verification {
|
||
session.complete_hint_verification(state.verification_productive);
|
||
state.verification_productive && state.event_count >= session.pagination_threshold()
|
||
} else if state.event_count >= session.pagination_threshold() {
|
||
true
|
||
} else if state.event_count >= PAGINATION_THRESHOLD_FLOOR
|
||
&& (session.begin_hint_verification() || session.hint_verification_in_progress())
|
||
{
|
||
state.verification_baseline = Some(std::mem::take(&mut state.page_event_ids));
|
||
true
|
||
} else {
|
||
false
|
||
};
|
||
if completed_verification {
|
||
state.verification_baseline = None;
|
||
}
|
||
|
||
let Some(min_created_at) = continue_page.then_some(state.min_created_at).flatten()
|
||
else {
|
||
continue;
|
||
};
|
||
state.original_filter = state
|
||
.original_filter
|
||
.until(Timestamp::from(min_created_at.as_secs()));
|
||
state.event_count = 0;
|
||
state.min_created_at = None;
|
||
state.page_event_ids.clear();
|
||
state.verification_productive = false;
|
||
next.push(state);
|
||
}
|
||
(!next.is_empty()).then_some(Self { filters: next })
|
||
}
|
||
}
|
||
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
enum PaginationHint {
|
||
Absent,
|
||
Unverified(usize),
|
||
Verifying(usize),
|
||
Verified(usize),
|
||
Discarded,
|
||
}
|
||
|
||
#[derive(Debug, Clone)]
|
||
struct RelayPaginationSession {
|
||
largest_raw_page: usize,
|
||
hint: PaginationHint,
|
||
}
|
||
|
||
impl Default for RelayPaginationSession {
|
||
fn default() -> Self {
|
||
Self::new(None)
|
||
}
|
||
}
|
||
|
||
impl RelayPaginationSession {
|
||
fn new(advertised_default_limit: Option<usize>) -> Self {
|
||
Self {
|
||
largest_raw_page: 0,
|
||
hint: advertised_default_limit
|
||
.map(PaginationHint::Unverified)
|
||
.unwrap_or(PaginationHint::Absent),
|
||
}
|
||
}
|
||
|
||
fn observe_page(&mut self, raw_count: usize) {
|
||
self.largest_raw_page = self.largest_raw_page.max(raw_count);
|
||
}
|
||
|
||
fn active_hint(&self) -> Option<usize> {
|
||
match self.hint {
|
||
PaginationHint::Unverified(limit) | PaginationHint::Verified(limit) => Some(limit),
|
||
PaginationHint::Absent | PaginationHint::Verifying(_) | PaginationHint::Discarded => {
|
||
None
|
||
}
|
||
}
|
||
}
|
||
|
||
fn estimated_cap(&self) -> usize {
|
||
self.largest_raw_page.max(self.active_hint().unwrap_or(0))
|
||
}
|
||
|
||
fn pagination_threshold(&self) -> usize {
|
||
// Ten percent slack absorbs relay-side shrinkage such as expired-event filtering.
|
||
// The floor remains below Ditto's observed 100-event omitted-limit page, the smallest
|
||
// live default found in the relay audit. These are interoperability constants, not
|
||
// operator policy, so deliberately do not enlarge the four-source config surface.
|
||
PAGINATION_THRESHOLD_FLOOR.max(
|
||
self.estimated_cap()
|
||
.saturating_mul(PAGINATION_THRESHOLD_PERCENT)
|
||
/ 100,
|
||
)
|
||
}
|
||
|
||
fn begin_hint_verification(&mut self) -> bool {
|
||
match self.hint {
|
||
PaginationHint::Unverified(limit) => {
|
||
self.hint = PaginationHint::Verifying(limit);
|
||
true
|
||
}
|
||
_ => false,
|
||
}
|
||
}
|
||
|
||
fn hint_verification_in_progress(&self) -> bool {
|
||
matches!(self.hint, PaginationHint::Verifying(_))
|
||
}
|
||
|
||
fn complete_hint_verification(&mut self, productive: bool) {
|
||
match (self.hint, productive) {
|
||
(PaginationHint::Verifying(_) | PaginationHint::Verified(_), true) => {
|
||
// Several filters can share the grouped verification page. Any one of them
|
||
// finding an unseen event disproves the relay-wide hint, even if an earlier
|
||
// exhausted filter provisionally marked it verified.
|
||
self.hint = PaginationHint::Discarded;
|
||
}
|
||
(PaginationHint::Verifying(limit), false) => {
|
||
self.hint = PaginationHint::Verified(limit);
|
||
}
|
||
_ => {}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// A batch of items pending confirmation
|
||
#[derive(Debug, Clone)]
|
||
pub struct PendingBatch {
|
||
/// Unique ID for this batch - for debugging/logging
|
||
pub batch_id: u64,
|
||
/// Why this batch exists. Auxiliary descendant discovery must not mutate
|
||
/// core historic-sync completion or health state.
|
||
pub purpose: PendingBatchPurpose,
|
||
/// The items this batch is syncing
|
||
pub items: PendingItems,
|
||
/// Subscription IDs that must ALL receive EOSE before confirming (for ReqEose)
|
||
/// Empty for Negentropy sync method
|
||
pub outstanding_subs: HashSet<SubscriptionId>,
|
||
/// The sync method used for this batch
|
||
pub sync_method: SyncMethod,
|
||
/// Pagination tracking for REQ+EOSE subscriptions (empty for Negentropy)
|
||
/// Maps subscription ID to its pagination state
|
||
pub pagination_state: HashMap<SubscriptionId, PaginationState>,
|
||
/// Event IDs requested via negentropy ID-based fetch (None for REQ+EOSE)
|
||
/// Used to validate that all requested events were received
|
||
pub requested_event_ids: Option<HashSet<EventId>>,
|
||
/// Event IDs actually received for this batch (None for REQ+EOSE)
|
||
/// Compared against requested_event_ids to detect missing events
|
||
pub received_event_ids: Option<HashSet<EventId>>,
|
||
/// First-pass advertised and delivered counts, retained while retrying a residual.
|
||
pub initial_hydration_counts: Option<(usize, usize)>,
|
||
/// Number of retry attempts for missing events (Negentropy only)
|
||
/// Used to prevent infinite retry loops when relay consistently fails
|
||
pub retry_count: usize,
|
||
/// Whether this batch failed (completed with missing data)
|
||
/// Set to true when retry protection triggers or other failures occur
|
||
pub failed: bool,
|
||
}
|
||
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
pub enum PendingBatchPurpose {
|
||
Core,
|
||
Announcements,
|
||
Descendants,
|
||
}
|
||
|
||
#[derive(Debug, Clone, Default, PartialEq, Eq)]
|
||
struct DescendantFrontier {
|
||
event_ids: HashSet<EventId>,
|
||
coordinates: HashSet<String>,
|
||
/// Accepted participant authors and the exact repository roots reached by
|
||
/// their retained direct or recursive events.
|
||
author_roots: HashMap<PublicKey, HashSet<EventId>>,
|
||
}
|
||
|
||
#[derive(Debug, Default)]
|
||
struct DescendantSyncRotation {
|
||
coverage_fingerprint: Vec<String>,
|
||
filters: Vec<DescendantFilterCursor>,
|
||
next_filter: usize,
|
||
in_flight: Option<DescendantFilterInFlight>,
|
||
}
|
||
|
||
#[derive(Debug)]
|
||
struct DescendantFilterCursor {
|
||
filters: Vec<Filter>,
|
||
last_successful_until: Option<Timestamp>,
|
||
}
|
||
|
||
#[derive(Debug, Clone, Copy)]
|
||
struct DescendantFilterInFlight {
|
||
batch_id: u64,
|
||
filter_index: usize,
|
||
until: Timestamp,
|
||
}
|
||
|
||
#[derive(Debug)]
|
||
struct DescendantLiveCoverage {
|
||
desired_filters: HashSet<String>,
|
||
fallback_filters: Vec<Filter>,
|
||
subscription_ids: Vec<SubscriptionId>,
|
||
}
|
||
|
||
fn filter_group_fingerprint(filters: &[Filter]) -> String {
|
||
filters
|
||
.iter()
|
||
.map(Filter::as_json)
|
||
.collect::<Vec<_>>()
|
||
.join("\n")
|
||
}
|
||
|
||
fn rotation_fingerprint(filters: &[Filter], max_filters_per_req: usize) -> Vec<String> {
|
||
group_filters_for_req_with_max(filters, max_filters_per_req)
|
||
.iter()
|
||
.map(|group| filter_group_fingerprint(group))
|
||
.collect()
|
||
}
|
||
|
||
impl DescendantSyncRotation {
|
||
fn refresh(&mut self, filters: Vec<Filter>, max_filters_per_req: usize) {
|
||
let previous: HashMap<String, Option<Timestamp>> = self
|
||
.filters
|
||
.drain(..)
|
||
.map(|cursor| {
|
||
(
|
||
filter_group_fingerprint(&cursor.filters),
|
||
cursor.last_successful_until,
|
||
)
|
||
})
|
||
.collect();
|
||
self.filters = group_filters_for_req_with_max(&filters, max_filters_per_req)
|
||
.into_iter()
|
||
.map(|filters| DescendantFilterCursor {
|
||
last_successful_until: previous
|
||
.get(&filter_group_fingerprint(&filters))
|
||
.copied()
|
||
.flatten(),
|
||
filters,
|
||
})
|
||
.collect();
|
||
self.coverage_fingerprint = self
|
||
.filters
|
||
.iter()
|
||
.map(|cursor| filter_group_fingerprint(&cursor.filters))
|
||
.collect();
|
||
self.next_filter = 0;
|
||
self.in_flight = None;
|
||
}
|
||
|
||
fn next_request(&self, now: Timestamp) -> Option<(usize, Vec<Filter>, Timestamp)> {
|
||
if self.in_flight.is_some() || self.filters.is_empty() {
|
||
return None;
|
||
}
|
||
let filter_index = self.next_filter % self.filters.len();
|
||
let cursor = &self.filters[filter_index];
|
||
let filters = cursor
|
||
.filters
|
||
.iter()
|
||
.cloned()
|
||
.map(|filter| {
|
||
let filter = filter.until(now);
|
||
match cursor.last_successful_until {
|
||
Some(last_until) => filter.since(Timestamp::from(
|
||
last_until
|
||
.as_secs()
|
||
.saturating_sub(DESCENDANT_FALLBACK_OVERLAP_SECS),
|
||
)),
|
||
None => filter,
|
||
}
|
||
})
|
||
.collect();
|
||
Some((filter_index, filters, now))
|
||
}
|
||
|
||
fn mark_started(&mut self, batch_id: u64, filter_index: usize, until: Timestamp) {
|
||
self.in_flight = Some(DescendantFilterInFlight {
|
||
batch_id,
|
||
filter_index,
|
||
until,
|
||
});
|
||
}
|
||
|
||
fn mark_completed(&mut self, batch_id: u64, succeeded: bool) -> bool {
|
||
let Some(in_flight) = self
|
||
.in_flight
|
||
.filter(|request| request.batch_id == batch_id)
|
||
else {
|
||
return false;
|
||
};
|
||
self.in_flight = None;
|
||
if succeeded {
|
||
self.filters[in_flight.filter_index].last_successful_until = Some(in_flight.until);
|
||
self.next_filter = (in_flight.filter_index + 1) % self.filters.len();
|
||
}
|
||
true
|
||
}
|
||
}
|
||
|
||
fn descendant_event_coordinate(event: &Event) -> Option<String> {
|
||
if !(event.kind.is_replaceable() || event.kind.is_addressable()) {
|
||
return None;
|
||
}
|
||
let identifier = if event.kind.is_addressable() {
|
||
event
|
||
.tags
|
||
.iter()
|
||
.find(|tag| tag.kind() == "d")
|
||
.and_then(|tag| tag.content())?
|
||
} else {
|
||
""
|
||
};
|
||
Some(format!(
|
||
"{}:{}:{}",
|
||
event.kind.as_u16(),
|
||
event.pubkey.to_hex(),
|
||
identifier
|
||
))
|
||
}
|
||
|
||
fn descendant_reference_roots(
|
||
event: &Event,
|
||
root_events: &HashSet<EventId>,
|
||
event_root_owners: &HashMap<EventId, HashSet<EventId>>,
|
||
coordinate_root_owners: &HashMap<String, HashSet<EventId>>,
|
||
) -> HashSet<EventId> {
|
||
let (coordinates, event_ids) =
|
||
crate::nostr::policy::RelatedEventPolicy::extract_reference_tags(event);
|
||
let mut roots: HashSet<EventId> = event_ids
|
||
.iter()
|
||
.filter(|event_id| root_events.contains(event_id))
|
||
.copied()
|
||
.collect();
|
||
for event_id in event_ids {
|
||
if let Some(owners) = event_root_owners.get(&event_id) {
|
||
roots.extend(owners.iter().copied());
|
||
}
|
||
}
|
||
for coordinate in coordinates {
|
||
if let Some(owners) = coordinate_root_owners.get(&coordinate) {
|
||
roots.extend(owners.iter().copied());
|
||
}
|
||
}
|
||
roots
|
||
}
|
||
|
||
/// Derive a bounded transitive frontier from related events already accepted
|
||
/// into the local database.
|
||
///
|
||
/// A newly fetched event becomes a parent on the next reconciliation tick, so
|
||
/// remote traversal progresses without retaining another durable cursor. Both
|
||
/// event IDs and replaceable/addressable coordinates participate because
|
||
/// clients may continue a thread through either reference form.
|
||
async fn recursive_descendant_frontier(
|
||
database: &SharedDatabase,
|
||
root_events: &HashSet<EventId>,
|
||
max_recursive_events_per_branch: usize,
|
||
) -> DescendantFrontier {
|
||
recursive_descendant_frontier_with_limits(
|
||
database,
|
||
root_events,
|
||
MAX_DESCENDANT_FRONTIER_DEPTH,
|
||
max_recursive_events_per_branch,
|
||
)
|
||
.await
|
||
}
|
||
|
||
async fn recursive_descendant_frontier_with_limits(
|
||
database: &SharedDatabase,
|
||
root_events: &HashSet<EventId>,
|
||
max_depth: usize,
|
||
max_recursive_events_per_branch: usize,
|
||
) -> DescendantFrontier {
|
||
let mut frontier = DescendantFrontier::default();
|
||
if max_depth == 0 || root_events.is_empty() {
|
||
return frontier;
|
||
}
|
||
|
||
// Every event directly tagging a root owns an independent recursive
|
||
// subtree. This keeps a popular tangential branch from consuming the
|
||
// compatibility coverage of unrelated issues, patches, or repositories.
|
||
// Retain exact root provenance at the same time so participant mailbox
|
||
// probes cannot leak one repository's roots into another author's scope.
|
||
let mut direct_events = HashMap::<EventId, (Event, HashSet<EventId>)>::new();
|
||
let mut event_root_owners = HashMap::<EventId, HashSet<EventId>>::new();
|
||
let mut coordinate_root_owners = HashMap::<String, HashSet<EventId>>::new();
|
||
for filter in filters::tagged_one_of_our_root_event_filters(root_events, None) {
|
||
match database.query(filter).await {
|
||
Ok(events) => {
|
||
for event in events
|
||
.iter()
|
||
.filter(|event| !root_events.contains(&event.id))
|
||
{
|
||
let roots = descendant_reference_roots(
|
||
event,
|
||
root_events,
|
||
&event_root_owners,
|
||
&coordinate_root_owners,
|
||
);
|
||
if roots.is_empty() {
|
||
continue;
|
||
}
|
||
direct_events
|
||
.entry(event.id)
|
||
.and_modify(|(_, known_roots)| {
|
||
known_roots.extend(roots.iter().copied());
|
||
})
|
||
.or_insert_with(|| (event.clone(), roots));
|
||
}
|
||
}
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
%error,
|
||
root_event_count = root_events.len(),
|
||
"Failed to derive direct repository thread members"
|
||
);
|
||
return DescendantFrontier::default();
|
||
}
|
||
}
|
||
}
|
||
|
||
let mut direct_events: Vec<_> = direct_events.into_iter().collect();
|
||
direct_events.sort_by(|(left_id, (left, _)), (right_id, (right, _))| {
|
||
left.created_at
|
||
.cmp(&right.created_at)
|
||
.then_with(|| left_id.cmp(right_id))
|
||
});
|
||
let direct_event_ids: HashSet<_> = direct_events.iter().map(|(id, _)| *id).collect();
|
||
let mut branch_member_counts = HashMap::<EventId, usize>::new();
|
||
let mut event_branches = HashMap::<EventId, HashSet<EventId>>::new();
|
||
let mut coordinate_branches = HashMap::<String, HashSet<EventId>>::new();
|
||
let mut event_seeds = HashMap::<EventId, HashSet<EventId>>::new();
|
||
let mut coordinate_seeds = HashMap::<String, HashSet<EventId>>::new();
|
||
for (event_id, (event, roots)) in direct_events {
|
||
branch_member_counts.insert(event_id, 0);
|
||
event_branches.entry(event_id).or_default().insert(event_id);
|
||
frontier
|
||
.author_roots
|
||
.entry(event.pubkey)
|
||
.or_default()
|
||
.extend(roots.iter().copied());
|
||
event_root_owners.insert(event_id, roots.clone());
|
||
event_seeds.entry(event_id).or_default().insert(event_id);
|
||
if let Some(coordinate) = descendant_event_coordinate(&event) {
|
||
coordinate_branches
|
||
.entry(coordinate.clone())
|
||
.or_default()
|
||
.insert(event_id);
|
||
coordinate_seeds
|
||
.entry(coordinate.clone())
|
||
.or_default()
|
||
.insert(event_id);
|
||
coordinate_root_owners
|
||
.entry(coordinate)
|
||
.or_default()
|
||
.extend(roots);
|
||
}
|
||
}
|
||
|
||
if max_depth == 1 || (event_seeds.is_empty() && coordinate_seeds.is_empty()) {
|
||
frontier.event_ids.extend(event_seeds.into_keys());
|
||
frontier.coordinates.extend(coordinate_seeds.into_keys());
|
||
return frontier;
|
||
}
|
||
|
||
for depth in 2..=max_depth {
|
||
let mut layer_events = HashMap::<EventId, (Event, HashSet<EventId>)>::new();
|
||
let event_seed_ids: HashSet<_> = event_seeds.keys().copied().collect();
|
||
let coordinate_seed_ids: HashSet<_> = coordinate_seeds.keys().cloned().collect();
|
||
let mut layer_filters =
|
||
filters::tagged_one_of_our_root_event_filters(&event_seed_ids, None);
|
||
layer_filters.extend(filters::tagged_one_of_our_repo_event_filters(
|
||
&coordinate_seed_ids,
|
||
None,
|
||
));
|
||
|
||
for filter in layer_filters {
|
||
match database.query(filter).await {
|
||
Ok(events) => {
|
||
for event in events.iter() {
|
||
if root_events.contains(&event.id) || direct_event_ids.contains(&event.id) {
|
||
continue;
|
||
}
|
||
let roots = descendant_reference_roots(
|
||
event,
|
||
root_events,
|
||
&event_root_owners,
|
||
&coordinate_root_owners,
|
||
);
|
||
if roots.is_empty() {
|
||
continue;
|
||
}
|
||
layer_events
|
||
.entry(event.id)
|
||
.and_modify(|(_, known_roots)| {
|
||
known_roots.extend(roots.iter().copied());
|
||
})
|
||
.or_insert_with(|| (event.clone(), roots));
|
||
}
|
||
}
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
%error,
|
||
depth,
|
||
event_seed_count = event_seeds.len(),
|
||
coordinate_seed_count = coordinate_seeds.len(),
|
||
direct_event_count = direct_event_ids.len(),
|
||
"Failed to derive recursive repository thread members"
|
||
);
|
||
return frontier;
|
||
}
|
||
}
|
||
}
|
||
|
||
let mut layer_events: Vec<_> = layer_events.into_iter().collect();
|
||
layer_events.sort_by(|(left_id, (left, _)), (right_id, (right, _))| {
|
||
left.created_at
|
||
.cmp(&right.created_at)
|
||
.then_with(|| left_id.cmp(right_id))
|
||
});
|
||
|
||
let mut next_event_seeds = HashMap::<EventId, HashSet<EventId>>::new();
|
||
let mut next_coordinate_seeds = HashMap::<String, HashSet<EventId>>::new();
|
||
for (event_id, (event, roots)) in layer_events {
|
||
let (addressable_refs, event_refs) =
|
||
crate::nostr::policy::RelatedEventPolicy::extract_reference_tags(&event);
|
||
let mut inherited_branches = HashSet::new();
|
||
for parent in event_refs {
|
||
if let Some(branches) = event_branches.get(&parent) {
|
||
inherited_branches.extend(branches.iter().copied());
|
||
}
|
||
}
|
||
for parent in addressable_refs {
|
||
if let Some(branches) = coordinate_branches.get(&parent) {
|
||
inherited_branches.extend(branches.iter().copied());
|
||
}
|
||
}
|
||
if let Some(existing_branches) = event_branches.get(&event_id) {
|
||
inherited_branches.retain(|branch| !existing_branches.contains(branch));
|
||
}
|
||
|
||
let mut admitted_branches = HashSet::new();
|
||
for branch in inherited_branches {
|
||
let member_count = branch_member_counts.entry(branch).or_default();
|
||
if *member_count < max_recursive_events_per_branch {
|
||
*member_count += 1;
|
||
admitted_branches.insert(branch);
|
||
}
|
||
}
|
||
if admitted_branches.is_empty() {
|
||
continue;
|
||
}
|
||
frontier
|
||
.author_roots
|
||
.entry(event.pubkey)
|
||
.or_default()
|
||
.extend(roots.iter().copied());
|
||
event_root_owners
|
||
.entry(event_id)
|
||
.or_default()
|
||
.extend(roots.iter().copied());
|
||
event_branches
|
||
.entry(event_id)
|
||
.or_default()
|
||
.extend(admitted_branches.iter().copied());
|
||
let coordinate = descendant_event_coordinate(&event);
|
||
if let Some(coordinate) = &coordinate {
|
||
coordinate_branches
|
||
.entry(coordinate.clone())
|
||
.or_default()
|
||
.extend(admitted_branches.iter().copied());
|
||
coordinate_root_owners
|
||
.entry(coordinate.clone())
|
||
.or_default()
|
||
.extend(roots.iter().copied());
|
||
}
|
||
|
||
let expandable_branches: HashSet<_> = admitted_branches
|
||
.into_iter()
|
||
.filter(|branch| branch_member_counts[branch] < max_recursive_events_per_branch)
|
||
.collect();
|
||
if expandable_branches.is_empty() {
|
||
continue;
|
||
}
|
||
next_event_seeds.insert(event_id, expandable_branches.clone());
|
||
if let Some(coordinate) = coordinate {
|
||
next_coordinate_seeds.insert(coordinate, expandable_branches);
|
||
}
|
||
}
|
||
|
||
if next_event_seeds.is_empty() && next_coordinate_seeds.is_empty() {
|
||
break;
|
||
}
|
||
event_seeds = next_event_seeds;
|
||
coordinate_seeds = next_coordinate_seeds;
|
||
}
|
||
|
||
let exhausted_branches: HashSet<_> = branch_member_counts
|
||
.iter()
|
||
.filter(|(_, member_count)| **member_count >= max_recursive_events_per_branch)
|
||
.map(|(branch, _)| *branch)
|
||
.collect();
|
||
for (event_id, branches) in &event_branches {
|
||
if branches
|
||
.iter()
|
||
.any(|branch| !exhausted_branches.contains(branch))
|
||
{
|
||
frontier.event_ids.insert(*event_id);
|
||
}
|
||
}
|
||
for (coordinate, branches) in &coordinate_branches {
|
||
if branches
|
||
.iter()
|
||
.any(|branch| !exhausted_branches.contains(branch))
|
||
{
|
||
frontier.coordinates.insert(coordinate.clone());
|
||
}
|
||
}
|
||
|
||
tracing::debug!(
|
||
depth_limit = max_depth,
|
||
event_count = frontier.event_ids.len(),
|
||
coordinate_count = frontier.coordinates.len(),
|
||
branch_count = branch_member_counts.len(),
|
||
exhausted_branch_count = exhausted_branches.len(),
|
||
"Derived bounded recursive repository thread frontier"
|
||
);
|
||
frontier
|
||
}
|
||
|
||
#[cfg(test)]
|
||
fn descendant_frontier_filters(
|
||
frontier: &DescendantFrontier,
|
||
since: Option<Timestamp>,
|
||
) -> Vec<Filter> {
|
||
let mut filters = filters::tagged_one_of_our_root_event_filters(&frontier.event_ids, since);
|
||
filters.extend(filters::tagged_one_of_our_repo_event_filters(
|
||
&frontier.coordinates,
|
||
since,
|
||
));
|
||
filters
|
||
}
|
||
|
||
fn mailbox_probe_filters(
|
||
repositories: &HashSet<String>,
|
||
root_events: &HashSet<EventId>,
|
||
frontier: &DescendantFrontier,
|
||
) -> Vec<Filter> {
|
||
let coordinate_values: HashSet<_> =
|
||
repositories.union(&frontier.coordinates).cloned().collect();
|
||
let event_values: HashSet<_> = root_events.union(&frontier.event_ids).copied().collect();
|
||
let mut filters = filters::tagged_one_of_our_repo_event_filters(&coordinate_values, None);
|
||
filters.extend(filters::tagged_one_of_our_root_event_filters(
|
||
&event_values,
|
||
None,
|
||
));
|
||
filters
|
||
}
|
||
|
||
fn packed_historic_filters(items: &PendingItems, frontier: &DescendantFrontier) -> Vec<Filter> {
|
||
let all_repos: HashSet<_> = items
|
||
.repos
|
||
.union(&items.state_only_repos)
|
||
.cloned()
|
||
.collect();
|
||
let mut filters = filters::state_event_filters_for_our_repos(&all_repos, None);
|
||
|
||
let coordinate_values: HashSet<_> = items.repos.union(&frontier.coordinates).cloned().collect();
|
||
filters.extend(filters::tagged_one_of_our_repo_event_filters(
|
||
&coordinate_values,
|
||
None,
|
||
));
|
||
|
||
let event_values: HashSet<_> = items
|
||
.root_events
|
||
.union(&frontier.event_ids)
|
||
.cloned()
|
||
.collect();
|
||
filters.extend(filters::tagged_one_of_our_root_event_filters(
|
||
&event_values,
|
||
None,
|
||
));
|
||
filters
|
||
}
|
||
|
||
fn tiered_auxiliary_filters(
|
||
repos: &HashSet<String>,
|
||
root_events: &HashSet<EventId>,
|
||
frontier: &DescendantFrontier,
|
||
since: Option<Timestamp>,
|
||
) -> Vec<filters::TieredFilter> {
|
||
let mut entries = filters::tiered_auxiliary_core_filters(repos, root_events, since);
|
||
entries.extend(filters::tiered_root_event_filters(
|
||
&frontier.event_ids,
|
||
true,
|
||
since,
|
||
));
|
||
entries.extend(filters::tiered_repo_event_filters(
|
||
&frontier.coordinates,
|
||
true,
|
||
since,
|
||
));
|
||
entries.sort_by_key(|entry| entry.tier);
|
||
entries
|
||
}
|
||
|
||
fn split_live_tier_prefix(
|
||
entries: &[filters::TieredFilter],
|
||
max_filters_per_req: usize,
|
||
fits: impl Fn(&[Vec<Filter>]) -> bool,
|
||
) -> (Vec<Filter>, Vec<Filter>) {
|
||
use filters::CoverageTier;
|
||
|
||
let live_tiers = [
|
||
CoverageTier::RootUppercase,
|
||
CoverageTier::CoreCompatibility,
|
||
CoverageTier::DescendantCanonical,
|
||
CoverageTier::DescendantQuote,
|
||
];
|
||
let mut live = Vec::new();
|
||
for tier in live_tiers {
|
||
let mut candidate = live.clone();
|
||
candidate.extend(
|
||
entries
|
||
.iter()
|
||
.filter(|entry| entry.tier == tier)
|
||
.map(|entry| entry.filter.clone()),
|
||
);
|
||
let groups = live_filter_groups(&candidate, max_filters_per_req);
|
||
if fits(&groups) {
|
||
live = candidate;
|
||
} else {
|
||
break;
|
||
}
|
||
}
|
||
let live_json: HashSet<_> = live.iter().map(Filter::as_json).collect();
|
||
let rotated = entries
|
||
.iter()
|
||
.filter(|entry| !live_json.contains(&entry.filter.as_json()))
|
||
.map(|entry| entry.filter.clone())
|
||
.collect();
|
||
(live, rotated)
|
||
}
|
||
|
||
fn complete_auxiliary_fallback(entries: &[filters::TieredFilter]) -> Vec<Filter> {
|
||
entries.iter().map(|entry| entry.filter.clone()).collect()
|
||
}
|
||
|
||
/// Items included in a pending batch
|
||
#[derive(Debug, Clone, Default)]
|
||
pub struct PendingItems {
|
||
/// Repos receiving full L2/L3 sync in this batch
|
||
pub repos: HashSet<String>,
|
||
/// Repos receiving state-only sync in this batch
|
||
pub state_only_repos: HashSet<String>,
|
||
/// Root events being synced in this batch
|
||
pub root_events: HashSet<EventId>,
|
||
}
|
||
|
||
fn take_drained_batch_as_failed(
|
||
pending: &mut HashMap<String, Vec<PendingBatch>>,
|
||
relay_url: &str,
|
||
batch_id: u64,
|
||
) -> Option<PendingBatch> {
|
||
let (batch_index, remove_relay) = {
|
||
let batches = pending.get(relay_url)?;
|
||
let batch_index = batches
|
||
.iter()
|
||
.position(|batch| batch.batch_id == batch_id)?;
|
||
if !batches[batch_index].outstanding_subs.is_empty() {
|
||
return None;
|
||
}
|
||
(batch_index, batches.len() == 1)
|
||
};
|
||
|
||
let mut batch = pending.get_mut(relay_url)?.remove(batch_index);
|
||
batch.failed = true;
|
||
if remove_relay {
|
||
pending.remove(relay_url);
|
||
}
|
||
Some(batch)
|
||
}
|
||
|
||
fn take_batch_containing_subscription(
|
||
pending: &mut HashMap<String, Vec<PendingBatch>>,
|
||
relay_url: &str,
|
||
subscription_id: &SubscriptionId,
|
||
) -> Option<PendingBatch> {
|
||
let (batch_index, remove_relay) = {
|
||
let batches = pending.get(relay_url)?;
|
||
let batch_index = batches
|
||
.iter()
|
||
.position(|batch| batch.outstanding_subs.contains(subscription_id))?;
|
||
(batch_index, batches.len() == 1)
|
||
};
|
||
|
||
let batch = pending.get_mut(relay_url)?.remove(batch_index);
|
||
if remove_relay {
|
||
pending.remove(relay_url);
|
||
}
|
||
Some(batch)
|
||
}
|
||
|
||
fn register_negentropy_hydration_attempt(
|
||
batch: &mut PendingBatch,
|
||
subscription_ids: impl IntoIterator<Item = SubscriptionId>,
|
||
requested_event_ids: impl IntoIterator<Item = EventId>,
|
||
is_retry: bool,
|
||
) {
|
||
// Register the complete attempt before its first REQ is sent. A fast
|
||
// relay can deliver events and EOSE while later paced chunks are still
|
||
// waiting to open; pre-registration makes that whole stream attributable.
|
||
batch.outstanding_subs.extend(subscription_ids);
|
||
batch.requested_event_ids = Some(requested_event_ids.into_iter().collect());
|
||
batch.received_event_ids = Some(HashSet::new());
|
||
if is_retry {
|
||
batch.retry_count += 1;
|
||
}
|
||
}
|
||
|
||
fn mark_deferred_pagination(
|
||
batch: &mut PendingBatch,
|
||
completed_sub_id: &SubscriptionId,
|
||
) -> SubscriptionId {
|
||
let deferred_sub_id = SubscriptionId::new(format!(
|
||
"deferred-pagination-{}-{completed_sub_id}",
|
||
batch.batch_id
|
||
));
|
||
batch.outstanding_subs.insert(deferred_sub_id.clone());
|
||
deferred_sub_id
|
||
}
|
||
|
||
// =============================================================================
|
||
// SyncManager - Main Entry Point
|
||
// =============================================================================
|
||
|
||
/// Notification from spawned tasks about relay disconnections
|
||
#[derive(Debug)]
|
||
pub struct DisconnectNotification {
|
||
/// The relay URL that disconnected
|
||
pub relay_url: String,
|
||
}
|
||
|
||
/// Notification from spawned tasks about EOSE (End Of Stored Events)
|
||
#[derive(Debug)]
|
||
pub struct EoseNotification {
|
||
/// The relay URL that sent EOSE
|
||
pub relay_url: String,
|
||
/// The subscription ID that completed
|
||
pub sub_id: SubscriptionId,
|
||
}
|
||
|
||
#[derive(Debug)]
|
||
struct SubscriptionClosedNotification {
|
||
relay_url: String,
|
||
subscription_id: SubscriptionId,
|
||
reason: String,
|
||
generation: Option<u64>,
|
||
live_filter_count: Option<usize>,
|
||
}
|
||
|
||
// One global actor consumes terminal state changes from every outbound relay.
|
||
// A hostile peer can repeat EOSE/CLOSED indefinitely, so this queue must bound
|
||
// retained memory independently of the per-connection subscription ledger.
|
||
const LIFECYCLE_NOTIFICATION_CAPACITY: usize = 1_000;
|
||
|
||
fn lifecycle_notification_channel<T>(
|
||
) -> (tokio::sync::mpsc::Sender<T>, tokio::sync::mpsc::Receiver<T>) {
|
||
tokio::sync::mpsc::channel(LIFECYCLE_NOTIFICATION_CAPACITY)
|
||
}
|
||
|
||
fn is_rate_limit_message(message: &str) -> bool {
|
||
let message = message.to_lowercase();
|
||
(message.contains("rate") && message.contains("limit"))
|
||
|| message.contains("too many")
|
||
|| message.contains("slow down")
|
||
|| message.contains("throttl")
|
||
}
|
||
|
||
fn is_filter_count_refusal(message: &str) -> bool {
|
||
let message = message.to_ascii_lowercase();
|
||
message.contains("invalid number of filters") || message.contains("max filter count")
|
||
}
|
||
|
||
fn is_auth_required_message(message: &str) -> bool {
|
||
let message = message.to_ascii_lowercase();
|
||
message.contains("auth-required") || message.contains("authentication required")
|
||
}
|
||
|
||
fn reserve_authentication_retry(
|
||
attempts: &mut HashSet<(String, SubscriptionId)>,
|
||
relay_url: &str,
|
||
subscription_id: &SubscriptionId,
|
||
) -> bool {
|
||
let attempt = (relay_url.to_string(), subscription_id.clone());
|
||
if attempts.insert(attempt.clone()) {
|
||
true
|
||
} else {
|
||
attempts.remove(&attempt);
|
||
false
|
||
}
|
||
}
|
||
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
enum PolicyRefusal {
|
||
AuthenticationRequired,
|
||
Blocked,
|
||
Restricted,
|
||
MembershipRequired,
|
||
FilterIncompatible,
|
||
}
|
||
|
||
impl PolicyRefusal {
|
||
fn label(self) -> &'static str {
|
||
match self {
|
||
Self::AuthenticationRequired => "authentication_required",
|
||
Self::Blocked => "blocked",
|
||
Self::Restricted => "restricted",
|
||
Self::MembershipRequired => "membership_required",
|
||
Self::FilterIncompatible => "filter_incompatible",
|
||
}
|
||
}
|
||
}
|
||
|
||
fn policy_refusal(message: &str) -> Option<PolicyRefusal> {
|
||
if is_filter_count_refusal(message) {
|
||
return Some(PolicyRefusal::FilterIncompatible);
|
||
}
|
||
if is_rate_limit_message(message) {
|
||
return None;
|
||
}
|
||
let message = message.to_ascii_lowercase();
|
||
if message.contains("unsupported filter") || message.contains("filter validation failed") {
|
||
Some(PolicyRefusal::FilterIncompatible)
|
||
} else if message.contains("not a member") || message.contains("membership") {
|
||
Some(PolicyRefusal::MembershipRequired)
|
||
} else if message.contains("restricted") {
|
||
Some(PolicyRefusal::Restricted)
|
||
} else if message.contains("blocked") || message.contains("request rejected") {
|
||
Some(PolicyRefusal::Blocked)
|
||
} else {
|
||
None
|
||
}
|
||
}
|
||
|
||
fn subscription_state_byte_limit(message: &str) -> Option<usize> {
|
||
let lower = message.to_ascii_lowercase();
|
||
let marker = "active subscriptions exceed max size ";
|
||
let tail = lower.split_once(marker)?.1;
|
||
let digits = tail.split_whitespace().next()?;
|
||
digits.parse().ok()
|
||
}
|
||
|
||
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
|
||
struct ConnectAttemptToken(u64);
|
||
|
||
#[derive(Debug)]
|
||
enum ConnectAttemptOutcome {
|
||
Connected {
|
||
advertised_default_limit: Option<usize>,
|
||
advertised_max_subscriptions: Option<usize>,
|
||
advertised_owner: Option<PublicKey>,
|
||
advertised_grasp08: bool,
|
||
},
|
||
/// The relay's NIP-11 advertises the GRASP-08 private-service extension
|
||
/// while this instance is public. Detected before the WebSocket dial, so
|
||
/// no connection or AUTH exchange ever happened.
|
||
PrivateService,
|
||
Failed(String),
|
||
}
|
||
|
||
#[derive(Debug)]
|
||
struct ConnectAttemptResult {
|
||
relay_url: String,
|
||
token: ConnectAttemptToken,
|
||
outcome: ConnectAttemptOutcome,
|
||
}
|
||
|
||
const NIP65_DISCOVERY_BATCH_AUTHORS: usize = 100;
|
||
|
||
fn nip65_discovery_refresh_interval() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_secs(2)
|
||
} else {
|
||
Duration::from_secs(24 * 60 * 60)
|
||
}
|
||
}
|
||
|
||
fn nip65_discovery_retry_interval() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_millis(500)
|
||
} else {
|
||
Duration::from_secs(5 * 60)
|
||
}
|
||
}
|
||
|
||
fn nip65_inventory_interval() -> Duration {
|
||
if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_millis(200)
|
||
} else {
|
||
Duration::from_secs(60)
|
||
}
|
||
}
|
||
|
||
fn nip65_author_retry_after(
|
||
query_succeeded: bool,
|
||
represented: &HashSet<PublicKey>,
|
||
author: &PublicKey,
|
||
) -> Duration {
|
||
if query_succeeded && represented.contains(author) {
|
||
nip65_discovery_refresh_interval()
|
||
} else {
|
||
nip65_discovery_retry_interval()
|
||
}
|
||
}
|
||
|
||
#[derive(Debug)]
|
||
struct Nip65DiscoveryResult {
|
||
source_relay: String,
|
||
authors: HashSet<PublicKey>,
|
||
outcome: Result<Vec<Event>, String>,
|
||
}
|
||
|
||
#[derive(Debug)]
|
||
struct MailboxProbeResult {
|
||
source_relay: String,
|
||
filter_index: usize,
|
||
filter_count: usize,
|
||
outcome: Result<Vec<Event>, String>,
|
||
}
|
||
|
||
#[derive(Debug, Default)]
|
||
struct Nip65DiscoveryState {
|
||
/// Roots authored by each accepted root author. These preserve the
|
||
/// existing GRASP-03 live inbox overlay; non-root participants remain on
|
||
/// the paced history-only mailbox path below.
|
||
root_author_roots: HashMap<PublicKey, HashSet<EventId>>,
|
||
author_roots: HashMap<PublicKey, HashSet<EventId>>,
|
||
root_repositories: HashMap<EventId, String>,
|
||
eligible_authors: HashSet<PublicKey>,
|
||
author_sources: HashMap<PublicKey, HashSet<String>>,
|
||
relay_lists: HashMap<PublicKey, Event>,
|
||
author_inboxes: HashMap<PublicKey, HashSet<String>>,
|
||
author_mailboxes: HashMap<PublicKey, HashSet<String>>,
|
||
/// Eligible authors for whom at least one successful index query returned
|
||
/// no accepted relay list. They use the operator's bounded fallback set
|
||
/// until an accepted kind 10002 arrives.
|
||
fallback_authors: HashSet<PublicKey>,
|
||
inbox_roots: HashMap<String, HashSet<EventId>>,
|
||
mailbox_roots: HashMap<String, HashSet<EventId>>,
|
||
mailbox_probe_next_at: HashMap<String, Instant>,
|
||
mailbox_probe_next_filter: HashMap<String, usize>,
|
||
/// At most one ordinary fetch runs per mailbox relay. Different relays
|
||
/// remain independent and use their own connection capacity.
|
||
mailbox_probes_in_flight: HashSet<String>,
|
||
in_flight: HashSet<(String, PublicKey)>,
|
||
next_attempt_at: HashMap<(String, PublicKey), Instant>,
|
||
next_inventory_at: Option<Instant>,
|
||
}
|
||
|
||
impl Nip65DiscoveryState {
|
||
fn install_mailbox_overlay(
|
||
&mut self,
|
||
new_overlay: HashMap<String, HashSet<EventId>>,
|
||
now: Instant,
|
||
) {
|
||
let old_overlay = std::mem::replace(&mut self.mailbox_roots, new_overlay);
|
||
if old_overlay == self.mailbox_roots {
|
||
return;
|
||
}
|
||
self.mailbox_probe_next_at
|
||
.retain(|relay, _| self.mailbox_roots.contains_key(relay));
|
||
self.mailbox_probe_next_filter
|
||
.retain(|relay, _| self.mailbox_roots.contains_key(relay));
|
||
for (relay, roots) in &self.mailbox_roots {
|
||
if old_overlay.get(relay) != Some(roots) {
|
||
self.mailbox_probe_next_at.insert(relay.clone(), now);
|
||
self.mailbox_probe_next_filter
|
||
.entry(relay.clone())
|
||
.or_default();
|
||
}
|
||
}
|
||
}
|
||
|
||
fn record_probe_completion(
|
||
&mut self,
|
||
relay: &str,
|
||
next_filter: usize,
|
||
next_probe_in: Duration,
|
||
now: Instant,
|
||
) {
|
||
// A relay removed from the overlay while its probe was in flight must
|
||
// not leave orphan cursor or due-time entries; re-addition is reseeded
|
||
// by `install_mailbox_overlay`.
|
||
if !self.mailbox_roots.contains_key(relay) {
|
||
return;
|
||
}
|
||
self.mailbox_probe_next_filter
|
||
.insert(relay.to_string(), next_filter);
|
||
self.mailbox_probe_next_at
|
||
.insert(relay.to_string(), now + next_probe_in);
|
||
}
|
||
}
|
||
|
||
/// Quick reconnect window in seconds (15 minutes)
|
||
const QUICK_RECONNECT_WINDOW_SECS: u64 = 15 * 60;
|
||
|
||
/// Bound concurrent DNS and websocket handshakes so a large relay list cannot
|
||
/// exhaust network resources while keeping the sync actor responsive.
|
||
const MAX_CONCURRENT_CONNECT_ATTEMPTS: usize = 8;
|
||
|
||
/// Maximum number of collection members rendered in an INFO-level log field.
|
||
/// Counts remain authoritative; samples keep remotely influenced logs bounded.
|
||
const LOG_COLLECTION_SAMPLE_SIZE: usize = 5;
|
||
|
||
/// Adaptive per-relay threshold for historic REQ+EOSE pagination.
|
||
///
|
||
/// NIP-01 guarantees none of this; the model is empirical. A source audit of
|
||
/// nine relay implementations (2026-08-06, recorded with citations in
|
||
/// docs/explanation/sync-scaling-constraints.md) found result limits are
|
||
/// always applied per filter, never in aggregate across a REQ. Ditto is the
|
||
/// smallest live omitted-limit default found, at 100 events.
|
||
///
|
||
/// Every raw delivery matching a tracked filter counts, before write-policy
|
||
/// processing. Purgatory-routed, rejected, and repeated events therefore
|
||
/// consume both the relay's allowance and our page count, and the cursor is
|
||
/// derived from that same raw stream. Each connection session learns its
|
||
/// largest raw page and combines it with NIP-11 `default_limit` when present:
|
||
/// `threshold = max(90, floor(0.9 * max(observed, default_limit)))`. A hint is
|
||
/// verified once before it may stop pagination; a productive verification
|
||
/// page discards it for that session. `max_limit` is intentionally ignored
|
||
/// because these filters omit `limit`. State and NIP-11 data reset on every
|
||
/// reconnect. See "Per-query result limits and the pagination model" in
|
||
/// docs/explanation/sync-scaling-constraints.md.
|
||
const PAGINATION_THRESHOLD_FLOOR: usize = 90;
|
||
const PAGINATION_THRESHOLD_PERCENT: usize = 90;
|
||
|
||
/// Conservative number of OR filters carried by one NIP-01 REQ.
|
||
///
|
||
/// This keeps active subscription counts low without producing unusually
|
||
/// large REQ messages for relays that enforce their own per-REQ filter limits.
|
||
const MAX_FILTERS_PER_REQ: usize = 10;
|
||
|
||
/// Serialized byte budget for the filters carried by one NIP-01 REQ.
|
||
///
|
||
/// The whole REQ message must fit relay maximum message sizes (128 KB floor
|
||
/// observed via NIP-11). 96 KB of filters leaves 1.3× margin for the
|
||
/// envelope and holds three full byte-budgeted filter chunks (see
|
||
/// [`filters::FILTER_VALUE_BYTE_BUDGET`]), so grouped REQs also stay within
|
||
/// strfry's optional strict `filterValidation` limit of three filters per
|
||
/// REQ when chunks are full. See
|
||
/// docs/explanation/sync-scaling-constraints.md.
|
||
const REQ_MESSAGE_BYTE_BUDGET: usize = 96 * 1024;
|
||
/// Leave room for one maximum-sized transient REQ when a relay discloses a
|
||
/// cumulative retained-subscription byte cap. Byte-limited sessions serialize
|
||
/// transient REQs through a matching connection gate.
|
||
const SUBSCRIPTION_BYTE_RESERVED_MARGIN: usize = REQ_MESSAGE_BYTE_BUDGET + 256;
|
||
|
||
fn req_message_size(filters: &[Filter]) -> usize {
|
||
ClientMessage::req(SubscriptionId::generate(), filters.to_vec())
|
||
.as_json()
|
||
.len()
|
||
}
|
||
|
||
fn groups_within_subscription_byte_limit(
|
||
groups: Vec<Vec<Filter>>,
|
||
limit: Option<usize>,
|
||
already_used: usize,
|
||
) -> (Vec<Vec<Filter>>, usize) {
|
||
let Some(limit) = limit else {
|
||
return (groups, 0);
|
||
};
|
||
let live_budget = limit.saturating_sub(SUBSCRIPTION_BYTE_RESERVED_MARGIN);
|
||
let mut admitted = Vec::new();
|
||
let mut used = already_used;
|
||
let mut overflow = 0usize;
|
||
for group in groups {
|
||
let size = req_message_size(&group);
|
||
if used
|
||
.checked_add(size)
|
||
.is_some_and(|total| total <= live_budget)
|
||
{
|
||
used += size;
|
||
admitted.push(group);
|
||
} else {
|
||
overflow += 1;
|
||
}
|
||
}
|
||
(admitted, overflow)
|
||
}
|
||
|
||
/// Pack filters into REQ-sized groups.
|
||
///
|
||
/// Groups are cut when adding the next filter would exceed either the
|
||
/// serialized byte budget or the filter-count cap. A single oversized filter
|
||
/// still gets its own group, so progress is always made.
|
||
fn group_filters_for_req_with_max(filters: &[Filter], max_filters: usize) -> Vec<Vec<Filter>> {
|
||
let mut groups: Vec<Vec<Filter>> = Vec::new();
|
||
let mut current: Vec<Filter> = Vec::new();
|
||
let mut bytes = 0usize;
|
||
for filter in filters {
|
||
let cost = filter.as_json().len();
|
||
if !current.is_empty()
|
||
&& (current.len() >= max_filters.max(1) || bytes + cost > REQ_MESSAGE_BYTE_BUDGET)
|
||
{
|
||
groups.push(std::mem::take(&mut current));
|
||
bytes = 0;
|
||
}
|
||
current.push(filter.clone());
|
||
bytes += cost;
|
||
}
|
||
if !current.is_empty() {
|
||
groups.push(current);
|
||
}
|
||
groups
|
||
}
|
||
|
||
fn group_filters_for_req(filters: &[Filter]) -> Vec<Vec<Filter>> {
|
||
group_filters_for_req_with_max(filters, MAX_FILTERS_PER_REQ)
|
||
}
|
||
|
||
fn live_filter_groups(filters: &[Filter], max_filters: usize) -> Vec<Vec<Filter>> {
|
||
group_filters_for_req_with_max(filters, max_filters)
|
||
.into_iter()
|
||
.map(|group| group.into_iter().map(|filter| filter.limit(0)).collect())
|
||
.collect()
|
||
}
|
||
|
||
fn reserve_connect_attempt(
|
||
in_flight: &mut HashMap<String, ConnectAttemptToken>,
|
||
next_token: &mut u64,
|
||
relay_url: &str,
|
||
) -> Option<ConnectAttemptToken> {
|
||
if in_flight.contains_key(relay_url) {
|
||
return None;
|
||
}
|
||
*next_token = next_token
|
||
.checked_add(1)
|
||
.expect("connect attempt token exhausted");
|
||
let token = ConnectAttemptToken(*next_token);
|
||
in_flight.insert(relay_url.to_string(), token);
|
||
Some(token)
|
||
}
|
||
|
||
fn take_connect_attempt(
|
||
in_flight: &mut HashMap<String, ConnectAttemptToken>,
|
||
relay_url: &str,
|
||
token: ConnectAttemptToken,
|
||
) -> bool {
|
||
if in_flight.get(relay_url) != Some(&token) {
|
||
return false;
|
||
}
|
||
in_flight.remove(relay_url);
|
||
true
|
||
}
|
||
|
||
async fn begin_connect_attempt(
|
||
semaphore: Arc<Semaphore>,
|
||
health_tracker: Arc<RelayHealthTracker>,
|
||
relay_url: &str,
|
||
) -> Option<tokio::sync::OwnedSemaphorePermit> {
|
||
let permit = semaphore.acquire_owned().await.ok()?;
|
||
health_tracker.record_attempt(relay_url);
|
||
Some(permit)
|
||
}
|
||
|
||
fn grouped_subscription_count(filters: &[Filter]) -> usize {
|
||
group_filters_for_req(filters).len()
|
||
}
|
||
|
||
#[derive(Debug, Default)]
|
||
struct DeferredConsolidations {
|
||
relays: HashSet<String>,
|
||
}
|
||
|
||
impl DeferredConsolidations {
|
||
fn request(&mut self, relay_url: &str, has_pending_batches: bool) -> bool {
|
||
if has_pending_batches {
|
||
self.relays.insert(relay_url.to_string());
|
||
false
|
||
} else {
|
||
self.relays.remove(relay_url);
|
||
true
|
||
}
|
||
}
|
||
|
||
fn take_ready(&mut self, relay_url: &str, has_pending_batches: bool) -> bool {
|
||
!has_pending_batches && self.relays.remove(relay_url)
|
||
}
|
||
|
||
fn contains(&self, relay_url: &str) -> bool {
|
||
self.relays.contains(relay_url)
|
||
}
|
||
|
||
fn cancel(&mut self, relay_url: &str) -> bool {
|
||
self.relays.remove(relay_url)
|
||
}
|
||
|
||
fn clear(&mut self) {
|
||
self.relays.clear();
|
||
}
|
||
}
|
||
|
||
// =============================================================================
|
||
// Daily Timer
|
||
// =============================================================================
|
||
|
||
/// Run the daily timer for periodic fresh syncs
|
||
///
|
||
/// This function runs in a loop, sleeping for a random interval between
|
||
/// 23-25 hours, then triggering a daily sync for all relays. The random
|
||
/// interval prevents thundering herd effects across multiple ngit-grasp instances.
|
||
///
|
||
/// The daily sync:
|
||
/// - Unsubscribes from all current subscriptions
|
||
/// - Clears pending batches and sync state
|
||
/// - Re-discovers all repos and events from scratch
|
||
///
|
||
/// This detects state drift over time that might occur from missed events.
|
||
async fn run_daily_timer(
|
||
sync_manager: Arc<Mutex<SyncManager>>,
|
||
mut shutdown_rx: broadcast::Receiver<()>,
|
||
) {
|
||
use ::rand::RngExt;
|
||
|
||
loop {
|
||
// Random interval between 23-25 hours
|
||
let hours = 23.0 + ::rand::rng().random::<f64>() * 2.0;
|
||
let seconds = (hours * 3600.0) as u64;
|
||
|
||
tracing::info!(
|
||
hours = format!("{:.1}", hours),
|
||
"Daily timer scheduled to fire in {:.1} hours",
|
||
hours
|
||
);
|
||
|
||
tokio::select! {
|
||
_ = tokio::time::sleep(Duration::from_secs(seconds)) => {
|
||
// Timer fired - do daily sync
|
||
// Get list of relays
|
||
let relay_urls: Vec<String> = {
|
||
let manager = sync_manager.lock().await;
|
||
let index = manager.relay_sync_index.read().await;
|
||
let urls: Vec<String> = index.keys().cloned().collect();
|
||
drop(index);
|
||
urls
|
||
};
|
||
|
||
tracing::info!(
|
||
relay_count = relay_urls.len(),
|
||
"Daily timer fired, starting daily sync for all relays"
|
||
);
|
||
|
||
// Trigger daily sync for each relay
|
||
for relay_url in relay_urls {
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.daily_sync(&relay_url).await;
|
||
}
|
||
}
|
||
_ = shutdown_rx.recv() => {
|
||
tracing::info!("Daily timer received shutdown signal");
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Background task that periodically syncs purgatory announcements into repo_sync_index.
|
||
///
|
||
/// Runs every 5 seconds by default (200ms when `NGIT_TEST=1`).
|
||
/// For each announcement currently in purgatory, ensures there is a `StateOnly` entry in
|
||
/// `repo_sync_index`. New entries trigger `handle_new_sync_filters` which connects to the
|
||
/// relay URLs listed in the announcement and subscribes to state events (kind 30618).
|
||
///
|
||
/// This is the sole registration path for purgatory announcements:
|
||
/// - Sync-path announcements: registered here within one interval of arriving.
|
||
/// - User-submitted purgatory announcements: the SelfSubscriber never sees them
|
||
/// (they're rejected from DB), so this timer is the only registration path.
|
||
///
|
||
/// The same tick also drives bounded recovery of events that relays reported
|
||
/// during negentropy reconciliation but failed to deliver on exact-ID fetches
|
||
/// (see [`missing_events`]). Recovery attempts are backed off per relay, so
|
||
/// the tick itself stays cheap when nothing is due.
|
||
///
|
||
/// Finally, one relay's live descendant admission is reconciled and every
|
||
/// constrained relay may advance one queued descendant history query. Reusing
|
||
/// this five-second cadence avoids another scheduler or configuration surface.
|
||
async fn run_purgatory_announcement_sync(
|
||
sync_manager: Arc<Mutex<SyncManager>>,
|
||
mut shutdown_rx: broadcast::Receiver<()>,
|
||
) {
|
||
let interval = if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_millis(200)
|
||
} else {
|
||
Duration::from_secs(5)
|
||
};
|
||
loop {
|
||
tokio::select! {
|
||
_ = tokio::time::sleep(interval) => {
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.sync_purgatory_announcements_to_index().await;
|
||
manager.reconcile_private_membership().await;
|
||
manager.tick_missing_event_recovery().await;
|
||
manager.tick_descendant_sync().await;
|
||
}
|
||
_ = shutdown_rx.recv() => {
|
||
tracing::debug!("Purgatory announcement sync timer received shutdown signal");
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Combined Health and Metrics Checker
|
||
|
||
/// Background task for cleaning up expired entries from the rejected events index
|
||
///
|
||
/// This task runs two cleanup operations at different intervals:
|
||
/// 1. **Hot cache cleanup (60s)**: Remove events older than 2 minutes from hot cache
|
||
/// 2. **Cold index cleanup (daily)**: Remove metadata older than 7 days from cold index
|
||
///
|
||
/// A single `RejectedEventsIndex` handles both announcement and state events,
|
||
/// differentiated by `EventType`. Each cleanup pass processes both types.
|
||
///
|
||
/// The hot cache cleanup runs frequently to keep memory usage low (events expire quickly).
|
||
/// The cold index cleanup runs daily since metadata is small and expires slowly.
|
||
async fn run_rejected_index_cleanup(
|
||
sync_manager: Arc<Mutex<SyncManager>>,
|
||
mut shutdown_rx: broadcast::Receiver<()>,
|
||
) {
|
||
let hot_cache_interval = Duration::from_secs(60);
|
||
let cold_index_interval = Duration::from_secs(86400); // 24 hours
|
||
|
||
tracing::info!("Rejected index cleanup started (hot cache: 60s, cold index: daily)");
|
||
|
||
let mut hot_cache_timer = tokio::time::interval(hot_cache_interval);
|
||
let mut cold_index_timer = tokio::time::interval(cold_index_interval);
|
||
|
||
// Tick immediately to set the initial delay
|
||
hot_cache_timer.tick().await;
|
||
cold_index_timer.tick().await;
|
||
|
||
loop {
|
||
tokio::select! {
|
||
_ = hot_cache_timer.tick() => {
|
||
let manager = sync_manager.lock().await;
|
||
|
||
// Clean up hot cache for both event types (single index handles both)
|
||
// Note: cleanup_expired_for_type updates metrics with type label
|
||
let (ann_hot_expired, _) = manager.rejected_events_index.cleanup_expired_for_type("announcement");
|
||
let (state_hot_expired, _) = manager.rejected_events_index.cleanup_expired_for_type("state");
|
||
|
||
if ann_hot_expired + state_hot_expired > 0 {
|
||
tracing::debug!(
|
||
announcements = ann_hot_expired,
|
||
states = state_hot_expired,
|
||
"Cleaned up expired entries from rejected events hot cache"
|
||
);
|
||
}
|
||
}
|
||
_ = cold_index_timer.tick() => {
|
||
let manager = sync_manager.lock().await;
|
||
|
||
// Clean up cold index for both event types (single index handles both)
|
||
let (_, ann_cold_expired) = manager.rejected_events_index.cleanup_expired_for_type("announcement");
|
||
let (_, state_cold_expired) = manager.rejected_events_index.cleanup_expired_for_type("state");
|
||
let unrecoverable_expired = manager.rejected_events_index.cleanup_expired_unrecoverable();
|
||
let related_expired = manager.rejected_events_index.cleanup_expired_related();
|
||
|
||
if ann_cold_expired + state_cold_expired + unrecoverable_expired + related_expired > 0 {
|
||
tracing::info!(
|
||
announcements = ann_cold_expired,
|
||
states = state_cold_expired,
|
||
unrecoverable = unrecoverable_expired,
|
||
related = related_expired,
|
||
"Cleaned up expired entries from rejected events cold index"
|
||
);
|
||
}
|
||
}
|
||
_ = shutdown_rx.recv() => {
|
||
tracing::info!("Rejected index cleanup received shutdown signal");
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Background task for checking relay health and updating metrics
|
||
///
|
||
/// This task runs every 2 seconds and performs three operations:
|
||
///
|
||
/// 1. **Disconnect checking**: Check for empty relays and disconnect non-bootstrap ones
|
||
/// 2. **Rate limit recovery**: Check for relays whose rate limit cooldown has expired
|
||
/// 3. **Metrics update**: Update Prometheus metrics with current health states from health_tracker
|
||
///
|
||
/// The metrics update ensures that health states are kept current in metrics even when
|
||
/// they change due to timeouts, cooldowns expiring, or stability periods completing.
|
||
///
|
||
/// The 2-second interval provides a good balance between responsiveness and overhead.
|
||
/// While disconnect checking traditionally ran at 60s intervals, the faster cadence here
|
||
/// is acceptable since the operations are lightweight (just index checks, no I/O).
|
||
async fn run_health_and_metrics_checker(
|
||
sync_manager: Arc<Mutex<SyncManager>>,
|
||
mut shutdown_rx: broadcast::Receiver<()>,
|
||
) {
|
||
let interval = Duration::from_secs(2);
|
||
tracing::info!("Health and metrics checker started with 2s interval");
|
||
|
||
loop {
|
||
tokio::select! {
|
||
_ = tokio::time::sleep(interval) => {
|
||
let mut manager = sync_manager.lock().await;
|
||
|
||
// 1. Reset failure history only after a recovered connection
|
||
// has survived the stability period under normal sync load.
|
||
manager.health_tracker.promote_stable_connections();
|
||
|
||
// 2. Check for disconnects and retry disconnected relays
|
||
manager.check_disconnects().await;
|
||
manager.retry_disconnected_relays().await;
|
||
|
||
// 3. Check for rate limit recovery
|
||
manager.check_rate_limit_recovery().await;
|
||
|
||
// 4. Keep deliberately bounded live coverage complete through
|
||
// paced incremental history rather than retrying an impossible
|
||
// persistent set.
|
||
manager.sync_due_byte_limited_relay().await;
|
||
|
||
// 5. Discover one bounded batch of participant NIP-65 mailboxes.
|
||
// The query uses an ordinary transient ledger slot and yields
|
||
// to historic work when no slot is immediately available.
|
||
manager.schedule_nip65_discovery().await;
|
||
|
||
// 6. Start at most one participant mailbox history filter.
|
||
// Mailbox coverage is history-only: it does not turn every
|
||
// participant relay into a permanent live source.
|
||
manager.schedule_mailbox_probe().await;
|
||
|
||
// 7. Check for naughty list expiration
|
||
if let Some(naughty_list) = manager.health_tracker.naughty_list() {
|
||
let recovered = naughty_list.expire_old_entries();
|
||
for url in recovered {
|
||
tracing::info!(
|
||
relay = %url,
|
||
"Relay removed from naughty list after expiration, will retry"
|
||
);
|
||
}
|
||
}
|
||
|
||
// 8. Update metrics with current health states and naughty list
|
||
if let Some(ref metrics) = manager.metrics {
|
||
// Get all tracked relay URLs
|
||
let relay_urls: Vec<String> = {
|
||
let index = manager.relay_sync_index.read().await;
|
||
index.keys().cloned().collect()
|
||
};
|
||
|
||
// Update health state for each relay
|
||
for relay_url in relay_urls {
|
||
let state = manager.health_tracker.get_state(&relay_url);
|
||
metrics.record_health_state(&relay_url, state);
|
||
}
|
||
|
||
// Update naughty list metrics
|
||
if let Some(naughty_list) = manager.health_tracker.naughty_list() {
|
||
let entries = naughty_list.get_all();
|
||
metrics.update_naughty_list(entries);
|
||
}
|
||
|
||
let (pending_batches, pending_subscriptions) = {
|
||
let pending = manager.pending_sync_index.read().await;
|
||
(
|
||
pending.values().map(Vec::len).sum(),
|
||
pending
|
||
.values()
|
||
.flatten()
|
||
.map(|batch| batch.outstanding_subs.len())
|
||
.sum(),
|
||
)
|
||
};
|
||
for (class, count) in [
|
||
("pending_batches", pending_batches),
|
||
("pending_subscriptions", pending_subscriptions),
|
||
("auth_retries", manager.auth_required_attempts.len()),
|
||
("queued_connection_attempts", manager.in_flight_connect_attempts.len()),
|
||
("purgatory_dependencies", manager.purgatory_dependency_attempts.len()),
|
||
("dependency_relays", manager.dependency_relay_deadlines.len()),
|
||
("related_dependency_events", manager.rejected_events_index.related_len()),
|
||
("deferred_consolidations", manager.deferred_consolidations.relays.len()),
|
||
("descendant_rotations", manager.descendant_sync_rotations.len()),
|
||
(
|
||
"nip65_eligible_authors",
|
||
manager.nip65_discovery.eligible_authors.len(),
|
||
),
|
||
(
|
||
"mailbox_probe_relays",
|
||
manager.nip65_discovery.mailbox_roots.len(),
|
||
),
|
||
(
|
||
"mailbox_probe_cursors",
|
||
manager.nip65_discovery.mailbox_probe_next_filter.len(),
|
||
),
|
||
(
|
||
"mailbox_probe_active",
|
||
manager.nip65_discovery.mailbox_probes_in_flight.len(),
|
||
),
|
||
] {
|
||
metrics.set_retained_state(class, count);
|
||
}
|
||
}
|
||
}
|
||
_ = shutdown_rx.recv() => {
|
||
tracing::info!("Health and metrics checker received shutdown signal");
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Manages proactive synchronization with external relays
|
||
///
|
||
/// The SyncManager runs as a background task, subscribing to repository
|
||
/// announcements on the local relay and syncing data from external relays
|
||
/// listed in those announcements.
|
||
pub struct SyncManager {
|
||
/// Bootstrap relay URL for initial sync (optional)
|
||
bootstrap_relay_url: Option<String>,
|
||
/// Our service domain - used for filtering relevant repos
|
||
service_domain: String,
|
||
/// Database for event storage and queries
|
||
database: SharedDatabase,
|
||
/// Write policy for validating incoming events
|
||
write_policy: Nip34WritePolicy,
|
||
/// Purgatory for read-only access to events awaiting git data
|
||
purgatory: Arc<crate::purgatory::Purgatory>,
|
||
/// Local relay for submitting synced events (enables broadcast to WebSocket subscribers)
|
||
local_relay: LocalRelay,
|
||
/// Configuration reference for sync settings
|
||
config: Config,
|
||
/// What we want to sync (source of truth)
|
||
repo_sync_index: RepoSyncIndex,
|
||
root_candidate_index: RootCandidateIndex,
|
||
proactive_participant_authors: crate::nostr::policy::SharedProactiveParticipantAuthorIndex,
|
||
/// GRASP-08 access shared with the inbound HTTP/WebSocket boundary.
|
||
private_access: Option<PrivateAccess>,
|
||
/// Operator-configured members form the permanent base of private access.
|
||
configured_private_members: HashSet<PublicKey>,
|
||
/// Latest NIP-11 owner learned for each connected repository relay whose
|
||
/// NIP-11 also advertises GRASP-08. Only private-service owners can mint
|
||
/// derived membership: the owner of a public relay gains nothing
|
||
/// legitimate from private membership, since their relay enforces no
|
||
/// confidentiality for the repositories it mirrors.
|
||
relay_owners: HashMap<String, PublicKey>,
|
||
/// What we've confirmed syncing + connection state
|
||
relay_sync_index: RelaySyncIndex,
|
||
/// In-flight subscription batches
|
||
pending_sync_index: PendingSyncIndex,
|
||
/// Rejected events (30617/30618) - two-tier storage for re-processing
|
||
/// Handles both announcement and state events via EventType discriminator
|
||
rejected_events_index: Arc<RejectedEventsIndex>,
|
||
/// Active relay connections - keyed by relay URL
|
||
connections: HashMap<String, RelayConnection>,
|
||
/// Connections opened only to discover accepted authors' kind 0/10002.
|
||
///
|
||
/// These share the ordinary connection, safety, and reconnect machinery,
|
||
/// but must not turn a user-index or outbox relay into a repository sync
|
||
/// source merely because it answered a control-plane lookup.
|
||
nip65_discovery_only_relays: HashSet<String>,
|
||
/// Adaptive pagination learning for each relay's current connection session.
|
||
pagination_sessions: HashMap<String, RelayPaginationSession>,
|
||
/// Event-directed relay targets rejected by the outbound target policy.
|
||
///
|
||
/// Rejected URLs stay in `repo_sync_index` (they come from stored events),
|
||
/// so without this memo every sync pass would re-derive, re-reject, and
|
||
/// re-log the same forbidden target. Bounded by the set of distinct relay
|
||
/// URLs in stored/purgatory events, which the indexes already carry.
|
||
rejected_relay_targets: HashSet<String>,
|
||
/// Relays whose NIP-11 advertises GRASP-08 while this instance is public.
|
||
///
|
||
/// Laundering guard: a public mirror must not present private credentials
|
||
/// to such a relay nor hammer a service that will never admit it, so the
|
||
/// target is parked before the dial (no WebSocket, no AUTH exchange).
|
||
/// Held in memory only, re-probed at most once per process lifetime.
|
||
private_service_relays: HashSet<String>,
|
||
/// Last exact-ID dependency recovery attempt, used to bound retries.
|
||
dependency_refetch_attempts: Arc<std::sync::Mutex<HashMap<EventId, Instant>>>,
|
||
/// Events relays reported during negentropy reconciliation but failed to
|
||
/// deliver on exact-ID fetches. Retried with bounded backoff by the sync
|
||
/// maintenance timer instead of being dropped with their failed batch.
|
||
missing_event_recovery: Arc<std::sync::Mutex<missing_events::MissingEventRecoveryIndex>>,
|
||
/// Last dependency pass for each purgatory announcement.
|
||
purgatory_dependency_attempts: HashMap<EventId, Instant>,
|
||
/// Temporary source relays retained while rejected dependencies are recovered.
|
||
dependency_relay_deadlines: HashMap<String, Instant>,
|
||
/// First auth-required CLOSED per subscription. rust-nostr retains the
|
||
/// same ID while it answers NIP-42; a repeat means authentication did not
|
||
/// authorize this query and becomes a bounded policy refusal.
|
||
auth_required_attempts: HashSet<(String, SubscriptionId)>,
|
||
/// Health tracker for relay connection state
|
||
health_tracker: Arc<RelayHealthTracker>,
|
||
/// Counter for generating unique batch IDs
|
||
next_batch_id: u64,
|
||
/// Monotonic identity for rejecting stale connection results.
|
||
next_connect_attempt_token: u64,
|
||
/// One active or queued connection attempt per canonical relay URL.
|
||
in_flight_connect_attempts: HashMap<String, ConnectAttemptToken>,
|
||
/// Shared bound for DNS and websocket handshakes.
|
||
connect_attempt_semaphore: Arc<Semaphore>,
|
||
/// Relays whose subscription consolidation waits for in-flight batches to drain.
|
||
deferred_consolidations: DeferredConsolidations,
|
||
/// Relays whose complete persistent filter set exceeds a learned remote
|
||
/// byte cap, mapped to their next bounded catch-up deadline.
|
||
byte_limited_live_relays: HashMap<String, Instant>,
|
||
/// Per-relay snapshots for low-priority, EOSE-closing descendant queries.
|
||
descendant_sync_rotations: HashMap<String, DescendantSyncRotation>,
|
||
/// Auxiliary persistent descendant coverage, kept separate from the core
|
||
/// desired set so it can be retired without replacing healthy core REQs.
|
||
descendant_live_coverage: HashMap<String, DescendantLiveCoverage>,
|
||
/// Round-robin cursor so only one relay performs descendant live-mode
|
||
/// derivation and admission work per maintenance tick.
|
||
descendant_relay_cursor: usize,
|
||
/// Narrow GRASP-03 identity discovery and paced mailbox-probe state.
|
||
nip65_discovery: Nip65DiscoveryState,
|
||
/// Channel for disconnect notifications (set during run)
|
||
disconnect_tx: Option<tokio::sync::mpsc::Sender<DisconnectNotification>>,
|
||
/// Channel for EOSE notifications (set during run)
|
||
eose_tx: Option<tokio::sync::mpsc::Sender<EoseNotification>>,
|
||
/// Serializes CLOSED recovery and pending-batch cleanup through the actor.
|
||
subscription_closed_tx: Option<tokio::sync::mpsc::Sender<SubscriptionClosedNotification>>,
|
||
/// Returns connection outcomes to the sync actor for serialized state changes.
|
||
connect_attempt_result_tx: Option<tokio::sync::mpsc::Sender<ConnectAttemptResult>>,
|
||
nip65_discovery_result_tx: Option<tokio::sync::mpsc::Sender<Nip65DiscoveryResult>>,
|
||
mailbox_probe_result_tx: Option<tokio::sync::mpsc::Sender<MailboxProbeResult>>,
|
||
/// Channel for broadcasting shutdown signal to all background tasks
|
||
shutdown_tx: Option<broadcast::Sender<()>>,
|
||
/// Prometheus metrics for sync operations (None if metrics disabled)
|
||
metrics: Option<SyncMetrics>,
|
||
}
|
||
|
||
impl SyncManager {
|
||
/// Create a new SyncManager
|
||
///
|
||
/// # Arguments
|
||
/// * `bootstrap_relay_url` - Optional relay URL for initial historical sync
|
||
/// * `service_domain` - The domain this relay serves (for filtering repos)
|
||
/// * `database` - Shared database for event storage
|
||
/// * `write_policy` - Policy for validating events before storage
|
||
/// * `local_relay` - Local relay for submitting synced events (enables WebSocket broadcast)
|
||
/// * `config` - Configuration for sync settings
|
||
/// * `data_path` - Path to git data directory (for persistence)
|
||
/// * `sync_metrics` - Optional pre-registered SyncMetrics (passed from Metrics if metrics are enabled)
|
||
#[allow(clippy::too_many_arguments)]
|
||
pub fn new(
|
||
bootstrap_relay_url: Option<String>,
|
||
service_domain: String,
|
||
database: SharedDatabase,
|
||
write_policy: Nip34WritePolicy,
|
||
local_relay: LocalRelay,
|
||
config: &Config,
|
||
data_path: PathBuf,
|
||
sync_metrics: Option<SyncMetrics>,
|
||
private_access: Option<PrivateAccess>,
|
||
) -> Self {
|
||
// Extract purgatory from write_policy for read-only access
|
||
let purgatory = write_policy.purgatory().clone();
|
||
|
||
// Create rejected events index
|
||
let rejected_events_index = Arc::new(if let Some(ref metrics) = sync_metrics {
|
||
RejectedEventsIndex::with_metrics(
|
||
Duration::from_secs(config.rejected_hot_cache_duration_secs),
|
||
Duration::from_secs(config.rejected_cold_index_expiry_secs),
|
||
metrics.clone(),
|
||
)
|
||
} else {
|
||
RejectedEventsIndex::new(
|
||
Duration::from_secs(config.rejected_hot_cache_duration_secs),
|
||
Duration::from_secs(config.rejected_cold_index_expiry_secs),
|
||
)
|
||
});
|
||
|
||
// Attempt to restore rejected events index from disk
|
||
let rejected_index_path = data_path.join("rejected-events-cache.json");
|
||
if rejected_index_path.exists() {
|
||
match rejected_events_index.restore_from_disk(&rejected_index_path) {
|
||
Ok(()) => {
|
||
tracing::info!("Restored rejected events index from disk");
|
||
}
|
||
Err(e) => {
|
||
tracing::warn!(
|
||
"Failed to restore rejected events index: {}, starting empty",
|
||
e
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
let proactive_participant_authors = write_policy.proactive_participant_authors();
|
||
let configured_private_members = config
|
||
.parse_private_members()
|
||
.expect("private members were validated before SyncManager construction")
|
||
.into_iter()
|
||
.collect();
|
||
Self {
|
||
bootstrap_relay_url,
|
||
service_domain,
|
||
database,
|
||
write_policy,
|
||
purgatory,
|
||
local_relay,
|
||
config: config.clone(),
|
||
repo_sync_index: Arc::new(RwLock::new(HashMap::new())),
|
||
root_candidate_index: Arc::new(RwLock::new(HashMap::new())),
|
||
proactive_participant_authors,
|
||
private_access,
|
||
configured_private_members,
|
||
relay_owners: HashMap::new(),
|
||
relay_sync_index: Arc::new(RwLock::new(HashMap::new())),
|
||
pending_sync_index: Arc::new(RwLock::new(HashMap::new())),
|
||
rejected_events_index,
|
||
connections: HashMap::new(),
|
||
nip65_discovery_only_relays: HashSet::new(),
|
||
pagination_sessions: HashMap::new(),
|
||
rejected_relay_targets: HashSet::new(),
|
||
private_service_relays: HashSet::new(),
|
||
dependency_refetch_attempts: Arc::new(std::sync::Mutex::new(HashMap::new())),
|
||
missing_event_recovery: Arc::new(std::sync::Mutex::new(
|
||
missing_events::MissingEventRecoveryIndex::default(),
|
||
)),
|
||
purgatory_dependency_attempts: HashMap::new(),
|
||
dependency_relay_deadlines: HashMap::new(),
|
||
auth_required_attempts: HashSet::new(),
|
||
health_tracker: Arc::new(RelayHealthTracker::new(config)),
|
||
next_batch_id: 0,
|
||
next_connect_attempt_token: 0,
|
||
in_flight_connect_attempts: HashMap::new(),
|
||
connect_attempt_semaphore: Arc::new(Semaphore::new(MAX_CONCURRENT_CONNECT_ATTEMPTS)),
|
||
deferred_consolidations: DeferredConsolidations::default(),
|
||
byte_limited_live_relays: HashMap::new(),
|
||
descendant_sync_rotations: HashMap::new(),
|
||
descendant_live_coverage: HashMap::new(),
|
||
descendant_relay_cursor: 0,
|
||
nip65_discovery: Nip65DiscoveryState::default(),
|
||
disconnect_tx: None,
|
||
eose_tx: None,
|
||
subscription_closed_tx: None,
|
||
connect_attempt_result_tx: None,
|
||
nip65_discovery_result_tx: None,
|
||
mailbox_probe_result_tx: None,
|
||
shutdown_tx: None,
|
||
metrics: sync_metrics,
|
||
}
|
||
}
|
||
|
||
/// Generate a unique batch ID
|
||
///
|
||
/// Increments the internal counter and returns the new value.
|
||
/// Used for tracking pending batches and debugging/logging.
|
||
fn next_batch_id(&mut self) -> u64 {
|
||
self.next_batch_id += 1;
|
||
self.next_batch_id
|
||
}
|
||
|
||
/// Get a clone of the rejected events index Arc.
|
||
///
|
||
/// This allows access to the rejected events index for persistence
|
||
/// even after the SyncManager has been moved into a task.
|
||
///
|
||
/// # Returns
|
||
/// Arc clone of the rejected events index
|
||
pub fn rejected_events_index(&self) -> Arc<RejectedEventsIndex> {
|
||
self.rejected_events_index.clone()
|
||
}
|
||
|
||
/// Save rejected events index to disk.
|
||
///
|
||
/// This is called during shutdown to persist the rejected events cache,
|
||
/// allowing us to avoid re-downloading rejected events after restart.
|
||
///
|
||
/// # Arguments
|
||
/// * `path` - Path to save the rejected index file
|
||
///
|
||
/// # Returns
|
||
/// Ok(()) on success, Err if save fails
|
||
pub fn save_rejected_index(&self, path: &Path) -> Result<(), Box<dyn std::error::Error>> {
|
||
self.rejected_events_index.save_to_disk(path)
|
||
}
|
||
|
||
/// Handle EOSE (End Of Stored Events) for a subscription
|
||
///
|
||
/// This method:
|
||
/// - Finds the PendingBatch containing this subscription ID
|
||
/// - Removes the subscription from outstanding_subs
|
||
/// - When all subscriptions complete (outstanding_subs empty):
|
||
/// - Calls confirm_batch to move items to confirmed state
|
||
async fn handle_eose(&mut self, relay_url: &str, sub_id: SubscriptionId) {
|
||
// Check if relay is in Disconnecting state
|
||
let is_disconnecting = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.get(relay_url)
|
||
.map(|s| s.connection_status == ConnectionStatus::Disconnecting)
|
||
.unwrap_or(false)
|
||
};
|
||
|
||
// 1. Find and update the pending batch
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
|
||
let Some(batches) = pending.get_mut(relay_url) else {
|
||
// This can happen when EOSE arrives after batch has already been confirmed/removed.
|
||
// Common causes:
|
||
// 1. During intentional disconnect (cleanup in progress)
|
||
// 2. Duplicate/late EOSE from relay (e.g., live_sync REQ subscriptions may send
|
||
// multiple EOSE messages - some relays do this)
|
||
// 3. Race condition between batch confirmation and EOSE arrival
|
||
//
|
||
// NOTE: If we wanted to investigate whether these are truly duplicate EOSEs,
|
||
// we could track recently-completed subscription IDs (with timestamps) and
|
||
// check if this sub_id was recently confirmed. This would distinguish between:
|
||
// - Duplicate EOSE (sub_id was recently in outstanding_subs)
|
||
// - Truly unknown subscription (sub_id never tracked)
|
||
if is_disconnecting {
|
||
// Expected during intentional disconnect - suppress noisy log
|
||
tracing::trace!(
|
||
relay = %relay_url,
|
||
sub_id = %sub_id,
|
||
"EOSE received during disconnect cleanup - ignoring"
|
||
);
|
||
} else {
|
||
// Expected when batch completes before late/duplicate EOSE arrives
|
||
tracing::trace!(
|
||
relay = %relay_url,
|
||
sub_id = %sub_id,
|
||
"EOSE received after batch already completed (late or duplicate EOSE)"
|
||
);
|
||
}
|
||
return;
|
||
};
|
||
|
||
// Find the batch containing this subscription
|
||
let batch_index = batches
|
||
.iter()
|
||
.position(|b| b.outstanding_subs.contains(&sub_id));
|
||
|
||
let Some(batch_idx) = batch_index else {
|
||
// Live subscriptions (limit:0, no auto-close) are not tracked in PendingBatch.
|
||
// They complete immediately with EOSE (no historic events) and stay open for new events.
|
||
// Observed in production: sync_live() subscriptions trigger this path (expected).
|
||
// Also possible: duplicate/late EOSE from relay after batch already completed.
|
||
tracing::trace!(
|
||
relay = %relay_url,
|
||
sub_id = %sub_id,
|
||
"EOSE received for subscription not tracked in batch (live subscription or late EOSE)"
|
||
);
|
||
return;
|
||
};
|
||
|
||
// Remove the subscription from outstanding_subs
|
||
let batch = &mut batches[batch_idx];
|
||
batch.outstanding_subs.remove(&sub_id);
|
||
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
sub_id = %sub_id,
|
||
batch_id = batch.batch_id,
|
||
remaining_subs = batch.outstanding_subs.len(),
|
||
"EOSE processed for subscription"
|
||
);
|
||
|
||
// Check for pagination using this relay connection session's observed page sizes and
|
||
// verified NIP-11 default-limit hint.
|
||
if let Some(pagination_state) = batch.pagination_state.remove(&sub_id) {
|
||
let next_page = pagination_state.next_page(
|
||
self.pagination_sessions
|
||
.entry(relay_url.to_string())
|
||
.or_default(),
|
||
);
|
||
if let Some(next_page) = next_page {
|
||
let next_filters = next_page.filters();
|
||
let relay_url_for_pagination = relay_url.to_string();
|
||
let batch_id = batch.batch_id;
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
sub_id = %sub_id,
|
||
batch_id,
|
||
filter_count = next_filters.len(),
|
||
"Grouped subscription hit pagination threshold, fetching next page"
|
||
);
|
||
|
||
// A NOTICE can arrive immediately before this page's EOSE.
|
||
// Keep a sentinel in the batch and let a detached worker
|
||
// resume the exact grouped page after the cooldown.
|
||
if self.health_tracker.is_subscription_paused(relay_url) {
|
||
let deferred_sub_id = mark_deferred_pagination(batch, &sub_id);
|
||
drop(pending);
|
||
|
||
let Some(connection) = self.connections.get(&relay_url_for_pagination).cloned()
|
||
else {
|
||
tracing::error!(
|
||
relay = %relay_url_for_pagination,
|
||
batch_id,
|
||
"Cannot defer rate-limited pagination without a relay connection"
|
||
);
|
||
return;
|
||
};
|
||
Self::spawn_deferred_pagination(
|
||
connection,
|
||
self.health_tracker.clone(),
|
||
self.pending_sync_index.clone(),
|
||
relay_url_for_pagination,
|
||
batch_id,
|
||
deferred_sub_id,
|
||
next_page,
|
||
);
|
||
return;
|
||
}
|
||
|
||
drop(pending);
|
||
|
||
let mut next_page_started = false;
|
||
if let Some(conn) = self.connections.get(&relay_url_for_pagination) {
|
||
match conn
|
||
.subscribe_filters(next_filters.clone(), next_page.request_class())
|
||
.await
|
||
{
|
||
Ok(new_sub_id) => {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_for_pagination) {
|
||
if let Some(batch) =
|
||
batches.iter_mut().find(|b| b.batch_id == batch_id)
|
||
{
|
||
batch.outstanding_subs.insert(new_sub_id.clone());
|
||
next_page_started = true;
|
||
batch.pagination_state.insert(new_sub_id.clone(), next_page);
|
||
tracing::info!(
|
||
relay = %relay_url_for_pagination,
|
||
new_sub_id = %new_sub_id,
|
||
batch_id,
|
||
"Next grouped page subscription created"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
Err(error) => {
|
||
tracing::error!(
|
||
relay = %relay_url_for_pagination,
|
||
batch_id,
|
||
error = %error,
|
||
"Failed to create grouped pagination subscription"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
if !next_page_started {
|
||
let completed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
take_drained_batch_as_failed(
|
||
&mut pending,
|
||
&relay_url_for_pagination,
|
||
batch_id,
|
||
)
|
||
};
|
||
if let Some(batch) = completed_batch {
|
||
tracing::warn!(
|
||
relay = %relay_url_for_pagination,
|
||
batch_id,
|
||
"Pagination could not continue; completing drained batch as failed"
|
||
);
|
||
self.confirm_batch(&relay_url_for_pagination, batch).await;
|
||
}
|
||
}
|
||
|
||
return;
|
||
}
|
||
}
|
||
|
||
// Check if batch is complete
|
||
if !batch.outstanding_subs.is_empty() {
|
||
return;
|
||
}
|
||
|
||
// 2. Batch complete - validate negentropy ID fetches before confirming
|
||
// For negentropy batches, check if all requested events were received
|
||
if batch.sync_method == SyncMethod::Negentropy {
|
||
if let (Some(requested), Some(received)) =
|
||
(&batch.requested_event_ids, &batch.received_event_ids)
|
||
{
|
||
let missing: Vec<EventId> = requested.difference(received).cloned().collect();
|
||
|
||
if !missing.is_empty() {
|
||
let requested_count = requested.len();
|
||
let received_count = received.len();
|
||
let retry_count = batch.retry_count;
|
||
let initial_hydration_counts = batch
|
||
.initial_hydration_counts
|
||
.unwrap_or((requested_count, received_count));
|
||
if retry_count == 0 {
|
||
batch.initial_hydration_counts = Some(initial_hydration_counts);
|
||
}
|
||
|
||
// A semantic fallback is reserved for relays whose first exact-ID fetch is
|
||
// materially incompatible with their advertised inventory. Small residuals
|
||
// can be caused by indexing lag, expiry, or concurrent deletion and must not
|
||
// disable NIP-77 for an otherwise healthy relay.
|
||
if retry_count > 0
|
||
&& received_count == 0
|
||
&& should_use_semantic_fallback(
|
||
initial_hydration_counts.0,
|
||
initial_hydration_counts.1,
|
||
)
|
||
{
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id = batch.batch_id,
|
||
retry_count = retry_count,
|
||
requested_count = requested_count,
|
||
missing_count = missing.len(),
|
||
initial_requested_count = initial_hydration_counts.0,
|
||
initial_received_count = initial_hydration_counts.1,
|
||
missing_ids_sample = ?missing.iter().take(5).map(|id| id.to_hex()).collect::<Vec<_>>(),
|
||
"Negentropy exact-ID hydration delivered at most 10% of a material batch. \
|
||
Marking relay as incompatible with negentropy hydration and falling back to REQ+EOSE."
|
||
);
|
||
|
||
// Mark relay as not supporting negentropy so future batches skip it
|
||
if let Some(conn) = self.connections.get(relay_url) {
|
||
conn.mark_negentropy_unsupported();
|
||
}
|
||
|
||
// Prepare for REQ+EOSE fallback using semantic filters
|
||
// (not ID-based queries which already failed)
|
||
let relay_url_for_fallback = relay_url.to_string();
|
||
let batch_id = batch.batch_id;
|
||
let batch_full_repos = batch.items.repos.clone();
|
||
let batch_state_only_repos = batch.items.state_only_repos.clone();
|
||
let batch_root_events = batch.items.root_events.clone();
|
||
let missing_count = missing.len();
|
||
|
||
// Drop the lock before async operations
|
||
drop(pending);
|
||
|
||
// Create REQ+EOSE subscriptions using original semantic filters
|
||
// This queries by kind/author/tags instead of by ID, which may
|
||
// succeed even when ID-based queries fail.
|
||
// Preserve the level this batch was created with. The desired
|
||
// level may have advanced from StateOnly to Full while the
|
||
// historic request was in flight.
|
||
let fallback_filters = filters::build_sync_level_aware_filters(
|
||
&batch_full_repos,
|
||
&batch_state_only_repos,
|
||
&batch_root_events,
|
||
None,
|
||
);
|
||
|
||
if fallback_filters.is_empty() {
|
||
tracing::warn!(
|
||
relay = %relay_url_for_fallback,
|
||
batch_id = batch_id,
|
||
full_repos = batch_full_repos.len(),
|
||
state_only_repos = batch_state_only_repos.len(),
|
||
root_events = batch_root_events.len(),
|
||
"Cannot create semantic fallback filters - no repos or root_events in batch"
|
||
);
|
||
// Fall through to ID-based fallback as last resort
|
||
}
|
||
|
||
let mut new_sub_ids = HashSet::new();
|
||
if let Some(conn) = self.connections.get(&relay_url_for_fallback) {
|
||
for filter_group in group_filters_for_req(&fallback_filters) {
|
||
match conn
|
||
.subscribe_filters(
|
||
filter_group,
|
||
TransientRequestClass::NegentropyFallback,
|
||
)
|
||
.await
|
||
{
|
||
Ok(sub_id) => {
|
||
new_sub_ids.insert(sub_id);
|
||
}
|
||
Err(e) => {
|
||
tracing::error!(
|
||
relay = %relay_url_for_fallback,
|
||
batch_id = batch_id,
|
||
error = %e,
|
||
"Failed to create REQ+EOSE fallback subscription"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
if !new_sub_ids.is_empty() {
|
||
// Re-acquire lock and update batch to use REQ+EOSE
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_for_fallback) {
|
||
if let Some(batch) =
|
||
batches.iter_mut().find(|b| b.batch_id == batch_id)
|
||
{
|
||
// Switch to REQ+EOSE sync method
|
||
batch.sync_method = SyncMethod::ReqEose;
|
||
// Clear negentropy-specific tracking
|
||
batch.requested_event_ids = None;
|
||
batch.received_event_ids = None;
|
||
// Reset retry count for REQ+EOSE flow
|
||
batch.retry_count = 0;
|
||
// Add new subscriptions to outstanding_subs
|
||
batch.outstanding_subs.extend(new_sub_ids.clone());
|
||
|
||
tracing::info!(
|
||
relay = %relay_url_for_fallback,
|
||
batch_id = batch_id,
|
||
fallback_subs = new_sub_ids.len(),
|
||
missing_events = missing_count,
|
||
"Switched batch to REQ+EOSE fallback, waiting for EOSE"
|
||
);
|
||
}
|
||
}
|
||
// Early return - batch not complete yet, waiting for REQ+EOSE EOSE
|
||
return;
|
||
} else {
|
||
// No fallback subscriptions could be created (for
|
||
// announcement batches there is no semantic metadata
|
||
// at all). Finalize the batch as failed but keep the
|
||
// missing IDs represented so the maintenance timer
|
||
// retries them with bounded backoff instead of
|
||
// forgetting them until the next daily sync.
|
||
let relay_already_degraded = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.get(&relay_url_for_fallback)
|
||
.map(|state| state.historic_sync_had_failures)
|
||
.unwrap_or(false)
|
||
};
|
||
let register_outcome =
|
||
self.missing_event_recovery.lock().unwrap().register(
|
||
&relay_url_for_fallback,
|
||
batch_id,
|
||
missing.iter().copied(),
|
||
relay_already_degraded,
|
||
Instant::now(),
|
||
);
|
||
tracing::error!(
|
||
relay = %relay_url_for_fallback,
|
||
batch_id = batch_id,
|
||
missing_count = missing_count,
|
||
pending_recovery = register_outcome.pending_total,
|
||
"Failed to create REQ+EOSE fallback subscriptions - completing batch with partial results; bounded missing-event recovery scheduled"
|
||
);
|
||
|
||
// Re-acquire lock to extract the batch
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_for_fallback) {
|
||
if let Some(idx) =
|
||
batches.iter().position(|b| b.batch_id == batch_id)
|
||
{
|
||
let mut completed_batch = batches.remove(idx);
|
||
completed_batch.failed = true; // Mark as failed
|
||
if batches.is_empty() {
|
||
pending.remove(&relay_url_for_fallback);
|
||
}
|
||
drop(pending);
|
||
self.confirm_batch(&relay_url_for_fallback, completed_batch)
|
||
.await;
|
||
}
|
||
}
|
||
return;
|
||
}
|
||
}
|
||
|
||
if retry_count > 0 && received_count == 0 {
|
||
let relay_url_for_recovery = relay_url.to_string();
|
||
let batch_id = batch.batch_id;
|
||
let missing_count = missing.len();
|
||
|
||
drop(pending);
|
||
|
||
let relay_already_degraded = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.get(&relay_url_for_recovery)
|
||
.map(|state| state.historic_sync_had_failures)
|
||
.unwrap_or(false)
|
||
};
|
||
let register_outcome =
|
||
self.missing_event_recovery.lock().unwrap().register(
|
||
&relay_url_for_recovery,
|
||
batch_id,
|
||
missing.iter().copied(),
|
||
relay_already_degraded,
|
||
Instant::now(),
|
||
);
|
||
tracing::warn!(
|
||
relay = %relay_url_for_recovery,
|
||
batch_id = batch_id,
|
||
retry_count = retry_count,
|
||
missing_count = missing_count,
|
||
pending_recovery = register_outcome.pending_total,
|
||
"Negentropy residual retry made no progress; retaining NIP-77 and scheduling bounded missing-event recovery"
|
||
);
|
||
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_for_recovery) {
|
||
if let Some(idx) = batches.iter().position(|b| b.batch_id == batch_id) {
|
||
let mut completed_batch = batches.remove(idx);
|
||
completed_batch.failed = true;
|
||
if batches.is_empty() {
|
||
pending.remove(&relay_url_for_recovery);
|
||
}
|
||
drop(pending);
|
||
self.confirm_batch(&relay_url_for_recovery, completed_batch)
|
||
.await;
|
||
}
|
||
}
|
||
return;
|
||
}
|
||
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
batch_id = batch.batch_id,
|
||
retry_count = retry_count,
|
||
requested_count = requested_count,
|
||
received_count = received_count,
|
||
missing_count = missing.len(),
|
||
missing_ids_sample = ?missing.iter().take(5).map(|id| id.to_hex()).collect::<Vec<_>>(),
|
||
"Negentropy sync incomplete - relay returned fewer events than requested. \
|
||
This may indicate a relay limit on ID-based queries. \
|
||
Retrying missing events."
|
||
);
|
||
|
||
// Create retry subscription for missing events
|
||
// Chunk by 300 to avoid overly large filters
|
||
let relay_url_for_retry = relay_url.to_string();
|
||
let batch_id = batch.batch_id;
|
||
|
||
// Drop the lock before async operations
|
||
drop(pending);
|
||
|
||
// Create new subscriptions for missing events
|
||
let retry_subscriptions: Vec<_> = missing
|
||
.chunks(300)
|
||
.map(|chunk| {
|
||
(
|
||
SubscriptionId::generate(),
|
||
Filter::new().ids(chunk.iter().copied()),
|
||
)
|
||
})
|
||
.collect();
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batch) =
|
||
pending.get_mut(&relay_url_for_retry).and_then(|batches| {
|
||
batches.iter_mut().find(|batch| batch.batch_id == batch_id)
|
||
})
|
||
{
|
||
register_negentropy_hydration_attempt(
|
||
batch,
|
||
retry_subscriptions
|
||
.iter()
|
||
.map(|(subscription_id, _)| subscription_id.clone()),
|
||
missing.iter().copied(),
|
||
true,
|
||
);
|
||
}
|
||
}
|
||
|
||
let mut successful_subscriptions = 0usize;
|
||
for (subscription_id, filter) in &retry_subscriptions {
|
||
let result = if let Some(conn) = self.connections.get(&relay_url_for_retry)
|
||
{
|
||
conn.subscribe_filter_with_id(
|
||
filter.clone(),
|
||
TransientRequestClass::NegentropyRetry,
|
||
subscription_id.clone(),
|
||
)
|
||
.await
|
||
} else {
|
||
Err("Relay connection disappeared before hydration retry".to_string())
|
||
};
|
||
match result {
|
||
Ok(_) => successful_subscriptions += 1,
|
||
Err(e) => {
|
||
tracing::error!(
|
||
relay = %relay_url_for_retry,
|
||
batch_id = batch_id,
|
||
error = %e,
|
||
"Failed to create retry subscription for missing events"
|
||
);
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batch) =
|
||
pending.get_mut(&relay_url_for_retry).and_then(|batches| {
|
||
batches.iter_mut().find(|batch| batch.batch_id == batch_id)
|
||
})
|
||
{
|
||
batch.outstanding_subs.remove(subscription_id);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
if successful_subscriptions > 0 {
|
||
// Immediate EOSEs may have drained every successful
|
||
// subscription before the final failed send unwound.
|
||
// Complete that otherwise signal-less edge here.
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
let completed_batch = take_drained_batch_as_failed(
|
||
&mut pending,
|
||
&relay_url_for_retry,
|
||
batch_id,
|
||
);
|
||
let retry_attempt = pending
|
||
.get(&relay_url_for_retry)
|
||
.and_then(|batches| {
|
||
batches.iter().find(|batch| batch.batch_id == batch_id)
|
||
})
|
||
.map(|batch| batch.retry_count);
|
||
drop(pending);
|
||
if let Some(batch) = completed_batch {
|
||
self.confirm_batch(&relay_url_for_retry, batch).await;
|
||
}
|
||
tracing::info!(
|
||
relay = %relay_url_for_retry,
|
||
batch_id = batch_id,
|
||
retry_subs = successful_subscriptions,
|
||
missing_events = missing.len(),
|
||
retry_attempt = retry_attempt.unwrap_or(retry_count + 1),
|
||
"Created retry subscriptions for missing negentropy events"
|
||
);
|
||
// Early return - batch not complete yet, waiting for retry EOSE
|
||
return;
|
||
} else {
|
||
// Failed to create retry subscriptions. Finalize the
|
||
// batch as failed (an incomplete batch must not be
|
||
// reported as successfully complete) and keep the
|
||
// missing IDs represented for bounded recovery.
|
||
let relay_already_degraded = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.get(&relay_url_for_retry)
|
||
.map(|state| state.historic_sync_had_failures)
|
||
.unwrap_or(false)
|
||
};
|
||
let register_outcome =
|
||
self.missing_event_recovery.lock().unwrap().register(
|
||
&relay_url_for_retry,
|
||
batch_id,
|
||
missing.iter().copied(),
|
||
relay_already_degraded,
|
||
Instant::now(),
|
||
);
|
||
tracing::error!(
|
||
relay = %relay_url_for_retry,
|
||
batch_id = batch_id,
|
||
missing_count = missing.len(),
|
||
pending_recovery = register_outcome.pending_total,
|
||
"Failed to retry missing events - confirming batch with partial results; bounded missing-event recovery scheduled"
|
||
);
|
||
|
||
// Re-acquire lock to extract the batch
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_for_retry) {
|
||
if let Some(idx) = batches.iter().position(|b| b.batch_id == batch_id) {
|
||
let mut completed_batch = batches.remove(idx);
|
||
completed_batch.failed = true;
|
||
if batches.is_empty() {
|
||
pending.remove(&relay_url_for_retry);
|
||
}
|
||
drop(pending);
|
||
self.confirm_batch(&relay_url_for_retry, completed_batch)
|
||
.await;
|
||
}
|
||
}
|
||
return;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// 3. Batch complete - extract and remove
|
||
let completed_batch = batches.remove(batch_idx);
|
||
|
||
// Clean up empty relay entry
|
||
if batches.is_empty() {
|
||
pending.remove(relay_url);
|
||
}
|
||
|
||
// Drop the pending lock before confirm_batch
|
||
drop(pending);
|
||
|
||
// 4. Confirm the batch (moves items to RelayState)
|
||
self.confirm_batch(relay_url, completed_batch).await;
|
||
}
|
||
|
||
#[allow(clippy::too_many_arguments)]
|
||
fn spawn_deferred_pagination(
|
||
connection: RelayConnection,
|
||
health_tracker: Arc<RelayHealthTracker>,
|
||
pending_sync_index: PendingSyncIndex,
|
||
relay_url: String,
|
||
batch_id: u64,
|
||
deferred_sub_id: SubscriptionId,
|
||
next_page: PaginationState,
|
||
) {
|
||
tokio::spawn(async move {
|
||
let next_filters = next_page.filters();
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
filter_count = next_filters.len(),
|
||
"Rate limited during historic pagination; deferring the exact next page without blocking the sync actor"
|
||
);
|
||
|
||
loop {
|
||
if let Some(remaining) = health_tracker.get_remaining_backoff(&relay_url) {
|
||
tokio::time::sleep(remaining + Duration::from_millis(10)).await;
|
||
continue;
|
||
}
|
||
|
||
// Hold only the pending-index lock while subscribing. This
|
||
// prevents a very fast EOSE from reaching the actor before its
|
||
// new subscription ID is registered, without blocking the
|
||
// SyncManager mutex or the purgatory timer.
|
||
let mut pending = pending_sync_index.write().await;
|
||
let Some(batch) = pending.get_mut(&relay_url).and_then(|batches| {
|
||
batches.iter_mut().find(|batch| batch.batch_id == batch_id)
|
||
}) else {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
"Deferred pagination batch no longer exists"
|
||
);
|
||
return;
|
||
};
|
||
if !batch.outstanding_subs.contains(&deferred_sub_id) {
|
||
return;
|
||
}
|
||
|
||
match connection
|
||
.subscribe_filters(next_filters.clone(), next_page.request_class())
|
||
.await
|
||
{
|
||
Ok(new_sub_id) => {
|
||
batch.outstanding_subs.remove(&deferred_sub_id);
|
||
batch.outstanding_subs.insert(new_sub_id.clone());
|
||
batch.pagination_state.insert(new_sub_id.clone(), next_page);
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
new_sub_id = %new_sub_id,
|
||
batch_id,
|
||
"Deferred pagination resumed after rate-limit cooldown"
|
||
);
|
||
return;
|
||
}
|
||
Err(error) => {
|
||
drop(pending);
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
error = %error,
|
||
"Deferred pagination subscription failed; retrying"
|
||
);
|
||
tokio::time::sleep(Duration::from_secs(2)).await;
|
||
}
|
||
}
|
||
}
|
||
});
|
||
}
|
||
|
||
/// Drive bounded recovery of events relays reported during negentropy
|
||
/// reconciliation but failed to deliver on exact-ID fetches.
|
||
///
|
||
/// Runs on the sync maintenance timer. For each relay with pending IDs:
|
||
/// 1. IDs already present locally (live sync, user submission) are
|
||
/// cleared promptly without consuming attempt budget.
|
||
/// 2. When the relay's backoff deadline has passed and it has a live
|
||
/// connection, one bounded exact-ID fetch is spawned. Network I/O runs
|
||
/// outside the sync actor lock, and per-relay single-flight keeps one
|
||
/// persistently incomplete relay from starving other relays or later
|
||
/// batches.
|
||
async fn tick_missing_event_recovery(&mut self) {
|
||
let idle_relays = self.missing_event_recovery.lock().unwrap().idle_relays();
|
||
if idle_relays.is_empty() {
|
||
return;
|
||
}
|
||
|
||
for relay_url in idle_relays {
|
||
let Some(pending) = self
|
||
.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.pending_ids(&relay_url)
|
||
else {
|
||
continue;
|
||
};
|
||
|
||
// 1. Clear IDs satisfied by other means.
|
||
let satisfied: Vec<EventId> = match self
|
||
.database
|
||
.query(Filter::new().ids(pending.iter().copied()))
|
||
.await
|
||
{
|
||
Ok(events) => events.into_iter().map(|event| event.id).collect(),
|
||
Err(_) => Vec::new(),
|
||
};
|
||
if !satisfied.is_empty() {
|
||
let outcome = self
|
||
.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.clear_satisfied(&relay_url, &satisfied);
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
satisfied = satisfied.len(),
|
||
"Pending missing events satisfied by local arrivals"
|
||
);
|
||
if let Some(outcome) = outcome {
|
||
Self::apply_recovery_outcome(
|
||
&relay_url,
|
||
outcome,
|
||
&self.relay_sync_index,
|
||
self.metrics.as_ref(),
|
||
)
|
||
.await;
|
||
continue;
|
||
}
|
||
}
|
||
|
||
// 2. Issue a due attempt over a live connection only.
|
||
let connection_ready = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.get(&relay_url)
|
||
.map(|state| {
|
||
matches!(
|
||
state.connection_status,
|
||
ConnectionStatus::Syncing
|
||
| ConnectionStatus::Connected
|
||
| ConnectionStatus::ConnectedHistoricSyncFailures
|
||
)
|
||
})
|
||
.unwrap_or(false)
|
||
};
|
||
let connection = self.connections.get(&relay_url).cloned();
|
||
let (Some(connection), true) = (connection, connection_ready) else {
|
||
// No usable connection: postpone without consuming the
|
||
// zero-progress budget so an unavailable relay can neither
|
||
// expire its pending IDs nor spin in a tight loop.
|
||
self.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.defer_attempt(&relay_url, Instant::now());
|
||
continue;
|
||
};
|
||
|
||
let Some(attempt) = self
|
||
.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.begin_attempt(&relay_url, Instant::now())
|
||
else {
|
||
continue;
|
||
};
|
||
|
||
tokio::spawn(Self::run_missing_event_recovery_attempt(
|
||
relay_url,
|
||
connection,
|
||
attempt,
|
||
Arc::clone(&self.database),
|
||
self.write_policy.clone(),
|
||
self.local_relay.clone(),
|
||
Arc::clone(&self.rejected_events_index),
|
||
Arc::clone(&self.missing_event_recovery),
|
||
self.relay_sync_index.clone(),
|
||
self.metrics.clone(),
|
||
));
|
||
}
|
||
}
|
||
|
||
/// One bounded exact-ID recovery fetch against a single relay.
|
||
///
|
||
/// Runs outside the sync actor lock. Every event the relay returns is
|
||
/// passed through the normal write policy. An ID counts as recovered only
|
||
/// once it has a durable terminal outcome; transient persistence and
|
||
/// dependency-sensitive policy failures remain pending.
|
||
#[allow(clippy::too_many_arguments)]
|
||
async fn run_missing_event_recovery_attempt(
|
||
relay_url: String,
|
||
connection: RelayConnection,
|
||
attempt: missing_events::RecoveryAttempt,
|
||
database: SharedDatabase,
|
||
write_policy: Nip34WritePolicy,
|
||
local_relay: LocalRelay,
|
||
rejected_events_index: Arc<RejectedEventsIndex>,
|
||
missing_event_recovery: Arc<std::sync::Mutex<missing_events::MissingEventRecoveryIndex>>,
|
||
relay_index: RelaySyncIndex,
|
||
metrics: Option<SyncMetrics>,
|
||
) {
|
||
let requested: HashSet<EventId> = attempt.ids.iter().copied().collect();
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
attempt = attempt.attempt_number,
|
||
requested = requested.len(),
|
||
"Retrying events missing from an incomplete historic sync batch"
|
||
);
|
||
if let Some(metrics) = metrics.as_ref() {
|
||
metrics.record_hydration_events(&relay_url, "recovery", "requested", requested.len());
|
||
}
|
||
|
||
let events = match connection
|
||
.fetch_events(
|
||
Filter::new().ids(attempt.ids.iter().copied()),
|
||
Duration::from_secs(10),
|
||
)
|
||
.await
|
||
{
|
||
Ok(events) => events,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
attempt = attempt.attempt_number,
|
||
requested = requested.len(),
|
||
error = %error,
|
||
"Missing-event recovery fetch failed"
|
||
);
|
||
Vec::new()
|
||
}
|
||
};
|
||
|
||
let mut delivered = HashSet::new();
|
||
let mut recovered = HashSet::new();
|
||
for event in events {
|
||
if !requested.contains(&event.id) {
|
||
continue;
|
||
}
|
||
if !delivered.insert(event.id) {
|
||
continue;
|
||
}
|
||
if let Some(metrics) = metrics.as_ref() {
|
||
metrics.record_hydration_events(&relay_url, "recovery", "delivered", 1);
|
||
}
|
||
// Permanent cached rejections account for the requested ID.
|
||
// Dependency-sensitive entries remain pending until their normal
|
||
// re-processing machinery observes the missing accepted event.
|
||
if rejected_events_index.contains(&event.id) {
|
||
let dependency_pending = rejected_events_index.is_dependency_pending(&event.id);
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
event_id = %event.id,
|
||
dependency_pending,
|
||
"Recovered missing event already tracked as rejected, skipping re-processing"
|
||
);
|
||
if let Some(metrics) = metrics.as_ref() {
|
||
metrics.record_hydration_events(&relay_url, "recovery", "rejected_cached", 1);
|
||
}
|
||
if !dependency_pending {
|
||
recovered.insert(event.id);
|
||
}
|
||
continue;
|
||
}
|
||
let result = Self::process_event_static(
|
||
&event,
|
||
&relay_url,
|
||
&database,
|
||
&write_policy,
|
||
&local_relay,
|
||
&rejected_events_index,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
if let Some(metrics) = metrics.as_ref() {
|
||
metrics.record_hydration_events(
|
||
&relay_url,
|
||
"recovery",
|
||
result.hydration_outcome(),
|
||
1,
|
||
);
|
||
if result == ProcessResult::Saved {
|
||
metrics.record_synced_event();
|
||
}
|
||
}
|
||
if result.is_rejected() || result == ProcessResult::PersistenceError {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
event_id = %event.id,
|
||
outcome = result.hydration_outcome(),
|
||
terminal = result.is_terminally_accounted(),
|
||
"Recovered missing event was not persisted"
|
||
);
|
||
}
|
||
if result.is_terminally_accounted() {
|
||
recovered.insert(event.id);
|
||
}
|
||
}
|
||
|
||
if let Some(metrics) = metrics.as_ref() {
|
||
metrics.record_hydration_events(
|
||
&relay_url,
|
||
"recovery",
|
||
"not_delivered",
|
||
requested.len().saturating_sub(delivered.len()),
|
||
);
|
||
}
|
||
|
||
let outcome = missing_event_recovery.lock().unwrap().complete_attempt(
|
||
&relay_url,
|
||
&recovered,
|
||
Instant::now(),
|
||
);
|
||
let Some(outcome) = outcome else {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Recovery attempt finished after the relay's pending work was reset"
|
||
);
|
||
return;
|
||
};
|
||
Self::apply_recovery_outcome(&relay_url, outcome, &relay_index, metrics.as_ref()).await;
|
||
}
|
||
|
||
/// Log a recovery outcome and, on full recovery, restore relay health.
|
||
///
|
||
/// A relay is only promoted from `ConnectedHistoricSyncFailures` back to
|
||
/// `Connected` when every registered missing ID was recovered and no
|
||
/// unrelated batch failure was observed for the relay.
|
||
async fn apply_recovery_outcome(
|
||
relay_url: &str,
|
||
outcome: missing_events::AttemptOutcome,
|
||
relay_index: &RelaySyncIndex,
|
||
metrics: Option<&SyncMetrics>,
|
||
) {
|
||
use missing_events::AttemptOutcome;
|
||
|
||
match outcome {
|
||
AttemptOutcome::FullyRecovered {
|
||
recovered,
|
||
total_recovered,
|
||
can_restore_health,
|
||
} => {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
recovered,
|
||
total_recovered,
|
||
can_restore_health,
|
||
"All events missing from historic sync fully recovered"
|
||
);
|
||
if !can_restore_health {
|
||
return;
|
||
}
|
||
let mut index = relay_index.write().await;
|
||
if let Some(state) = index.get_mut(relay_url) {
|
||
state.historic_sync_had_failures = false;
|
||
if state.connection_status == ConnectionStatus::ConnectedHistoricSyncFailures {
|
||
state.connection_status = ConnectionStatus::Connected;
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"Historic sync failures resolved - relay promoted to Connected"
|
||
);
|
||
if let Some(metrics) = metrics {
|
||
metrics
|
||
.record_connection_status(relay_url, ConnectionStatus::Connected);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
AttemptOutcome::PartiallyRecovered {
|
||
recovered,
|
||
remaining,
|
||
next_attempt_in,
|
||
} => {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
recovered,
|
||
remaining,
|
||
next_attempt_in_secs = next_attempt_in.as_secs_f64(),
|
||
"Partially recovered events missing from historic sync - retry scheduled"
|
||
);
|
||
}
|
||
AttemptOutcome::RetryScheduled {
|
||
attempt,
|
||
remaining,
|
||
next_attempt_in,
|
||
} => {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
attempt,
|
||
remaining,
|
||
next_attempt_in_secs = next_attempt_in.as_secs_f64(),
|
||
"No missing events recovered - retry scheduled with backoff"
|
||
);
|
||
}
|
||
AttemptOutcome::Expired {
|
||
remaining,
|
||
attempts,
|
||
} => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
remaining,
|
||
attempts,
|
||
"Missing-event recovery expired by policy after repeated zero-progress attempts - relay remains ConnectedHistoricSyncFailures until daily sync"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Confirm a completed batch by moving items to RelayState
|
||
///
|
||
/// This method is used by both sync paths (REQ+EOSE and Negentropy) to
|
||
/// move repos and root_events from pending to confirmed state. This unified
|
||
/// flow ensures consistent state tracking regardless of sync method.
|
||
///
|
||
/// For generic filter batches (identified by empty repos and root_events),
|
||
/// this sets the announcements_synced flag to enable incremental sync on reconnect.
|
||
///
|
||
/// # Arguments
|
||
/// * `relay_url` - The relay URL the batch belongs to
|
||
/// * `batch` - The completed batch to confirm
|
||
async fn confirm_batch(&mut self, relay_url: &str, batch: PendingBatch) {
|
||
let batch_id = batch.batch_id;
|
||
let full_repos_count = batch.items.repos.len();
|
||
let state_only_repos_count = batch.items.state_only_repos.len();
|
||
let events_count = batch.items.root_events.len();
|
||
let sync_method = batch.sync_method;
|
||
let is_generic_filter = batch.purpose == PendingBatchPurpose::Announcements;
|
||
|
||
if batch.purpose == PendingBatchPurpose::Descendants {
|
||
let succeeded = !batch.failed;
|
||
if let Some(rotation) = self.descendant_sync_rotations.get_mut(relay_url) {
|
||
rotation.mark_completed(batch_id, succeeded);
|
||
}
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
succeeded,
|
||
"Descendant historic query reached terminal batch state"
|
||
);
|
||
}
|
||
|
||
let mut relay_index = self.relay_sync_index.write().await;
|
||
|
||
if let Some(state) = relay_index.get_mut(relay_url) {
|
||
// A completed Full batch upgrades any earlier StateOnly
|
||
// confirmation. A late StateOnly completion must not downgrade a
|
||
// repository whose Full batch already completed.
|
||
state
|
||
.state_only_repos
|
||
.retain(|repo| !batch.items.repos.contains(repo));
|
||
state.repos.extend(batch.items.repos);
|
||
state.state_only_repos.extend(
|
||
batch
|
||
.items
|
||
.state_only_repos
|
||
.into_iter()
|
||
.filter(|repo| !state.repos.contains(repo)),
|
||
);
|
||
// Move root_events to confirmed
|
||
state.root_events.extend(batch.items.root_events.clone());
|
||
|
||
// Set announcements_synced flag for generic filter batches
|
||
if is_generic_filter {
|
||
state.announcements_synced = true;
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
sync_method = ?sync_method,
|
||
"Generic filter (announcements) historic sync complete - announcements_synced set to true"
|
||
);
|
||
|
||
// Provide helpful feedback for bootstrap relay
|
||
if state.is_bootstrap {
|
||
let announcement_count = events_count;
|
||
if announcement_count == 0 {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
domain = %self.config.domain,
|
||
"Bootstrap sync found no announcements for domain - verify domain is correct or try different bootstrap relay"
|
||
);
|
||
} else {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
domain = %self.config.domain,
|
||
announcement_count,
|
||
"Bootstrap sync discovered announcements for domain"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// Track if this batch failed (for ConnectedDegraded transition)
|
||
if batch.failed && batch.purpose != PendingBatchPurpose::Descendants {
|
||
state.historic_sync_had_failures = true;
|
||
// Failures unrelated to a relay's pending missing-event
|
||
// recovery mean full recovery must not restore its health.
|
||
self.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.note_failed_batch(relay_url, batch_id);
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
"Batch failed - will transition to ConnectedHistoricSyncFailures instead of Connected"
|
||
);
|
||
}
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
purpose = ?batch.purpose,
|
||
sync_method = ?sync_method,
|
||
full_repos_confirmed = full_repos_count,
|
||
state_only_repos_confirmed = state_only_repos_count,
|
||
root_events_confirmed = events_count,
|
||
root_events_sample = ?batch.items.root_events.iter().take(LOG_COLLECTION_SAMPLE_SIZE).map(|id| id.to_hex()).collect::<Vec<_>>(),
|
||
total_full_repos = state.repos.len(),
|
||
total_state_only_repos = state.state_only_repos.len(),
|
||
total_root_events = state.root_events.len(),
|
||
all_root_events_sample = ?state.root_events.iter().take(LOG_COLLECTION_SAMPLE_SIZE).map(|id| id.to_hex()).collect::<Vec<_>>(),
|
||
is_generic_filter = is_generic_filter,
|
||
announcements_synced = state.announcements_synced,
|
||
had_failures = state.historic_sync_had_failures,
|
||
"Batch confirmed - items moved from pending to confirmed"
|
||
);
|
||
} else {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
"Batch completed but no RelayState found for relay"
|
||
);
|
||
}
|
||
|
||
// Release lock before checking if historic sync is complete
|
||
drop(relay_index);
|
||
|
||
// Batch completion already runs inside the sync actor and all index
|
||
// guards are released. Process a ready deferral directly instead of
|
||
// retaining self-addressed wakeups in another queue.
|
||
if self.deferred_consolidations.contains(relay_url) {
|
||
// Consolidation can synchronously complete an empty historic
|
||
// batch, so erase that finite recursive future from the type.
|
||
Box::pin(self.process_deferred_consolidation(relay_url)).await;
|
||
}
|
||
|
||
// Spawn background task to check if historic sync is complete
|
||
// This avoids blocking the confirm_batch flow for 6 seconds
|
||
let relay_url = relay_url.to_string();
|
||
let pending_index = self.pending_sync_index.clone();
|
||
let relay_index = self.relay_sync_index.clone();
|
||
let metrics = self.metrics.clone();
|
||
|
||
tokio::spawn(async move {
|
||
Self::check_and_complete_historic_sync_impl(
|
||
&relay_url,
|
||
pending_index,
|
||
relay_index,
|
||
metrics,
|
||
)
|
||
.await;
|
||
});
|
||
}
|
||
|
||
/// Check if historic sync is complete and transition to Connected status
|
||
///
|
||
/// This method uses a double-check pattern to avoid race conditions with
|
||
/// the self-subscriber's batching window. The sequence is:
|
||
///
|
||
/// 1. First check: Are there pending batches?
|
||
/// 2. Wait for batch window + buffer (6 seconds)
|
||
/// 3. Second check: Are there still no pending batches?
|
||
/// 4. If still no pending batches, transition to Connected
|
||
///
|
||
/// This ensures that events received just before the first check have time
|
||
/// to be batched and create Layer 2/3 filters before we mark sync complete.
|
||
///
|
||
/// The 6-second delay is based on:
|
||
/// - Self-subscriber batch window: 5 seconds (200ms when `NGIT_TEST=1`)
|
||
/// - Buffer for processing: 1 second
|
||
///
|
||
/// Called after each batch is confirmed to detect completion.
|
||
/// Spawned as a background task to avoid blocking the confirm_batch flow.
|
||
async fn check_and_complete_historic_sync_impl(
|
||
relay_url: &str,
|
||
pending_index: PendingSyncIndex,
|
||
relay_index: RelaySyncIndex,
|
||
metrics: Option<SyncMetrics>,
|
||
) {
|
||
// First check: Are there any pending batches?
|
||
let has_pending = {
|
||
let pending = pending_index.read().await;
|
||
pending
|
||
.get(relay_url)
|
||
.is_some_and(|batches| !batches.is_empty())
|
||
};
|
||
|
||
if has_pending {
|
||
// Still syncing, don't transition yet
|
||
return;
|
||
}
|
||
|
||
// Wait for self-subscriber batch window + buffer to catch any in-flight events
|
||
// that might create new Layer 2/3 filters
|
||
tokio::time::sleep(Duration::from_millis(6000)).await;
|
||
|
||
// Second check: Are there still no pending batches?
|
||
let has_pending = {
|
||
let pending = pending_index.read().await;
|
||
pending
|
||
.get(relay_url)
|
||
.is_some_and(|batches| !batches.is_empty())
|
||
};
|
||
|
||
if has_pending {
|
||
// New batches appeared during the wait - still syncing
|
||
return;
|
||
}
|
||
|
||
// No pending batches after waiting - safe to transition to Connected or ConnectedDegraded
|
||
let mut relay_index_guard = relay_index.write().await;
|
||
if let Some(state) = relay_index_guard.get_mut(relay_url) {
|
||
if state.connection_status == ConnectionStatus::Syncing {
|
||
// Check if any batches failed during historic sync
|
||
let new_status = if state.historic_sync_had_failures {
|
||
ConnectionStatus::ConnectedHistoricSyncFailures
|
||
} else {
|
||
ConnectionStatus::Connected
|
||
};
|
||
|
||
state.connection_status = new_status;
|
||
state.historic_sync_completed = true;
|
||
state.historic_sync_completed_at = Some(Timestamp::now());
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
full_repos_synced = state.repos.len(),
|
||
state_only_repos_synced = state.state_only_repos.len(),
|
||
root_events_synced = state.root_events.len(),
|
||
had_failures = state.historic_sync_had_failures,
|
||
status = ?new_status,
|
||
"Historic sync complete - transitioned to {} status",
|
||
if state.historic_sync_had_failures { "ConnectedHistoricSyncFailures" } else { "Connected" }
|
||
);
|
||
|
||
// Update metrics
|
||
if let Some(ref metrics) = metrics {
|
||
metrics.record_connection_status(relay_url, new_status);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Perform a daily sync for a specific relay
|
||
///
|
||
/// This method:
|
||
/// - Unsubscribes from all current subscriptions on the relay
|
||
/// - Clears pending batches for this relay
|
||
/// - Clears sync state (repos and root_events) in RelayState
|
||
/// - Recomputes actions to re-discover all repos/events
|
||
///
|
||
/// This is triggered by the daily timer to detect state drift over time.
|
||
async fn daily_sync(&mut self, relay_url: &str) {
|
||
tracing::info!(relay = %relay_url, "Starting daily sync");
|
||
self.cancel_deferred_consolidation(relay_url, "daily sync reset");
|
||
let _ = self
|
||
.close_descendant_live_coverage(relay_url, "daily sync reset")
|
||
.await;
|
||
self.descendant_sync_rotations.remove(relay_url);
|
||
|
||
// Get connection
|
||
let connection = match self.connections.get(relay_url) {
|
||
Some(conn) => conn,
|
||
None => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
"No connection for relay, skipping daily sync"
|
||
);
|
||
return;
|
||
}
|
||
};
|
||
|
||
// Unsubscribe all current subscriptions
|
||
connection.unsubscribe_all().await;
|
||
|
||
// Daily sync re-discovers everything from scratch, superseding any
|
||
// pending missing-event recovery for this relay.
|
||
let dropped_recovery_ids = self
|
||
.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.clear_relay(relay_url);
|
||
if dropped_recovery_ids > 0 {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
dropped_missing_ids = dropped_recovery_ids,
|
||
"Daily sync reset pending missing-event recovery"
|
||
);
|
||
}
|
||
|
||
// Clear pending batches for this relay
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
pending.remove(relay_url);
|
||
}
|
||
|
||
// Get relay state and clear sync state (repos and root_events)
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
if let Some(state) = index.get_mut(relay_url) {
|
||
let repos_cleared = state.repos.len() + state.state_only_repos.len();
|
||
let events_cleared = state.root_events.len();
|
||
state.clear_sync_state();
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
repos_cleared = repos_cleared,
|
||
events_cleared = events_cleared,
|
||
"Cleared sync state for daily sync"
|
||
);
|
||
}
|
||
}
|
||
|
||
// maybe we just run start fresh with a daily flag? make sture so start layer 1 filters
|
||
self.fresh_start(relay_url).await;
|
||
|
||
// if let Some(ref metrics) = self.metrics {
|
||
// metrics.record_event(event_source::DAILY);
|
||
// }
|
||
|
||
// tracing::info!(relay = %relay_url, "Daily sync complete");
|
||
}
|
||
|
||
/// Run the sync manager
|
||
///
|
||
/// Coordinates all sync components:
|
||
/// 1. Spawns self-subscriber to monitor own relay for announcements
|
||
/// 2. Spawns daily timer for periodic fresh syncs
|
||
/// 3. Connects to bootstrap relay if configured
|
||
/// 4. Handles relay actions from self-subscriber
|
||
/// 5. Handles disconnect/EOSE notifications and connection-worker results
|
||
pub async fn run(mut self) {
|
||
use tokio::sync::mpsc;
|
||
|
||
tracing::info!(
|
||
bootstrap_relay = ?self.bootstrap_relay_url,
|
||
service_domain = %self.service_domain,
|
||
"SyncManager starting"
|
||
);
|
||
|
||
// 1. Create action channel for self-subscriber -> manager communication
|
||
let (action_tx, mut action_rx) = mpsc::channel::<AddFilters>(100);
|
||
|
||
// 2. Create disconnect channel for spawned tasks -> manager communication
|
||
let (disconnect_tx, mut disconnect_rx) = mpsc::channel::<DisconnectNotification>(100);
|
||
|
||
// 3. Create EOSE channel for spawned tasks -> manager communication
|
||
// The independent terminal-control listener has already released
|
||
// transient resources. These actor notifications may therefore apply
|
||
// bounded backpressure without stranding subscription-ledger slots.
|
||
let (eose_tx, mut eose_rx) = lifecycle_notification_channel::<EoseNotification>();
|
||
let (subscription_closed_tx, mut subscription_closed_rx) =
|
||
lifecycle_notification_channel::<SubscriptionClosedNotification>();
|
||
|
||
// Connection workers never mutate manager state. At most the global
|
||
// connection-attempt cap can complete while the actor is busy.
|
||
let (connect_attempt_result_tx, mut connect_attempt_result_rx) =
|
||
mpsc::channel::<ConnectAttemptResult>(MAX_CONCURRENT_CONNECT_ATTEMPTS);
|
||
|
||
let (nip65_discovery_result_tx, mut nip65_discovery_result_rx) =
|
||
mpsc::channel::<Nip65DiscoveryResult>(1);
|
||
let (mailbox_probe_result_tx, mut mailbox_probe_result_rx) =
|
||
mpsc::channel::<MailboxProbeResult>(1);
|
||
|
||
// 4b. Create shutdown broadcast channel for graceful shutdown
|
||
let (shutdown_tx, _shutdown_rx) = broadcast::channel(1);
|
||
|
||
// 5. Spawn self-subscriber with shutdown receiver
|
||
let self_subscriber = SelfSubscriber::new(
|
||
format!("ws://{}", self.config.bind_address),
|
||
self.service_domain.clone(),
|
||
Arc::clone(&self.repo_sync_index),
|
||
Arc::clone(&self.root_candidate_index),
|
||
action_tx,
|
||
self.database.clone(),
|
||
);
|
||
let subscriber_shutdown = shutdown_tx.subscribe();
|
||
tokio::spawn(async move { self_subscriber.run(Some(subscriber_shutdown)).await });
|
||
|
||
// 5b. Store channel senders for use by handlers
|
||
self.disconnect_tx = Some(disconnect_tx.clone());
|
||
self.eose_tx = Some(eose_tx.clone());
|
||
self.subscription_closed_tx = Some(subscription_closed_tx);
|
||
self.connect_attempt_result_tx = Some(connect_attempt_result_tx);
|
||
self.nip65_discovery_result_tx = Some(nip65_discovery_result_tx);
|
||
self.mailbox_probe_result_tx = Some(mailbox_probe_result_tx);
|
||
self.shutdown_tx = Some(shutdown_tx.clone());
|
||
|
||
// 6. Connect to bootstrap relay if configured
|
||
if let Some(ref bootstrap_url) = self.bootstrap_relay_url.clone() {
|
||
match canonical_relay_key(bootstrap_url) {
|
||
Ok(relay_url) => {
|
||
if self.register_relay(relay_url.clone(), true, false).await {
|
||
self.schedule_connect_relay(&relay_url).await;
|
||
}
|
||
}
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %bootstrap_url,
|
||
error = %error,
|
||
"Rejecting invalid bootstrap relay target"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// 7. Wrap self in Arc<Mutex> for sharing with timer task
|
||
let sync_manager = Arc::new(Mutex::new(self));
|
||
|
||
// 8. Spawn daily timer task with shutdown receiver
|
||
let timer_manager = Arc::clone(&sync_manager);
|
||
let timer_shutdown = shutdown_tx.subscribe();
|
||
tokio::spawn(async move {
|
||
run_daily_timer(timer_manager, timer_shutdown).await;
|
||
});
|
||
|
||
// 9. Spawn health and metrics checker task with shutdown receiver
|
||
// This combines disconnect checking, rate limit recovery, and metrics updates
|
||
let checker_manager = Arc::clone(&sync_manager);
|
||
let checker_shutdown = shutdown_tx.subscribe();
|
||
tokio::spawn(async move {
|
||
run_health_and_metrics_checker(checker_manager, checker_shutdown).await;
|
||
});
|
||
|
||
// 10. Spawn rejected events index cleanup task
|
||
// Hot cache cleanup every 60s, cold index cleanup daily
|
||
let cleanup_manager = Arc::clone(&sync_manager);
|
||
let cleanup_shutdown = shutdown_tx.subscribe();
|
||
tokio::spawn(async move {
|
||
run_rejected_index_cleanup(cleanup_manager, cleanup_shutdown).await;
|
||
});
|
||
|
||
// 11. Spawn purgatory announcement sync timer (every 5s)
|
||
// Ensures purgatory announcements (including user-submitted ones that never
|
||
// touch the DB) are registered in repo_sync_index as StateOnly so that
|
||
// state event subscriptions are established on their listed relay URLs.
|
||
let purgatory_sync_manager = Arc::clone(&sync_manager);
|
||
let purgatory_sync_shutdown = shutdown_tx.subscribe();
|
||
tokio::spawn(async move {
|
||
run_purgatory_announcement_sync(purgatory_sync_manager, purgatory_sync_shutdown).await;
|
||
});
|
||
|
||
// 12. Main loop - handle actions, lifecycle notifications, and worker results
|
||
loop {
|
||
// Wait for an event without holding the lock
|
||
tokio::select! {
|
||
// Lifecycle completions release resources and unblock later
|
||
// work. Drain them ahead of a continuously-ready discovery
|
||
// queue so retirement cannot remain half-finished under load.
|
||
biased;
|
||
disconnect = disconnect_rx.recv() => {
|
||
match disconnect {
|
||
Some(notification) => {
|
||
// Acquire lock to process disconnect
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.handle_disconnect(¬ification.relay_url).await;
|
||
}
|
||
None => {
|
||
// All disconnect senders dropped - unlikely but handle gracefully
|
||
tracing::debug!("Disconnect channel closed");
|
||
}
|
||
}
|
||
}
|
||
eose = eose_rx.recv() => {
|
||
match eose {
|
||
Some(notification) => {
|
||
// Acquire lock to process EOSE
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.handle_eose(¬ification.relay_url, notification.sub_id).await;
|
||
}
|
||
None => {
|
||
// All EOSE senders dropped - unlikely but handle gracefully
|
||
tracing::debug!("EOSE channel closed");
|
||
}
|
||
}
|
||
}
|
||
closed = subscription_closed_rx.recv() => {
|
||
if let Some(notification) = closed {
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.handle_subscription_closed(
|
||
¬ification.relay_url,
|
||
notification.subscription_id,
|
||
¬ification.reason,
|
||
notification.generation,
|
||
notification.live_filter_count,
|
||
).await;
|
||
}
|
||
}
|
||
result = mailbox_probe_result_rx.recv() => {
|
||
if let Some(result) = result {
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.handle_mailbox_probe_result(result).await;
|
||
}
|
||
}
|
||
result = connect_attempt_result_rx.recv() => {
|
||
match result {
|
||
Some(result) => {
|
||
// Connection state changes remain serialized through the actor.
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.handle_connect_attempt_result(result).await;
|
||
}
|
||
None => {
|
||
tracing::debug!("Connect attempt result channel closed");
|
||
}
|
||
}
|
||
}
|
||
result = nip65_discovery_result_rx.recv() => {
|
||
if let Some(result) = result {
|
||
let mut manager = sync_manager.lock().await;
|
||
manager.handle_nip65_discovery_result(result).await;
|
||
}
|
||
}
|
||
action = action_rx.recv() => {
|
||
match action {
|
||
Some(add_filters) => {
|
||
// SelfSubscriber actions are dirty-relay signals.
|
||
// Recompute against current pending and confirmed
|
||
// state instead of trusting a queued full-index
|
||
// snapshot that may already be stale.
|
||
let mut manager = sync_manager.lock().await;
|
||
// A dirty relay can contain a newly accepted root
|
||
// or participant. Refresh the local inventory now;
|
||
// the per-author retry deadlines still bound remote
|
||
// NIP-65 queries when only repository state changed.
|
||
manager.nip65_discovery.next_inventory_at = None;
|
||
manager
|
||
.recompute_new_sync_filters_for_relay(&add_filters.relay_url)
|
||
.await;
|
||
manager.schedule_nip65_discovery().await;
|
||
}
|
||
None => break,
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Handle AddFilters action - subscribe to filters on a relay
|
||
///
|
||
/// This method handles all filter additions:
|
||
/// - For new relays: creates entry with Connecting status, spawns connection
|
||
/// - For existing connected relays: subscribes to filters, creates PendingBatch
|
||
/// - For disconnected/connecting relays: returns (will be handled on connection)
|
||
async fn handle_new_sync_filters(&mut self, mut action: AddFilters) {
|
||
action.relay_url = match canonical_relay_key(&action.relay_url) {
|
||
Ok(relay_url) => relay_url,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %action.relay_url,
|
||
error = %error,
|
||
"Rejecting invalid sync relay target"
|
||
);
|
||
return;
|
||
}
|
||
};
|
||
|
||
// The local self-subscriber already observes everything served here.
|
||
// Enforce this at the shared AddFilters boundary too, because
|
||
// purgatory recomputation creates actions without passing through the
|
||
// self-subscriber's own-relay exclusion.
|
||
if is_own_sync_target(&action.relay_url, &self.service_domain) {
|
||
tracing::debug!(
|
||
relay = %action.relay_url,
|
||
service_domain = %self.service_domain,
|
||
"Skipping sync action targeting this relay"
|
||
);
|
||
return;
|
||
}
|
||
|
||
// Targets already rejected by the outbound policy stay in the sync
|
||
// indexes (they come from stored events); skip them quietly instead
|
||
// of re-rejecting on every recomputation.
|
||
if self.rejected_relay_targets.contains(&action.relay_url) {
|
||
tracing::debug!(
|
||
relay = %action.relay_url,
|
||
"Skipping relay target rejected by outbound policy"
|
||
);
|
||
return;
|
||
}
|
||
|
||
// A relay first reached through NIP-65 may later become an ordinary
|
||
// repository target. Promote that existing connection before reading
|
||
// its state so subsequent reconnects and sync work use the full role.
|
||
let promoted_from_discovery = self.nip65_discovery_only_relays.remove(&action.relay_url);
|
||
if promoted_from_discovery {
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
"Promoting NIP-65 discovery connection to repository sync"
|
||
);
|
||
}
|
||
|
||
// Step 1: Check if relay exists in relay_sync_index
|
||
let connection_status = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index.get(&action.relay_url).map(|s| s.connection_status)
|
||
};
|
||
|
||
match connection_status {
|
||
None => {
|
||
// New relay - register and connect
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
full_repos = action.items.repos.len(),
|
||
state_only_repos = action.items.state_only_repos.len(),
|
||
"Registering and connecting to new relay"
|
||
);
|
||
|
||
// Register relay (creates RelayConnection, initializes RelayState, updates metrics)
|
||
if self
|
||
.register_relay(action.relay_url.clone(), false, false)
|
||
.await
|
||
{
|
||
self.schedule_connect_relay(&action.relay_url).await;
|
||
}
|
||
// Connection will trigger handle_connect_or_reconnect which will process items
|
||
return;
|
||
}
|
||
Some(ConnectionStatus::Disconnected)
|
||
| Some(ConnectionStatus::Connecting)
|
||
| Some(ConnectionStatus::Disconnecting) => {
|
||
// Will be handled when connection succeeds (or ignored if disconnecting)
|
||
tracing::debug!(
|
||
relay = %action.relay_url,
|
||
status = ?connection_status,
|
||
"Relay not connected, action will be processed on connection"
|
||
);
|
||
return;
|
||
}
|
||
Some(ConnectionStatus::Syncing)
|
||
| Some(ConnectionStatus::Connected)
|
||
| Some(ConnectionStatus::ConnectedHistoricSyncFailures) => {
|
||
// Continue to subscribe - live sync is active, can accept new filters
|
||
}
|
||
}
|
||
|
||
// Step 2: Check if relay is rate-limited before creating new pending items
|
||
if self
|
||
.health_tracker
|
||
.is_subscription_paused(&action.relay_url)
|
||
{
|
||
tracing::debug!(
|
||
relay = %action.relay_url,
|
||
full_repos = action.items.repos.len(),
|
||
state_only_repos = action.items.state_only_repos.len(),
|
||
root_events = action.items.root_events.len(),
|
||
"Skipping AddFilters for rate-limited relay, will recompute after cooldown"
|
||
);
|
||
return;
|
||
}
|
||
|
||
// Subscribe to each filter and collect subscription IDs
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
filter_count = action.filters.len(),
|
||
full_repo_count = action.items.repos.len(),
|
||
state_only_repo_count = action.items.state_only_repos.len(),
|
||
root_event_count = action.items.root_events.len(),
|
||
"handle_add_filters: calling sync_live and historic_sync"
|
||
);
|
||
|
||
let essential_live = filters::build_essential_live_filters(
|
||
&action.items.repos,
|
||
&action.items.state_only_repos,
|
||
&action.items.root_events,
|
||
None,
|
||
);
|
||
if let Err(error) = self.sync_live(&action.relay_url, &essential_live).await {
|
||
tracing::warn!(
|
||
relay = %action.relay_url,
|
||
%error,
|
||
"Live coverage could not be extended; continuing bounded historic sync"
|
||
);
|
||
if error.starts_with("Minimum-churn live extension needs")
|
||
|| error.starts_with("Minimum-churn live extension exceeds")
|
||
{
|
||
let has_pending_batches = self.has_pending_batches(&action.relay_url).await;
|
||
if self
|
||
.deferred_consolidations
|
||
.request(&action.relay_url, has_pending_batches)
|
||
{
|
||
let _ = self.consolidate(&action.relay_url).await;
|
||
} else {
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
"Minimum-churn extension reached capacity; full consolidation deferred until pending batches drain"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
self.historic_sync(&action.relay_url, action.filters, action.items, None)
|
||
.await;
|
||
}
|
||
|
||
/// Handle a connection success (called when a relay connects or reconnects)
|
||
///
|
||
/// This method:
|
||
/// 1. Updates RelayState to Connected
|
||
/// 2. Spawns event loop (MUST happen on every connection/reconnect)
|
||
/// 3. Dispatches to appropriate reconnection strategy based on disconnect time
|
||
async fn handle_connect_or_reconnect(&mut self, relay_url: &str) {
|
||
use tokio::sync::mpsc;
|
||
|
||
// 1. Capture old last_connected BEFORE updating state
|
||
// This is critical for correct first-connection detection
|
||
let old_last_connected = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index.get(relay_url).and_then(|s| s.last_connected)
|
||
};
|
||
|
||
// 2. Update state to Syncing (will transition to Connected after historic sync completes)
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
let state = index.entry(relay_url.to_string()).or_default();
|
||
state.connection_status = ConnectionStatus::Syncing;
|
||
state.last_connected = Some(Timestamp::now());
|
||
state.disconnected_at = None;
|
||
}
|
||
|
||
// Update metrics - record as syncing initially
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_status(relay_url, ConnectionStatus::Syncing);
|
||
metrics.inc_connected_count();
|
||
}
|
||
|
||
// 2. SPAWN EVENT LOOP (moved from spawn_relay_connection)
|
||
// This MUST happen on every connection (initial or reconnect)
|
||
// because event loops die on disconnect and cannot be reused
|
||
let connection = match self.connections.get(relay_url) {
|
||
Some(c) => c.clone(),
|
||
None => {
|
||
tracing::error!(relay = %relay_url, "No RelayConnection found for connected relay");
|
||
return;
|
||
}
|
||
};
|
||
|
||
let (event_tx, mut event_rx) =
|
||
mpsc::channel::<RelayEvent>(relay_connection::RELAY_EVENT_BUFFER_CAPACITY);
|
||
|
||
// Spawn event loop task
|
||
let relay_url_for_loop = relay_url.to_string();
|
||
tokio::spawn(async move {
|
||
connection.run_event_loop(event_tx).await;
|
||
tracing::debug!(relay = %relay_url_for_loop, "Event loop terminated");
|
||
});
|
||
|
||
// Spawn event processor task
|
||
let relay_url_clone = relay_url.to_string();
|
||
let database = Arc::clone(&self.database);
|
||
let write_policy = self.write_policy.clone();
|
||
let local_relay = self.local_relay.clone();
|
||
let disconnect_tx = self.disconnect_tx.as_ref().unwrap().clone();
|
||
let eose_tx = self.eose_tx.as_ref().unwrap().clone();
|
||
let subscription_closed_tx = self.subscription_closed_tx.as_ref().unwrap().clone();
|
||
let metrics_clone = self.metrics.clone();
|
||
let pending_sync_index = Arc::clone(&self.pending_sync_index);
|
||
let health_tracker = Arc::clone(&self.health_tracker);
|
||
let rejected_events_index = Arc::clone(&self.rejected_events_index);
|
||
|
||
tokio::spawn(async move {
|
||
let mut disconnect_sent = false;
|
||
let mut pipeline_window = EventPipelineWindow::default();
|
||
|
||
while let Some(relay_event) = event_rx.recv().await {
|
||
match relay_event {
|
||
RelayEvent::Event(event, subscription_id, data_lane_arrival) => {
|
||
let queue_delay = data_lane_arrival.elapsed();
|
||
let processing_started = std::time::Instant::now();
|
||
// Count raw deliveries before deduplication or write policy. Relays spend
|
||
// their result allowance on every matching delivery, including events we
|
||
// route to purgatory, reject, or have already stored; pagination must use
|
||
// that same stream for both its page count and `until` cursor.
|
||
{
|
||
let mut pending = pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_clone) {
|
||
for batch in batches.iter_mut() {
|
||
if let Some(state) =
|
||
batch.pagination_state.get_mut(&subscription_id)
|
||
{
|
||
state.record_event(&event);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Skip events we've already rejected (announcements only)
|
||
if (event.kind == Kind::GitRepoAnnouncement
|
||
|| event.kind == Kind::RepoState)
|
||
&& rejected_events_index.contains(&event.id)
|
||
{
|
||
tracing::trace!(
|
||
event_id = %event.id,
|
||
kind = %event.kind.as_u16(),
|
||
relay = %relay_url_clone,
|
||
"Skipping previously rejected announcement event"
|
||
);
|
||
pipeline_window.record(
|
||
ProcessResult::Rejected(PolicyRejection::PreviouslyRejected),
|
||
queue_delay,
|
||
processing_started.elapsed(),
|
||
);
|
||
if let Some(ref metrics) = metrics_clone {
|
||
metrics.record_hydration_events(
|
||
&relay_url_clone,
|
||
"stream",
|
||
"delivered",
|
||
1,
|
||
);
|
||
metrics.record_hydration_events(
|
||
&relay_url_clone,
|
||
"stream",
|
||
"rejected_cached",
|
||
1,
|
||
);
|
||
}
|
||
pipeline_window.report_if_due(&relay_url_clone, event_rx.len());
|
||
continue;
|
||
}
|
||
|
||
let result = Self::process_event_static(
|
||
&event,
|
||
&relay_url_clone,
|
||
&database,
|
||
&write_policy,
|
||
&local_relay,
|
||
&rejected_events_index,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
if let Some(ref metrics) = metrics_clone {
|
||
metrics.record_hydration_events(
|
||
&relay_url_clone,
|
||
"stream",
|
||
"delivered",
|
||
1,
|
||
);
|
||
metrics.record_hydration_events(
|
||
&relay_url_clone,
|
||
"stream",
|
||
result.hydration_outcome(),
|
||
1,
|
||
);
|
||
}
|
||
// Only record metric when event is actually saved
|
||
if result == ProcessResult::Saved {
|
||
if let Some(ref metrics) = metrics_clone {
|
||
metrics.record_synced_event();
|
||
}
|
||
}
|
||
|
||
// For sync-triggered events that go to purgatory, trigger immediate sync
|
||
// (instead of the default 3-minute delay for user-submitted events)
|
||
//
|
||
// Note: announcement events (kind 30617) are registered in repo_sync_index
|
||
// by the purgatory announcement sync timer (run_purgatory_announcement_sync)
|
||
// rather than inline here.
|
||
if result == ProcessResult::Purgatory {
|
||
// State events (kind 30618) - extract identifier and trigger immediate sync
|
||
if event.kind.as_u16() == 30618 {
|
||
if let Some(identifier) = event.tags.iter().find_map(|tag| {
|
||
let tag_vec = tag.clone().to_vec();
|
||
if tag_vec.len() >= 2 && tag_vec[0] == "d" {
|
||
Some(tag_vec[1].clone())
|
||
} else {
|
||
None
|
||
}
|
||
}) {
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
identifier = %identifier,
|
||
"Triggering immediate sync for synced state event in purgatory"
|
||
);
|
||
write_policy.purgatory().enqueue_sync_immediate(&identifier);
|
||
}
|
||
}
|
||
// PR events (kind 1617/1618) - extract identifier from 'a' tag
|
||
else if event.kind.as_u16() == 1617 || event.kind.as_u16() == 1618 {
|
||
if let Some(identifier) =
|
||
crate::git::sync::extract_identifier_from_pr_event(&event)
|
||
{
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
identifier = %identifier,
|
||
"Triggering immediate sync for synced PR event in purgatory"
|
||
);
|
||
write_policy.purgatory().enqueue_sync_immediate(&identifier);
|
||
}
|
||
}
|
||
}
|
||
|
||
// Track received event IDs for negentropy batches. Unlike REQ+EOSE
|
||
// pagination above, negentropy completion is concerned with events that
|
||
// were actually saved or already present locally.
|
||
if result.is_terminally_accounted() {
|
||
let mut pending = pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(&relay_url_clone) {
|
||
for batch in batches.iter_mut() {
|
||
// Track received event IDs (negentropy path)
|
||
// Only track if this batch has requested_event_ids set
|
||
// and the subscription is one we're waiting on
|
||
if batch.requested_event_ids.is_some()
|
||
&& batch.outstanding_subs.contains(&subscription_id)
|
||
{
|
||
if let Some(ref mut received) = batch.received_event_ids {
|
||
received.insert(event.id);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
}
|
||
pipeline_window.record(result, queue_delay, processing_started.elapsed());
|
||
pipeline_window.report_if_due(&relay_url_clone, event_rx.len());
|
||
}
|
||
RelayEvent::EndOfStoredEvents(sub_id, data_lane_arrival) => {
|
||
let data_lane_delay = data_lane_arrival.elapsed();
|
||
if data_lane_delay >= std::time::Duration::from_secs(5) {
|
||
tracing::info!(
|
||
relay = %relay_url_clone,
|
||
sub_id = %sub_id,
|
||
data_lane_delay_seconds = data_lane_delay.as_secs_f64(),
|
||
queue_depth = event_rx.len(),
|
||
queue_capacity = relay_connection::RELAY_EVENT_BUFFER_CAPACITY,
|
||
"EOSE reached sync manager after data-lane delay"
|
||
);
|
||
}
|
||
tracing::debug!(
|
||
relay = %relay_url_clone,
|
||
sub_id = %sub_id,
|
||
"EOSE received, notifying SyncManager"
|
||
);
|
||
let _ = eose_tx
|
||
.send(EoseNotification {
|
||
relay_url: relay_url_clone.clone(),
|
||
sub_id,
|
||
})
|
||
.await;
|
||
}
|
||
RelayEvent::Notice(notice) => {
|
||
if is_rate_limit_message(¬ice) {
|
||
tracing::warn!(
|
||
relay = %relay_url_clone,
|
||
notice = %notice,
|
||
"Rate limiting NOTICE detected from relay"
|
||
);
|
||
|
||
// Mark relay as rate limited
|
||
health_tracker.record_rate_limit(&relay_url_clone);
|
||
|
||
// Update metrics with new health state
|
||
if let Some(ref metrics) = metrics_clone {
|
||
let state = health_tracker.get_state(&relay_url_clone);
|
||
metrics.record_health_state(&relay_url_clone, state);
|
||
}
|
||
} else {
|
||
// Log at TRACE level to avoid duplicate with nostr_relay_pool's DEBUG log
|
||
// (nostr-sdk already logs all NOTICE messages at DEBUG level)
|
||
tracing::trace!(
|
||
relay = %relay_url_clone,
|
||
notice = %notice,
|
||
"Relay issued notice"
|
||
);
|
||
}
|
||
}
|
||
RelayEvent::Closed {
|
||
subscription_id,
|
||
reason,
|
||
live_generation,
|
||
live_filter_count,
|
||
} => {
|
||
// CLOSED message means one subscription was closed, not the whole connection
|
||
// This is normal behavior (e.g., when historic_sync completes)
|
||
tracing::debug!(
|
||
relay = %relay_url_clone,
|
||
sub_id = %subscription_id,
|
||
reason = %reason,
|
||
"Relay closed a subscription (not a connection close)"
|
||
);
|
||
if is_rate_limit_message(&reason)
|
||
&& !is_filter_count_refusal(&reason)
|
||
&& subscription_state_byte_limit(&reason).is_none()
|
||
{
|
||
let already_paused = health_tracker.is_rate_limited(&relay_url_clone);
|
||
if already_paused {
|
||
tracing::debug!(
|
||
relay = %relay_url_clone,
|
||
reason = %reason,
|
||
"Repeated rate-limiting CLOSED during active cooldown"
|
||
);
|
||
} else {
|
||
tracing::info!(
|
||
relay = %relay_url_clone,
|
||
reason = %reason,
|
||
"Rate limiting CLOSED detected from relay"
|
||
);
|
||
}
|
||
health_tracker.record_rate_limit(&relay_url_clone);
|
||
if let Some(ref metrics) = metrics_clone {
|
||
metrics.record_health_state(
|
||
&relay_url_clone,
|
||
health_tracker.get_state(&relay_url_clone),
|
||
);
|
||
}
|
||
}
|
||
let _ = subscription_closed_tx
|
||
.send(SubscriptionClosedNotification {
|
||
relay_url: relay_url_clone.clone(),
|
||
subscription_id,
|
||
reason,
|
||
generation: live_generation,
|
||
live_filter_count,
|
||
})
|
||
.await;
|
||
}
|
||
RelayEvent::Shutdown => {
|
||
tracing::info!(relay = %relay_url_clone, "Relay shutdown detected");
|
||
if !disconnect_sent {
|
||
let _ = disconnect_tx
|
||
.send(DisconnectNotification {
|
||
relay_url: relay_url_clone.clone(),
|
||
})
|
||
.await;
|
||
disconnect_sent = true;
|
||
}
|
||
break;
|
||
}
|
||
}
|
||
}
|
||
|
||
// If the event channel closed without a Closed/Shutdown event
|
||
if !disconnect_sent {
|
||
tracing::info!(
|
||
relay = %relay_url_clone,
|
||
"Event channel closed, notifying SyncManager of disconnect"
|
||
);
|
||
let _ = disconnect_tx
|
||
.send(DisconnectNotification {
|
||
relay_url: relay_url_clone,
|
||
})
|
||
.await;
|
||
}
|
||
});
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"Event loop and processor spawned for connected relay"
|
||
);
|
||
|
||
if self.nip65_discovery_only_relays.contains(relay_url) {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"NIP-65 discovery connection ready without repository sync"
|
||
);
|
||
return;
|
||
}
|
||
|
||
// 3. Decide reconnection strategy based on OLD last_connected time
|
||
// Use the value captured BEFORE the update to correctly detect first connections
|
||
if let Some(last) = old_last_connected {
|
||
let elapsed = Timestamp::now().as_secs().saturating_sub(last.as_secs());
|
||
if elapsed < QUICK_RECONNECT_WINDOW_SECS {
|
||
// Short disconnect - quick reconnect
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
disconnect_secs = elapsed,
|
||
"Short disconnection - initiating quick_reconnect"
|
||
);
|
||
self.quick_reconnect(relay_url, Timestamp::from(elapsed))
|
||
.await;
|
||
} else {
|
||
// Long disconnect - fresh start
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
disconnect_secs = elapsed,
|
||
"Long disconnection - initiating fresh_start"
|
||
);
|
||
self.fresh_start(relay_url).await;
|
||
}
|
||
} else {
|
||
// First connection - fresh start
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"First connection - initiating fresh_start"
|
||
);
|
||
self.fresh_start(relay_url).await;
|
||
}
|
||
}
|
||
|
||
/// Fresh start - clears state and does full sync
|
||
///
|
||
/// Called by: initial connect, long_reconnect, daily_sync
|
||
///
|
||
/// Flow:
|
||
/// 1. Clear PendingSyncIndex for this relay
|
||
/// 2. Clear RelaySyncIndex sync state (repos/root_events)
|
||
/// 3. Update connection state to Connected
|
||
/// 4. L1 live + L1 historic (negentropy if available)
|
||
/// 5. compute_actions → AddFilters → sync_computed_filters for L2+L3
|
||
async fn fresh_start(&mut self, relay_url: &str) {
|
||
let _now = Timestamp::now();
|
||
|
||
tracing::info!(relay = %relay_url, "Starting fresh_start");
|
||
self.cancel_deferred_consolidation(relay_url, "fresh start");
|
||
|
||
// Step 1: Clear PendingSyncIndex for this relay
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if pending.remove(relay_url).is_some() {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Cleared pending batches in fresh_start"
|
||
);
|
||
}
|
||
}
|
||
|
||
// Step 2: Clear RelaySyncIndex sync state (but preserve connection metadata)
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
if let Some(state) = index.get_mut(relay_url) {
|
||
let repos_cleared = state.repos.len() + state.state_only_repos.len();
|
||
let events_cleared = state.root_events.len();
|
||
state.clear_sync_state();
|
||
if repos_cleared > 0 || events_cleared > 0 {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
repos_cleared = repos_cleared,
|
||
events_cleared = events_cleared,
|
||
"Cleared sync state in fresh_start"
|
||
);
|
||
}
|
||
// Only sync if we're connected (live sync active)
|
||
if state.connection_status.is_live_sync_active() {
|
||
drop(index);
|
||
self.sync_generic_filters(relay_url, None).await;
|
||
// Step 5: compute_actions for L2+L3 (will be triggered by EOSE)
|
||
self.recompute_new_sync_filters_for_relay(relay_url).await;
|
||
}
|
||
} else {
|
||
drop(index);
|
||
}
|
||
}
|
||
}
|
||
|
||
async fn sync_generic_filters(&mut self, relay_url: &str, since: Option<Timestamp>) {
|
||
let filters = vec![filters::build_announcement_filter(None)];
|
||
|
||
// Create live subscription for ongoing announcements
|
||
if self.sync_live(relay_url, &filters).await.is_err() {
|
||
return;
|
||
}
|
||
|
||
// Use historic_sync with empty PendingItems for generic filters
|
||
// Generic filters (announcements) don't have associated repos or root_events
|
||
let items = PendingItems::default();
|
||
let _batch_id = self
|
||
.historic_sync_with_options(
|
||
relay_url,
|
||
filters,
|
||
items,
|
||
since,
|
||
PendingBatchPurpose::Announcements,
|
||
false,
|
||
)
|
||
.await;
|
||
}
|
||
|
||
async fn sync_generic_history(&mut self, relay_url: &str, since: Option<Timestamp>) {
|
||
let filters = vec![filters::build_announcement_filter(None)];
|
||
let _batch_id = self
|
||
.historic_sync_with_options(
|
||
relay_url,
|
||
filters,
|
||
PendingItems::default(),
|
||
since,
|
||
PendingBatchPurpose::Announcements,
|
||
false,
|
||
)
|
||
.await;
|
||
}
|
||
|
||
/// Find the bounded transitive members of repository root threads.
|
||
async fn descendant_thread_members(
|
||
&self,
|
||
root_events: &HashSet<EventId>,
|
||
) -> DescendantFrontier {
|
||
recursive_descendant_frontier(
|
||
&self.database,
|
||
root_events,
|
||
self.config.sync_recursive_descendant_limit,
|
||
)
|
||
.await
|
||
}
|
||
|
||
async fn close_descendant_live_coverage(
|
||
&mut self,
|
||
relay_url: &str,
|
||
reason: &'static str,
|
||
) -> bool {
|
||
let Some(coverage) = self.descendant_live_coverage.remove(relay_url) else {
|
||
return true;
|
||
};
|
||
if let Some(connection) = self.connections.get(relay_url) {
|
||
if let Err(error) = connection
|
||
.close_live_subscriptions(&coverage.subscription_ids)
|
||
.await
|
||
{
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
%error,
|
||
reason,
|
||
"Could not retire auxiliary descendant live coverage"
|
||
);
|
||
self.descendant_live_coverage
|
||
.insert(relay_url.to_string(), coverage);
|
||
return false;
|
||
}
|
||
}
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
subscription_count = coverage.subscription_ids.len(),
|
||
reason,
|
||
"Retired auxiliary descendant live coverage"
|
||
);
|
||
true
|
||
}
|
||
|
||
async fn reconcile_descendant_mode(
|
||
&mut self,
|
||
relay_url: &str,
|
||
target: &algorithms::RelaySyncNeeds,
|
||
) {
|
||
let frontier = self.descendant_thread_members(&target.root_events).await;
|
||
let historic_entries =
|
||
tiered_auxiliary_filters(&target.repos, &target.root_events, &frontier, None);
|
||
let live_since = Timestamp::from(
|
||
Timestamp::now()
|
||
.as_secs()
|
||
.saturating_sub(DESCENDANT_FALLBACK_OVERLAP_SECS),
|
||
);
|
||
if let Some(connection) = self.connections.get(relay_url).cloned() {
|
||
let (live_filters, rotated_filters) = split_live_tier_prefix(
|
||
&historic_entries,
|
||
connection.max_filters_per_req(),
|
||
|groups| connection.can_admit_auxiliary_live_groups(groups),
|
||
);
|
||
let desired_live_filters: HashSet<_> =
|
||
live_filters.iter().map(Filter::as_json).collect();
|
||
if let Some(coverage) = self.descendant_live_coverage.get_mut(relay_url) {
|
||
if desired_live_filters == coverage.desired_filters {
|
||
let max_filters = connection.max_filters_per_req();
|
||
let rotation = self
|
||
.descendant_sync_rotations
|
||
.entry(relay_url.to_string())
|
||
.or_default();
|
||
if rotation.coverage_fingerprint
|
||
!= rotation_fingerprint(&rotated_filters, max_filters)
|
||
{
|
||
rotation.refresh(rotated_filters, max_filters);
|
||
}
|
||
coverage.fallback_filters = complete_auxiliary_fallback(&historic_entries);
|
||
return;
|
||
}
|
||
}
|
||
if self.descendant_live_coverage.contains_key(relay_url)
|
||
&& !self
|
||
.close_descendant_live_coverage(relay_url, "tiered coverage changed")
|
||
.await
|
||
{
|
||
return;
|
||
}
|
||
let live_filters: Vec<_> = live_filters
|
||
.into_iter()
|
||
.map(|filter| filter.since(live_since))
|
||
.collect();
|
||
let groups = live_filter_groups(&live_filters, connection.max_filters_per_req());
|
||
if !groups.is_empty() {
|
||
match connection
|
||
.subscribe_auxiliary_live_filter_groups(groups)
|
||
.await
|
||
{
|
||
Ok(subscription_ids) => {
|
||
// A relay can close any one member of this live set.
|
||
// Keep the complete frontier ready for rotation so
|
||
// the CLOSED path preserves every tier immediately,
|
||
// including those that had been live until now.
|
||
let fallback_filters = complete_auxiliary_fallback(&historic_entries);
|
||
let filter_count = live_filters.len();
|
||
let event_id_count = frontier.event_ids.len();
|
||
let coordinate_count = frontier.coordinates.len();
|
||
self.descendant_live_coverage.insert(
|
||
relay_url.to_string(),
|
||
DescendantLiveCoverage {
|
||
desired_filters: desired_live_filters,
|
||
fallback_filters,
|
||
subscription_ids: subscription_ids.clone(),
|
||
},
|
||
);
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
event_id_count,
|
||
coordinate_count,
|
||
filter_count,
|
||
subscription_count = subscription_ids.len(),
|
||
rotated_filter_count = rotated_filters.len(),
|
||
"Installed priority-bounded auxiliary live coverage"
|
||
);
|
||
// `limit:0` protects the future only. Pair every new
|
||
// live frontier with a complete EOSE-closing baseline
|
||
// so events stored before admission are not skipped.
|
||
let _ = self
|
||
.historic_sync_with_options(
|
||
relay_url,
|
||
historic_entries
|
||
.iter()
|
||
.map(|entry| entry.filter.clone())
|
||
.collect(),
|
||
PendingItems::default(),
|
||
None,
|
||
PendingBatchPurpose::Descendants,
|
||
true,
|
||
)
|
||
.await;
|
||
let rotation = self
|
||
.descendant_sync_rotations
|
||
.entry(relay_url.to_string())
|
||
.or_default();
|
||
let max_filters = connection.max_filters_per_req();
|
||
if rotation.coverage_fingerprint
|
||
!= rotation_fingerprint(&rotated_filters, max_filters)
|
||
{
|
||
rotation.refresh(rotated_filters, max_filters);
|
||
}
|
||
return;
|
||
}
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
%error,
|
||
"Descendant live admission failed; using historic rotation"
|
||
);
|
||
}
|
||
}
|
||
} else {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
filter_count = live_filters.len(),
|
||
"No auxiliary live tier fits while preserving historic capacity"
|
||
);
|
||
}
|
||
}
|
||
|
||
let rotated_filters: Vec<_> = historic_entries
|
||
.into_iter()
|
||
.map(|entry| entry.filter)
|
||
.collect();
|
||
let max_filters = self
|
||
.connections
|
||
.get(relay_url)
|
||
.map(|connection| connection.max_filters_per_req())
|
||
.unwrap_or(MAX_FILTERS_PER_REQ);
|
||
let rotation = self
|
||
.descendant_sync_rotations
|
||
.entry(relay_url.to_string())
|
||
.or_default();
|
||
if rotation.coverage_fingerprint != rotation_fingerprint(&rotated_filters, max_filters) {
|
||
rotation.refresh(rotated_filters, max_filters);
|
||
}
|
||
}
|
||
|
||
async fn start_descendant_fallback(&mut self, relay_url: &str) {
|
||
let Some((filter_index, filters, until)) = self
|
||
.descendant_sync_rotations
|
||
.get(relay_url)
|
||
.and_then(|rotation| rotation.next_request(Timestamp::now()))
|
||
else {
|
||
return;
|
||
};
|
||
let filter_count = self.descendant_sync_rotations[relay_url].filters.len();
|
||
if let Some(batch_id) = self
|
||
.historic_sync_with_options(
|
||
relay_url,
|
||
filters,
|
||
PendingItems::default(),
|
||
None,
|
||
PendingBatchPurpose::Descendants,
|
||
true,
|
||
)
|
||
.await
|
||
{
|
||
if let Some(rotation) = self.descendant_sync_rotations.get_mut(relay_url) {
|
||
rotation.mark_started(batch_id, filter_index, until);
|
||
}
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
filter_index,
|
||
filter_count,
|
||
until = until.as_secs(),
|
||
"Started queued descendant fallback query"
|
||
);
|
||
}
|
||
}
|
||
|
||
/// Reconcile one live-mode decision and advance every constrained relay by
|
||
/// at most one EOSE-closing request on each five-second maintenance tick.
|
||
async fn tick_descendant_sync(&mut self) {
|
||
let targets = self.derive_targets().await;
|
||
let states = self.relay_sync_index.read().await;
|
||
let mut relay_urls: Vec<String> = targets
|
||
.iter()
|
||
.filter(|(relay_url, needs)| {
|
||
(!needs.repos.is_empty() || !needs.root_events.is_empty())
|
||
&& states.get(*relay_url).is_some_and(|state| {
|
||
matches!(
|
||
state.connection_status,
|
||
ConnectionStatus::Connected
|
||
| ConnectionStatus::ConnectedHistoricSyncFailures
|
||
)
|
||
})
|
||
&& !self.health_tracker.is_subscription_paused(relay_url)
|
||
})
|
||
.map(|(relay_url, _)| relay_url.clone())
|
||
.collect();
|
||
drop(states);
|
||
relay_urls.sort_unstable();
|
||
if relay_urls.is_empty() {
|
||
return;
|
||
}
|
||
|
||
let reconcile_relay = relay_urls[self.descendant_relay_cursor % relay_urls.len()].clone();
|
||
self.descendant_relay_cursor = self.descendant_relay_cursor.wrapping_add(1);
|
||
self.reconcile_descendant_mode(&reconcile_relay, &targets[&reconcile_relay])
|
||
.await;
|
||
|
||
let constrained: Vec<String> = relay_urls
|
||
.into_iter()
|
||
.filter(|relay_url| self.descendant_sync_rotations.contains_key(relay_url))
|
||
.collect();
|
||
for relay_url in constrained {
|
||
self.start_descendant_fallback(&relay_url).await;
|
||
}
|
||
}
|
||
|
||
/// Build the complete persistent coverage for one connection. L1 must be
|
||
/// admitted in the same transaction as rebuilt L2/L3 so a tight relay
|
||
/// budget cannot leave a successful generic REQ hiding partial repo
|
||
/// coverage.
|
||
async fn complete_live_filters(
|
||
&self,
|
||
relay_url: &str,
|
||
since: Option<Timestamp>,
|
||
) -> Vec<Filter> {
|
||
let mut filters = vec![filters::build_announcement_filter(None)];
|
||
let index = self.relay_sync_index.read().await;
|
||
if let Some(state) = index.get(relay_url) {
|
||
filters.extend(filters::build_essential_live_filters(
|
||
&state.repos,
|
||
&state.state_only_repos,
|
||
&state.root_events,
|
||
since,
|
||
));
|
||
}
|
||
filters
|
||
}
|
||
|
||
async fn desired_items_for_relay(&self, relay_url: &str) -> PendingItems {
|
||
let target = self
|
||
.derive_targets()
|
||
.await
|
||
.remove(relay_url)
|
||
.unwrap_or_default();
|
||
PendingItems {
|
||
repos: target.repos,
|
||
state_only_repos: target.state_only_repos,
|
||
root_events: target.root_events,
|
||
}
|
||
}
|
||
|
||
/// Quick reconnect - for disconnections < 15 minutes
|
||
///
|
||
/// Re-establishes subscriptions after a brief disconnection by:
|
||
/// 1. Clearing stale PendingSyncIndex entries
|
||
/// 2. Syncing L1 filters with since timestamp (announcements)
|
||
/// 3. Rebuilding L2+L3 from preserved RelaySyncIndex state
|
||
/// 4. Computing actions for new items discovered during catchup
|
||
///
|
||
/// Basic connection health is managed by the connection worker and its
|
||
/// stability timer. This method handles reconnect-specific sync and metrics.
|
||
async fn quick_reconnect(&mut self, relay_url: &str, since: Timestamp) {
|
||
self.cancel_deferred_consolidation(relay_url, "quick reconnect");
|
||
|
||
// Step 1: Clear PendingSyncIndex for this relay
|
||
// Old subscriptions are dead after disconnect
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
pending.remove(relay_url);
|
||
}
|
||
|
||
// Record reconnect-specific metrics (not basic connection metrics)
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_health_state(relay_url, self.health_tracker.get_state(relay_url));
|
||
}
|
||
|
||
// Step 2: L1 live + L1 historic with since filter (or full sync if announcements never completed)
|
||
let announcement_since = {
|
||
let index = self.relay_sync_index.read().await;
|
||
if let Some(state) = index.get(relay_url) {
|
||
if state.announcements_synced {
|
||
Some(since) // Can use incremental sync
|
||
} else {
|
||
None // Need full sync - announcements never completed
|
||
}
|
||
} else {
|
||
None
|
||
}
|
||
};
|
||
|
||
let complete_live = self.complete_live_filters(relay_url, Some(since)).await;
|
||
if self.sync_live(relay_url, &complete_live).await.is_err() {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
"Complete live coverage did not fit on reconnect; historic work deferred"
|
||
);
|
||
return;
|
||
}
|
||
self.sync_generic_history(relay_url, announcement_since)
|
||
.await;
|
||
|
||
// Step 4: compute_actions for any NEW items discovered while disconnected
|
||
self.recompute_new_sync_filters_for_relay(relay_url).await;
|
||
}
|
||
|
||
/// Register a relay for managed connection/reconnection
|
||
///
|
||
/// Creates a RelayConnection object and stores it in the connections HashMap.
|
||
/// Also initializes RelayState if it doesn't exist.
|
||
/// Does NOT connect - connection happens via the bounded scheduler.
|
||
/// The RelayConnection persists forever and is reused on reconnects.
|
||
///
|
||
/// Returns `false` when the target was rejected and no connection exists,
|
||
/// so callers must not schedule a connection attempt. Event-directed URLs
|
||
/// are vetted by the outbound target policy; the operator-configured
|
||
/// bootstrap relay is exempt. An event URL identical to an already
|
||
/// registered connection (e.g. the bootstrap relay itself) reuses that
|
||
/// connection rather than being re-authorized, which keeps the configured
|
||
/// exception exact: only URLs that canonicalize to the same key share it.
|
||
async fn register_relay(
|
||
&mut self,
|
||
relay_url: String,
|
||
is_bootstrap: bool,
|
||
nip65_discovery_only: bool,
|
||
) -> bool {
|
||
let relay_url = match canonical_relay_key(&relay_url) {
|
||
Ok(relay_url) => relay_url,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
error = %error,
|
||
"Rejecting invalid relay registration"
|
||
);
|
||
return false;
|
||
}
|
||
};
|
||
|
||
// A GRASP-08 private service is not a sync target for a public
|
||
// instance; never re-enter the connection lifecycle for it.
|
||
if !self.config.private_mode && self.private_service_relays.contains(&relay_url) {
|
||
tracing::trace!(
|
||
relay = %relay_url,
|
||
"Skipping registration of GRASP-08 private-service relay"
|
||
);
|
||
return false;
|
||
}
|
||
|
||
// An ordinary sync registration upgrades a connection that was first
|
||
// opened only for NIP-65 discovery. Discovery must never downgrade an
|
||
// existing repository source.
|
||
if !nip65_discovery_only {
|
||
self.nip65_discovery_only_relays.remove(&relay_url);
|
||
}
|
||
|
||
// Create RelayConnection if not exists
|
||
if !self.connections.contains_key(&relay_url) {
|
||
let policy = self.outbound_target_policy();
|
||
let source = if is_bootstrap {
|
||
RelayTargetSource::OperatorConfigured
|
||
} else {
|
||
RelayTargetSource::EventDirected
|
||
};
|
||
|
||
// Refuse to even register unauthorized event-directed targets so
|
||
// they never enter the reconnect lifecycle. The connection worker
|
||
// re-checks (with DNS vetting) immediately before every dial.
|
||
if source == RelayTargetSource::EventDirected {
|
||
if let Err(reason) = policy.authorize(OutboundTargetKind::EventRelay, &relay_url) {
|
||
// Warn once per target; sync passes keep re-deriving the
|
||
// same URLs from stored events.
|
||
if self.rejected_relay_targets.insert(relay_url.clone()) {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
reason = %reason,
|
||
"Rejecting event-directed sync relay target"
|
||
);
|
||
}
|
||
return false;
|
||
}
|
||
}
|
||
|
||
// The relay owner key answers outbound NIP-42 challenges. Sync
|
||
// must keep working without it, so a missing key degrades to an
|
||
// unauthenticated connection instead of aborting registration.
|
||
let keys = match self.config.relay_owner_keys() {
|
||
Ok(keys) => Some(keys),
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
error = %error,
|
||
"Relay owner key unavailable; outbound NIP-42 authentication disabled for this connection"
|
||
);
|
||
None
|
||
}
|
||
};
|
||
|
||
let connection = RelayConnection::new_with_database(
|
||
relay_url.clone(),
|
||
Arc::clone(&self.database),
|
||
keys,
|
||
source,
|
||
policy,
|
||
);
|
||
self.connections.insert(relay_url.clone(), connection);
|
||
if nip65_discovery_only {
|
||
self.nip65_discovery_only_relays.insert(relay_url.clone());
|
||
}
|
||
tracing::debug!(relay = %relay_url, "Registered new relay connection");
|
||
}
|
||
|
||
// Initialize RelayState if not exists
|
||
let is_new = {
|
||
let mut index = self.relay_sync_index.write().await;
|
||
if !index.contains_key(&relay_url) {
|
||
let new_state = RelayState {
|
||
connection_status: ConnectionStatus::Disconnected,
|
||
is_bootstrap,
|
||
last_connected: None,
|
||
disconnected_at: None,
|
||
repos: HashSet::new(),
|
||
state_only_repos: HashSet::new(),
|
||
root_events: HashSet::new(),
|
||
announcements_synced: false,
|
||
historic_sync_completed: false,
|
||
historic_sync_completed_at: None,
|
||
historic_sync_had_failures: false,
|
||
};
|
||
index.insert(relay_url.clone(), new_state);
|
||
true
|
||
} else {
|
||
// If relay already exists and is_bootstrap is true, update the flag
|
||
if is_bootstrap {
|
||
if let Some(state) = index.get_mut(&relay_url) {
|
||
state.is_bootstrap = true;
|
||
}
|
||
}
|
||
false
|
||
}
|
||
};
|
||
|
||
// Track new relay in metrics
|
||
if is_new {
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.inc_tracked_count();
|
||
// Initialize connection status to disconnected
|
||
metrics.set_relay_connected(&relay_url, false);
|
||
}
|
||
tracing::info!(relay = %relay_url, "Registered new relay for tracking");
|
||
}
|
||
|
||
true
|
||
}
|
||
|
||
/// Outbound target policy derived from operator configuration.
|
||
fn outbound_target_policy(&self) -> OutboundTargetPolicy {
|
||
OutboundTargetPolicy {
|
||
allow_non_global: self.config.sync_allow_non_global_targets,
|
||
}
|
||
}
|
||
|
||
/// Queue one connection attempt without blocking the sync actor.
|
||
///
|
||
/// DNS and the websocket handshake run in a spawned worker, bounded globally
|
||
/// by `MAX_CONCURRENT_CONNECT_ATTEMPTS`. The actor owns every lifecycle
|
||
/// transition when it receives the worker result.
|
||
async fn schedule_connect_relay(&mut self, relay_url: &str) {
|
||
let relay_url = match canonical_relay_key(relay_url) {
|
||
Ok(relay_url) => relay_url,
|
||
Err(error) => {
|
||
tracing::warn!(relay = %relay_url, error = %error, "Rejecting invalid sync relay target");
|
||
return;
|
||
}
|
||
};
|
||
if self
|
||
.health_tracker
|
||
.naughty_list()
|
||
.is_some_and(|naughty_list| naughty_list.is_naughty(&relay_url))
|
||
{
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Suppressing connection attempt for naughty relay"
|
||
);
|
||
return;
|
||
}
|
||
if !self.config.private_mode && self.private_service_relays.contains(&relay_url) {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Suppressing connection attempt for GRASP-08 private-service relay"
|
||
);
|
||
return;
|
||
}
|
||
let Some(result_tx) = self.connect_attempt_result_tx.clone() else {
|
||
tracing::error!(relay = %relay_url, "Connection scheduler is not running");
|
||
return;
|
||
};
|
||
let Some(connection) = self.connections.get(&relay_url).cloned() else {
|
||
tracing::error!(relay = %relay_url, "No RelayConnection registered");
|
||
return;
|
||
};
|
||
|
||
let can_schedule = {
|
||
let mut index = self.relay_sync_index.write().await;
|
||
let can_schedule = index
|
||
.get(&relay_url)
|
||
.is_some_and(|state| state.connection_status == ConnectionStatus::Disconnected);
|
||
if can_schedule {
|
||
index
|
||
.get_mut(&relay_url)
|
||
.expect("relay state checked immediately above")
|
||
.connection_status = ConnectionStatus::Connecting;
|
||
}
|
||
can_schedule
|
||
};
|
||
if !can_schedule {
|
||
tracing::trace!(relay = %relay_url, "Relay does not need another connection attempt");
|
||
return;
|
||
}
|
||
|
||
let token = match reserve_connect_attempt(
|
||
&mut self.in_flight_connect_attempts,
|
||
&mut self.next_connect_attempt_token,
|
||
&relay_url,
|
||
) {
|
||
Some(token) => token,
|
||
None => {
|
||
tracing::trace!(relay = %relay_url, "Connection attempt already queued");
|
||
return;
|
||
}
|
||
};
|
||
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_status(&relay_url, ConnectionStatus::Connecting);
|
||
}
|
||
|
||
let health_tracker = Arc::clone(&self.health_tracker);
|
||
let semaphore = Arc::clone(&self.connect_attempt_semaphore);
|
||
let timeout = self.health_tracker.base_backoff_secs();
|
||
let private_mode = self.config.private_mode;
|
||
let Some(mut shutdown_rx) = self.shutdown_tx.as_ref().map(|sender| sender.subscribe())
|
||
else {
|
||
tracing::error!(relay = %relay_url, "Connection scheduler has no shutdown signal");
|
||
return;
|
||
};
|
||
tokio::spawn(async move {
|
||
let permit = tokio::select! {
|
||
permit = begin_connect_attempt(semaphore, health_tracker, &relay_url) => permit,
|
||
_ = shutdown_rx.recv() => return,
|
||
};
|
||
let Some(_permit) = permit else {
|
||
return;
|
||
};
|
||
let outcome = tokio::select! {
|
||
outcome = async {
|
||
// A GRASP-08 private service must be recognized before
|
||
// the dial so no WebSocket or AUTH exchange ever reaches
|
||
// it. Only public instances park, so only they pay the
|
||
// extra pre-dial probe; session hints still come from
|
||
// the post-connect fetch below, which runs on every
|
||
// attempt so they stay per-session.
|
||
if !private_mode && connection.preflight_limit_hints().await.grasp08 {
|
||
return ConnectAttemptOutcome::PrivateService;
|
||
}
|
||
match connection.connect(timeout).await {
|
||
Ok(()) => {
|
||
let hints = connection.fetch_limit_hints().await;
|
||
ConnectAttemptOutcome::Connected {
|
||
advertised_default_limit: hints.default_limit,
|
||
advertised_max_subscriptions: hints.max_subscriptions,
|
||
advertised_owner: hints.owner,
|
||
advertised_grasp08: hints.grasp08,
|
||
}
|
||
}
|
||
Err(error) => ConnectAttemptOutcome::Failed(error),
|
||
}
|
||
} => outcome,
|
||
_ = shutdown_rx.recv() => {
|
||
connection.disconnect().await;
|
||
return;
|
||
}
|
||
};
|
||
if result_tx
|
||
.send(ConnectAttemptResult {
|
||
relay_url,
|
||
token,
|
||
outcome,
|
||
})
|
||
.await
|
||
.is_err()
|
||
{
|
||
connection.disconnect().await;
|
||
}
|
||
});
|
||
}
|
||
|
||
async fn derive_targets(&self) -> HashMap<String, RelaySyncNeeds> {
|
||
let repo_index = self.repo_sync_index.read().await;
|
||
let mut targets = algorithms::derive_relay_targets(&repo_index);
|
||
discovery::merge_inbox_roots(&mut targets, &self.nip65_discovery.inbox_roots);
|
||
targets
|
||
}
|
||
|
||
/// Rebuild GRASP-08 membership from accepted-announcement sync state and
|
||
/// the latest NIP-11 owner learned for each referenced relay.
|
||
///
|
||
/// Purgatory announcements are deliberately excluded: they have not yet
|
||
/// passed repository admission and therefore cannot grant service access.
|
||
async fn reconcile_private_membership(&self) {
|
||
let Some(access) = &self.private_access else {
|
||
return;
|
||
};
|
||
let repo_index = self.repo_sync_index.read().await;
|
||
let accepted_relays: HashSet<String> = repo_index
|
||
.values()
|
||
.filter(|needs| needs.sync_level == SyncLevel::Full)
|
||
.flat_map(|needs| needs.relays.iter())
|
||
.filter_map(|relay| canonical_relay_key(relay).ok())
|
||
.collect();
|
||
drop(repo_index);
|
||
let members = effective_private_members(
|
||
&self.configured_private_members,
|
||
&accepted_relays,
|
||
&self.relay_owners,
|
||
);
|
||
if access.replace(members) {
|
||
tracing::info!(
|
||
configured_members = self.configured_private_members.len(),
|
||
accepted_relay_count = accepted_relays.len(),
|
||
effective_members = access.len(),
|
||
"Reconciled GRASP-08 service membership"
|
||
);
|
||
}
|
||
}
|
||
|
||
fn configured_nip65_fallback_relays(&self) -> HashSet<String> {
|
||
self.config
|
||
.parse_sync_plus_fallback_relays()
|
||
.into_iter()
|
||
.filter_map(|url| canonical_relay_key(&url).ok())
|
||
.collect()
|
||
}
|
||
|
||
async fn schedule_nip65_discovery(&mut self) {
|
||
// GRASP-03 is an optional overlay on the always-running GRASP-02
|
||
// manager. When disabled, do not inventory authors, retain identity
|
||
// authority, open discovery connections, or add mailbox roots.
|
||
if !self.config.sync_plus_enabled {
|
||
return;
|
||
}
|
||
|
||
let now = Instant::now();
|
||
if self
|
||
.nip65_discovery
|
||
.next_inventory_at
|
||
.is_none_or(|due| due <= now)
|
||
{
|
||
let announcements = self
|
||
.database
|
||
.query(Filter::new().kind(Kind::GitRepoAnnouncement))
|
||
.await
|
||
.unwrap_or_default();
|
||
let candidates = self.root_candidate_index.read().await;
|
||
let repo_index = self.repo_sync_index.read().await;
|
||
let roots = discovery::accepted_root_candidates(candidates.values(), &repo_index);
|
||
let mut eligible_authors =
|
||
discovery::accepted_repository_authors(announcements.iter(), &repo_index);
|
||
drop(candidates);
|
||
drop(repo_index);
|
||
let root_ids: HashSet<EventId> = roots.iter().map(|root| root.id).collect();
|
||
self.nip65_discovery.root_repositories = roots
|
||
.iter()
|
||
.map(|root| (root.id, root.repository.clone()))
|
||
.collect();
|
||
self.nip65_discovery.root_author_roots =
|
||
discovery::mailbox_author_roots(&roots, std::iter::empty());
|
||
|
||
// A root may be absent from the author's write relays while a
|
||
// participant's reaction, zap or reply is present there. Carry
|
||
// root provenance through indirect event/address descendants so
|
||
// each mailbox probe stays scoped to accepted threads in which
|
||
// that author actually participated.
|
||
let frontier = recursive_descendant_frontier(
|
||
&self.database,
|
||
&root_ids,
|
||
self.config.sync_recursive_descendant_limit,
|
||
)
|
||
.await;
|
||
let participant_author_count = frontier.author_roots.len();
|
||
let author_roots = discovery::mailbox_author_roots(&roots, frontier.author_roots);
|
||
eligible_authors.extend(author_roots.keys().copied());
|
||
self.nip65_discovery.author_roots = author_roots;
|
||
self.nip65_discovery.eligible_authors = eligible_authors;
|
||
self.proactive_participant_authors
|
||
.replace(self.nip65_discovery.eligible_authors.clone())
|
||
.await;
|
||
let current_authors = self.nip65_discovery.eligible_authors.clone();
|
||
|
||
// Rebuild mailbox ownership from already accepted local state before
|
||
// attempting a network refresh. GRASP-03 retains kind 10002 so a
|
||
// restart must not make conversation coverage depend on the index
|
||
// source still being reachable.
|
||
let retained_identity = if current_authors.is_empty() {
|
||
Some(std::collections::BTreeSet::<Event>::new())
|
||
} else {
|
||
self.database
|
||
.query(
|
||
Filter::new()
|
||
.kinds([Kind::Metadata, Kind::RelayList])
|
||
.authors(current_authors.iter().copied())
|
||
.limit(current_authors.len().saturating_mul(2)),
|
||
)
|
||
.await
|
||
.ok()
|
||
};
|
||
if let Some(retained_identity) = retained_identity {
|
||
let latest_identity = discovery::latest_identity_events(
|
||
retained_identity.iter().cloned(),
|
||
¤t_authors,
|
||
);
|
||
self.nip65_discovery.relay_lists =
|
||
discovery::latest_relay_lists(latest_identity, ¤t_authors);
|
||
self.nip65_discovery.author_inboxes = self
|
||
.nip65_discovery
|
||
.relay_lists
|
||
.iter()
|
||
.map(|(author, event)| (*author, discovery::inbox_relays(event)))
|
||
.collect();
|
||
self.nip65_discovery.author_mailboxes = self
|
||
.nip65_discovery
|
||
.relay_lists
|
||
.iter()
|
||
.map(|(author, event)| (*author, discovery::mailbox_relays(event)))
|
||
.collect();
|
||
}
|
||
|
||
let mut index_relays: HashSet<String> = self
|
||
.config
|
||
.parse_user_index_relays()
|
||
.into_iter()
|
||
.filter_map(|url| canonical_relay_key(&url).ok())
|
||
.collect();
|
||
if let Some(bootstrap) = self
|
||
.bootstrap_relay_url
|
||
.as_deref()
|
||
.and_then(|url| canonical_relay_key(url).ok())
|
||
{
|
||
index_relays.insert(bootstrap);
|
||
}
|
||
self.nip65_discovery.author_sources.clear();
|
||
for author in ¤t_authors {
|
||
let sources = self
|
||
.nip65_discovery
|
||
.author_sources
|
||
.entry(*author)
|
||
.or_default();
|
||
if let Some(relay_list) = self.nip65_discovery.relay_lists.get(author) {
|
||
sources.extend(
|
||
discovery::outbox_relays(relay_list)
|
||
.into_iter()
|
||
.filter_map(|url| canonical_relay_key(&url).ok()),
|
||
);
|
||
} else {
|
||
// Index/bootstrap relays only discover the first accepted
|
||
// list. Once present, its outboxes become the authoritative
|
||
// additive refresh graph and avoid redundant index traffic.
|
||
sources.extend(index_relays.iter().cloned());
|
||
}
|
||
}
|
||
self.nip65_discovery
|
||
.relay_lists
|
||
.retain(|author, _| current_authors.contains(author));
|
||
self.nip65_discovery
|
||
.author_inboxes
|
||
.retain(|author, _| current_authors.contains(author));
|
||
self.nip65_discovery
|
||
.author_mailboxes
|
||
.retain(|author, _| current_authors.contains(author));
|
||
self.nip65_discovery.fallback_authors.retain(|author| {
|
||
current_authors.contains(author)
|
||
&& !self.nip65_discovery.relay_lists.contains_key(author)
|
||
});
|
||
let current_sources = &self.nip65_discovery.author_sources;
|
||
self.nip65_discovery
|
||
.next_attempt_at
|
||
.retain(|(relay, author), _| {
|
||
current_sources
|
||
.get(author)
|
||
.is_some_and(|sources| sources.contains(relay))
|
||
});
|
||
self.nip65_discovery.next_inventory_at = Some(now + nip65_inventory_interval());
|
||
let fallback_relays = self.configured_nip65_fallback_relays();
|
||
let inbox_overlay = discovery::build_inbox_root_overlay(
|
||
&self.nip65_discovery.root_author_roots,
|
||
&self.nip65_discovery.author_inboxes,
|
||
&self.nip65_discovery.fallback_authors,
|
||
&fallback_relays,
|
||
);
|
||
let mailbox_overlay = discovery::build_inbox_root_overlay(
|
||
&self.nip65_discovery.author_roots,
|
||
&self.nip65_discovery.author_mailboxes,
|
||
&self.nip65_discovery.fallback_authors,
|
||
&fallback_relays,
|
||
);
|
||
self.install_nip65_inbox_overlay(inbox_overlay).await;
|
||
self.install_nip65_mailbox_overlay(mailbox_overlay);
|
||
tracing::info!(
|
||
root_count = root_ids.len(),
|
||
participant_author_count,
|
||
eligible_author_count = self.nip65_discovery.eligible_authors.len(),
|
||
mailbox_relay_count = self.nip65_discovery.mailbox_roots.len(),
|
||
fallback_authors = self.nip65_discovery.fallback_authors.len(),
|
||
"Reconciled proactive participant mailbox inventory"
|
||
);
|
||
}
|
||
|
||
// Discovery is deliberately single-flight. The immediate-capacity
|
||
// sample below can race with historic admission before fetch_events
|
||
// acquires its permit, but that race can create at most one waiter,
|
||
// never a maintenance-tick-sized queue of control-plane tasks.
|
||
if !self.nip65_discovery.in_flight.is_empty() {
|
||
return;
|
||
}
|
||
|
||
let mut candidates: HashMap<String, Vec<PublicKey>> = HashMap::new();
|
||
for (author, sources) in &self.nip65_discovery.author_sources {
|
||
for source in sources {
|
||
let key = (source.clone(), *author);
|
||
if self.nip65_discovery.in_flight.contains(&key)
|
||
|| self
|
||
.nip65_discovery
|
||
.next_attempt_at
|
||
.get(&key)
|
||
.is_some_and(|due| *due > now)
|
||
{
|
||
continue;
|
||
}
|
||
candidates.entry(source.clone()).or_default().push(*author);
|
||
}
|
||
}
|
||
let mut sources: Vec<String> = candidates.keys().cloned().collect();
|
||
sources.sort();
|
||
for source in sources {
|
||
let Some(connection) = self.connections.get(&source).cloned() else {
|
||
continue;
|
||
};
|
||
if !connection.has_immediate_transient_capacity().await {
|
||
continue;
|
||
}
|
||
let mut authors = candidates.remove(&source).unwrap_or_default();
|
||
authors.sort_by_key(PublicKey::to_hex);
|
||
authors.truncate(NIP65_DISCOVERY_BATCH_AUTHORS);
|
||
let authors: HashSet<PublicKey> = authors.into_iter().collect();
|
||
if authors.is_empty() {
|
||
continue;
|
||
}
|
||
self.nip65_discovery
|
||
.in_flight
|
||
.extend(authors.iter().map(|author| (source.clone(), *author)));
|
||
let filter = Filter::new()
|
||
.kinds([Kind::Metadata, Kind::RelayList])
|
||
.authors(authors.iter().copied())
|
||
.limit(authors.len().saturating_mul(2));
|
||
let Some(result_tx) = self.nip65_discovery_result_tx.clone() else {
|
||
return;
|
||
};
|
||
let source_relay = source.clone();
|
||
tokio::spawn(async move {
|
||
let outcome = connection
|
||
.fetch_events(filter, Duration::from_secs(30))
|
||
.await;
|
||
let _ = result_tx
|
||
.send(Nip65DiscoveryResult {
|
||
source_relay,
|
||
authors,
|
||
outcome,
|
||
})
|
||
.await;
|
||
});
|
||
break;
|
||
}
|
||
|
||
// No connected source could accept the single discovery query. Dial
|
||
// at most one new source per maintenance pass; registering the entire
|
||
// NIP-65 graph at once would turn a bounded query lane into an
|
||
// unbounded connection fan-out.
|
||
if self.nip65_discovery.in_flight.is_empty() {
|
||
if let Some(source) = candidates
|
||
.keys()
|
||
.filter(|source| !self.connections.contains_key(*source))
|
||
.min()
|
||
.cloned()
|
||
{
|
||
if self.register_relay(source.clone(), false, true).await {
|
||
self.schedule_connect_relay(&source).await;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
async fn install_nip65_inbox_overlay(
|
||
&mut self,
|
||
new_overlay: HashMap<String, HashSet<EventId>>,
|
||
) {
|
||
let old_overlay = std::mem::replace(&mut self.nip65_discovery.inbox_roots, new_overlay);
|
||
if old_overlay == self.nip65_discovery.inbox_roots {
|
||
return;
|
||
}
|
||
let mut dirty_relays: HashSet<String> = old_overlay.keys().cloned().collect();
|
||
dirty_relays.extend(self.nip65_discovery.inbox_roots.keys().cloned());
|
||
|
||
for relay in dirty_relays {
|
||
// Preserve the established root-author inbox behavior: additions
|
||
// enter ordinary live/rotating descendant coverage, while
|
||
// removal-only changes drain through that coverage's normal
|
||
// lifecycle instead of churning shared subscriptions.
|
||
self.recompute_new_sync_filters_for_relay(&relay).await;
|
||
}
|
||
}
|
||
|
||
fn install_nip65_mailbox_overlay(&mut self, new_overlay: HashMap<String, HashSet<EventId>>) {
|
||
self.nip65_discovery
|
||
.install_mailbox_overlay(new_overlay, Instant::now());
|
||
}
|
||
|
||
async fn schedule_mailbox_probe(&mut self) {
|
||
if !self.config.sync_plus_enabled {
|
||
return;
|
||
}
|
||
|
||
let now = Instant::now();
|
||
let due_relays: Vec<String> = due_mailbox_relays(
|
||
&self.nip65_discovery.mailbox_roots,
|
||
&self.nip65_discovery.mailbox_probe_next_at,
|
||
now,
|
||
)
|
||
.into_iter()
|
||
.filter(|relay| {
|
||
!self
|
||
.nip65_discovery
|
||
.mailbox_probes_in_flight
|
||
.contains(relay)
|
||
})
|
||
.collect();
|
||
let active_relays: HashSet<String> = self
|
||
.relay_sync_index
|
||
.read()
|
||
.await
|
||
.iter()
|
||
.filter(|(_, state)| state.connection_status.is_live_sync_active())
|
||
.map(|(relay, _)| relay.clone())
|
||
.collect();
|
||
let Some(relay) = select_due_mailbox_relay(&due_relays, &active_relays) else {
|
||
return;
|
||
};
|
||
if is_own_sync_target(&relay, &self.service_domain)
|
||
|| self.rejected_relay_targets.contains(&relay)
|
||
{
|
||
self.nip65_discovery
|
||
.mailbox_probe_next_at
|
||
.insert(relay, now + mailbox_probe_refresh_interval());
|
||
return;
|
||
}
|
||
|
||
if !self.connections.contains_key(&relay) {
|
||
if self.register_relay(relay.clone(), false, true).await {
|
||
self.schedule_connect_relay(&relay).await;
|
||
}
|
||
self.defer_mailbox_relay(&relay);
|
||
return;
|
||
}
|
||
|
||
let Some(connection) = self.connections.get(&relay).cloned() else {
|
||
self.defer_mailbox_relay(&relay);
|
||
return;
|
||
};
|
||
let socket_connected = connection.is_connected().await;
|
||
let connection_status = self
|
||
.relay_sync_index
|
||
.read()
|
||
.await
|
||
.get(&relay)
|
||
.map(|state| state.connection_status);
|
||
if !mailbox_probe_connection_ready(connection_status, socket_connected) {
|
||
if !socket_connected && self.health_tracker.should_attempt_connection(&relay) {
|
||
self.schedule_connect_relay(&relay).await;
|
||
} else if !socket_connected {
|
||
self.retire_idle_nip65_discovery_source(&relay).await;
|
||
}
|
||
self.defer_mailbox_relay(&relay);
|
||
return;
|
||
}
|
||
|
||
let roots = self.nip65_discovery.mailbox_roots[&relay].clone();
|
||
let repositories: HashSet<String> = roots
|
||
.iter()
|
||
.filter_map(|root| self.nip65_discovery.root_repositories.get(root).cloned())
|
||
.collect();
|
||
let frontier = recursive_descendant_frontier(
|
||
&self.database,
|
||
&roots,
|
||
self.config.sync_recursive_descendant_limit,
|
||
)
|
||
.await;
|
||
let mut filters = mailbox_probe_filters(&repositories, &roots, &frontier);
|
||
filters.sort_by_key(Filter::as_json);
|
||
let Some(filter_count) = (!filters.is_empty()).then_some(filters.len()) else {
|
||
self.nip65_discovery
|
||
.mailbox_probe_next_at
|
||
.insert(relay.clone(), now + mailbox_probe_refresh_interval());
|
||
self.retire_idle_nip65_discovery_source(&relay).await;
|
||
return;
|
||
};
|
||
let filter_index = self
|
||
.nip65_discovery
|
||
.mailbox_probe_next_filter
|
||
.get(&relay)
|
||
.copied()
|
||
.unwrap_or_default()
|
||
% filter_count;
|
||
let filter = filters.swap_remove(filter_index);
|
||
let Some(result_tx) = self.mailbox_probe_result_tx.clone() else {
|
||
self.defer_mailbox_relay(&relay);
|
||
return;
|
||
};
|
||
self.nip65_discovery
|
||
.mailbox_probes_in_flight
|
||
.insert(relay.clone());
|
||
let source_relay = relay.clone();
|
||
tokio::spawn(async move {
|
||
let outcome = fetch_mailbox_filter(connection, filter).await;
|
||
let _ = result_tx
|
||
.send(MailboxProbeResult {
|
||
source_relay,
|
||
filter_index,
|
||
filter_count,
|
||
outcome,
|
||
})
|
||
.await;
|
||
});
|
||
tracing::info!(
|
||
relay = %relay,
|
||
filter_index,
|
||
filter_count,
|
||
"Started bounded participant mailbox fetch"
|
||
);
|
||
}
|
||
|
||
fn defer_mailbox_relay(&mut self, relay: &str) {
|
||
if self.nip65_discovery.mailbox_roots.contains_key(relay) {
|
||
self.nip65_discovery.mailbox_probe_next_at.insert(
|
||
relay.to_string(),
|
||
Instant::now() + mailbox_probe_retry_interval(),
|
||
);
|
||
}
|
||
}
|
||
|
||
async fn handle_mailbox_probe_result(&mut self, result: MailboxProbeResult) {
|
||
if !self
|
||
.nip65_discovery
|
||
.mailbox_probes_in_flight
|
||
.remove(&result.source_relay)
|
||
{
|
||
tracing::debug!(
|
||
relay = %result.source_relay,
|
||
"Ignoring stale participant mailbox result"
|
||
);
|
||
return;
|
||
}
|
||
|
||
let (events, succeeded) = match result.outcome {
|
||
Ok(events) => (events, true),
|
||
Err(error) => {
|
||
tracing::debug!(relay = %result.source_relay, %error, "Participant mailbox fetch failed");
|
||
(Vec::new(), false)
|
||
}
|
||
};
|
||
let event_count = events.len();
|
||
for event in events {
|
||
let _ = Self::process_event_static(
|
||
&event,
|
||
&result.source_relay,
|
||
&self.database,
|
||
&self.write_policy,
|
||
&self.local_relay,
|
||
&self.rejected_events_index,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
}
|
||
|
||
let (next_filter, completed_cycle, next_probe_in) =
|
||
mailbox_probe_completion(result.filter_index, result.filter_count, succeeded);
|
||
self.nip65_discovery.record_probe_completion(
|
||
&result.source_relay,
|
||
next_filter,
|
||
next_probe_in,
|
||
Instant::now(),
|
||
);
|
||
tracing::info!(
|
||
relay = %result.source_relay,
|
||
filter_index = result.filter_index,
|
||
filter_count = result.filter_count,
|
||
event_count,
|
||
succeeded,
|
||
completed_cycle,
|
||
next_probe_in_secs = next_probe_in.as_secs_f64(),
|
||
"Participant mailbox fetch reached terminal state"
|
||
);
|
||
self.retire_idle_nip65_discovery_source(&result.source_relay)
|
||
.await;
|
||
}
|
||
|
||
async fn handle_nip65_discovery_result(&mut self, result: Nip65DiscoveryResult) {
|
||
for author in &result.authors {
|
||
self.nip65_discovery
|
||
.in_flight
|
||
.remove(&(result.source_relay.clone(), *author));
|
||
}
|
||
let (events, query_succeeded) = match result.outcome {
|
||
Ok(events) => (events, true),
|
||
Err(error) => {
|
||
tracing::debug!(relay = %result.source_relay, %error, "NIP-65 discovery query failed");
|
||
(Vec::new(), false)
|
||
}
|
||
};
|
||
let selected = discovery::latest_identity_events(events, &result.authors);
|
||
let mut represented_by_source = HashSet::new();
|
||
for event in selected {
|
||
let process_result = Self::process_event_static(
|
||
&event,
|
||
&result.source_relay,
|
||
&self.database,
|
||
&self.write_policy,
|
||
&self.local_relay,
|
||
&self.rejected_events_index,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
if event.kind == Kind::RelayList
|
||
&& matches!(
|
||
process_result,
|
||
ProcessResult::Saved | ProcessResult::Duplicate
|
||
)
|
||
{
|
||
represented_by_source.insert(event.pubkey);
|
||
}
|
||
}
|
||
// Drive relay ownership only from the accepted database. A source can
|
||
// return a validly signed but policy-ineligible relay list; processing
|
||
// its raw response must not be enough to make us dial its URLs.
|
||
let stored_relay_lists = self
|
||
.database
|
||
.query(
|
||
Filter::new()
|
||
.kind(Kind::RelayList)
|
||
.authors(result.authors.iter().copied())
|
||
.limit(result.authors.len()),
|
||
)
|
||
.await
|
||
.unwrap_or_default();
|
||
let latest = discovery::latest_relay_lists(stored_relay_lists, &result.authors);
|
||
let authors_with_lists: HashSet<PublicKey> = latest.keys().copied().collect();
|
||
let previous_fallback_authors = self.nip65_discovery.fallback_authors.clone();
|
||
let now = Instant::now();
|
||
for author in &result.authors {
|
||
let retry_after =
|
||
nip65_author_retry_after(query_succeeded, &represented_by_source, author);
|
||
self.nip65_discovery
|
||
.next_attempt_at
|
||
.insert((result.source_relay.clone(), *author), now + retry_after);
|
||
if authors_with_lists.contains(author) {
|
||
self.nip65_discovery.fallback_authors.remove(author);
|
||
} else if query_succeeded && self.nip65_discovery.author_roots.contains_key(author) {
|
||
self.nip65_discovery.fallback_authors.insert(*author);
|
||
}
|
||
}
|
||
|
||
let mut changed = previous_fallback_authors != self.nip65_discovery.fallback_authors;
|
||
for (author, candidate) in latest {
|
||
if !self.nip65_discovery.eligible_authors.contains(&author) {
|
||
continue;
|
||
}
|
||
let replace = self
|
||
.nip65_discovery
|
||
.relay_lists
|
||
.get(&author)
|
||
.is_none_or(|current| {
|
||
candidate.created_at > current.created_at
|
||
|| (candidate.created_at == current.created_at && candidate.id < current.id)
|
||
});
|
||
if replace {
|
||
self.nip65_discovery
|
||
.author_inboxes
|
||
.insert(author, discovery::inbox_relays(&candidate));
|
||
self.nip65_discovery
|
||
.author_mailboxes
|
||
.insert(author, discovery::mailbox_relays(&candidate));
|
||
let mut index_relays: HashSet<String> = self
|
||
.config
|
||
.parse_user_index_relays()
|
||
.into_iter()
|
||
.filter_map(|url| canonical_relay_key(&url).ok())
|
||
.collect();
|
||
if let Some(bootstrap) = self
|
||
.bootstrap_relay_url
|
||
.as_deref()
|
||
.and_then(|url| canonical_relay_key(url).ok())
|
||
{
|
||
index_relays.insert(bootstrap);
|
||
}
|
||
let sources = self
|
||
.nip65_discovery
|
||
.author_sources
|
||
.entry(author)
|
||
.or_default();
|
||
sources.retain(|source| !index_relays.contains(source));
|
||
sources.extend(
|
||
discovery::outbox_relays(&candidate)
|
||
.into_iter()
|
||
.filter_map(|url| canonical_relay_key(&url).ok()),
|
||
);
|
||
self.nip65_discovery.relay_lists.insert(author, candidate);
|
||
changed = true;
|
||
}
|
||
}
|
||
if !changed {
|
||
self.retire_idle_nip65_discovery_source(&result.source_relay)
|
||
.await;
|
||
self.schedule_nip65_discovery().await;
|
||
return;
|
||
}
|
||
let fallback_relays = self.configured_nip65_fallback_relays();
|
||
let inbox_overlay = discovery::build_inbox_root_overlay(
|
||
&self.nip65_discovery.root_author_roots,
|
||
&self.nip65_discovery.author_inboxes,
|
||
&self.nip65_discovery.fallback_authors,
|
||
&fallback_relays,
|
||
);
|
||
let mailbox_overlay = discovery::build_inbox_root_overlay(
|
||
&self.nip65_discovery.author_roots,
|
||
&self.nip65_discovery.author_mailboxes,
|
||
&self.nip65_discovery.fallback_authors,
|
||
&fallback_relays,
|
||
);
|
||
self.install_nip65_inbox_overlay(inbox_overlay).await;
|
||
self.install_nip65_mailbox_overlay(mailbox_overlay);
|
||
tracing::info!(
|
||
source = %result.source_relay,
|
||
authors = result.authors.len(),
|
||
mailbox_relays = self.nip65_discovery.mailbox_roots.len(),
|
||
fallback_authors = self.nip65_discovery.fallback_authors.len(),
|
||
"Updated proactive participant mailbox coverage from NIP-65"
|
||
);
|
||
|
||
self.retire_idle_nip65_discovery_source(&result.source_relay)
|
||
.await;
|
||
// Continue the single-flight round immediately. The health timer is a
|
||
// safety net, but ordinary sync work may hold the actor long enough
|
||
// that waiting for its next tick needlessly delays a newly discovered
|
||
// outbox.
|
||
self.schedule_nip65_discovery().await;
|
||
}
|
||
|
||
async fn retire_idle_nip65_discovery_source(&mut self, source: &str) {
|
||
if !self.nip65_discovery_only_relays.contains(source) {
|
||
return;
|
||
}
|
||
let now = Instant::now();
|
||
let has_author_work =
|
||
self.nip65_discovery
|
||
.author_sources
|
||
.iter()
|
||
.any(|(author, sources)| {
|
||
sources.contains(source)
|
||
&& (self
|
||
.nip65_discovery
|
||
.in_flight
|
||
.contains(&(source.to_string(), *author))
|
||
|| self
|
||
.nip65_discovery
|
||
.next_attempt_at
|
||
.get(&(source.to_string(), *author))
|
||
.is_none_or(|due| *due <= now))
|
||
});
|
||
let has_mailbox_work = self
|
||
.nip65_discovery
|
||
.mailbox_probes_in_flight
|
||
.contains(source)
|
||
|| (self.nip65_discovery.mailbox_roots.contains_key(source)
|
||
&& self
|
||
.nip65_discovery
|
||
.mailbox_probe_next_at
|
||
.get(source)
|
||
.is_none_or(|due| *due <= now));
|
||
if !has_author_work && !has_mailbox_work && !self.has_pending_batches(source).await {
|
||
tracing::debug!(
|
||
relay = %source,
|
||
"Retiring idle NIP-65 control-plane connection"
|
||
);
|
||
self.disconnect_relay(source).await;
|
||
}
|
||
}
|
||
|
||
async fn handle_connect_attempt_result(&mut self, result: ConnectAttemptResult) {
|
||
if !take_connect_attempt(
|
||
&mut self.in_flight_connect_attempts,
|
||
&result.relay_url,
|
||
result.token,
|
||
) {
|
||
tracing::debug!(
|
||
relay = %result.relay_url,
|
||
token = result.token.0,
|
||
"Ignoring stale connection attempt result"
|
||
);
|
||
return;
|
||
}
|
||
|
||
let still_connecting = self
|
||
.relay_sync_index
|
||
.read()
|
||
.await
|
||
.get(&result.relay_url)
|
||
.is_some_and(|state| state.connection_status == ConnectionStatus::Connecting);
|
||
if !still_connecting {
|
||
tracing::debug!(
|
||
relay = %result.relay_url,
|
||
token = result.token.0,
|
||
"Ignoring connection result after lifecycle state changed"
|
||
);
|
||
return;
|
||
}
|
||
|
||
let outcome = if matches!(result.outcome, ConnectAttemptOutcome::Connected { .. }) {
|
||
let still_connected = if let Some(connection) = self.connections.get(&result.relay_url)
|
||
{
|
||
connection.is_connected().await
|
||
} else {
|
||
false
|
||
};
|
||
if still_connected {
|
||
result.outcome
|
||
} else {
|
||
tracing::warn!(
|
||
relay = %result.relay_url,
|
||
"Rejecting stale connection success after relay disconnected during setup"
|
||
);
|
||
ConnectAttemptOutcome::Failed(
|
||
"Relay disconnected before connection setup completed".to_string(),
|
||
)
|
||
}
|
||
} else {
|
||
result.outcome
|
||
};
|
||
|
||
match outcome {
|
||
ConnectAttemptOutcome::Connected {
|
||
advertised_default_limit,
|
||
advertised_max_subscriptions,
|
||
advertised_owner,
|
||
advertised_grasp08,
|
||
} => {
|
||
// Only GRASP-08-advertising relays mint derived private
|
||
// membership: a public relay's owner gains nothing legitimate
|
||
// from private membership because their relay enforces no
|
||
// confidentiality for the mirrored repositories.
|
||
match advertised_owner.filter(|_| advertised_grasp08) {
|
||
Some(owner) => {
|
||
self.relay_owners.insert(result.relay_url.clone(), owner);
|
||
}
|
||
None => {
|
||
self.relay_owners.remove(&result.relay_url);
|
||
}
|
||
}
|
||
self.reconcile_private_membership().await;
|
||
if let Some(connection) = self.connections.get(&result.relay_url) {
|
||
connection.reset_subscription_budget(advertised_max_subscriptions);
|
||
}
|
||
// A session begins only after a successful WebSocket connection. Replacing this
|
||
// entry on every connection result resets learned caps and re-applies the NIP-11
|
||
// hint fetched for that exact session.
|
||
self.pagination_sessions.insert(
|
||
result.relay_url.clone(),
|
||
RelayPaginationSession::new(advertised_default_limit),
|
||
);
|
||
self.health_tracker.record_success(&result.relay_url);
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_attempt(&result.relay_url, true);
|
||
}
|
||
if self
|
||
.nip65_discovery
|
||
.mailbox_roots
|
||
.contains_key(&result.relay_url)
|
||
{
|
||
self.nip65_discovery
|
||
.mailbox_probe_next_at
|
||
.insert(result.relay_url.clone(), Instant::now());
|
||
}
|
||
self.handle_connect_or_reconnect(&result.relay_url).await;
|
||
}
|
||
ConnectAttemptOutcome::PrivateService => {
|
||
if self.private_service_relays.insert(result.relay_url.clone()) {
|
||
tracing::warn!(
|
||
relay = %result.relay_url,
|
||
"Relay advertises GRASP-08 private service; excluding it from public sync"
|
||
);
|
||
}
|
||
// Retire the target like `complete_ended_session`, but without
|
||
// the re-registration path: the park is permanent for this
|
||
// process. No connected gauge to decrement - we never dialed.
|
||
self.relay_sync_index
|
||
.write()
|
||
.await
|
||
.remove(&result.relay_url);
|
||
self.pending_sync_index
|
||
.write()
|
||
.await
|
||
.remove(&result.relay_url);
|
||
self.connections.remove(&result.relay_url);
|
||
self.nip65_discovery_only_relays.remove(&result.relay_url);
|
||
self.health_tracker.forget_relay(&result.relay_url);
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.forget_relay(&result.relay_url);
|
||
}
|
||
}
|
||
ConnectAttemptOutcome::Failed(error) => {
|
||
if let Some(category) = naughty_list::NaughtyListTracker::classify_error(&error) {
|
||
if let Some(ref naughty_list) = self.health_tracker.naughty_list() {
|
||
let is_new =
|
||
naughty_list.record(&result.relay_url, category, error.clone());
|
||
if is_new {
|
||
tracing::warn!(
|
||
relay = %result.relay_url,
|
||
category = ?category,
|
||
error = %error,
|
||
"Relay has persistent configuration issue, added to naughty list"
|
||
);
|
||
} else {
|
||
tracing::debug!(
|
||
relay = %result.relay_url,
|
||
category = ?category,
|
||
"Naughty relay failure (already tracked)"
|
||
);
|
||
}
|
||
}
|
||
} else {
|
||
tracing::debug!(
|
||
relay = %result.relay_url,
|
||
error = %error,
|
||
"Connection failed (transient issue, backoff active)"
|
||
);
|
||
}
|
||
self.record_connection_attempt_failure(&result.relay_url)
|
||
.await;
|
||
}
|
||
}
|
||
}
|
||
|
||
async fn record_connection_attempt_failure(&self, relay_url: &str) {
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
if let Some(state) = index.get_mut(relay_url) {
|
||
state.connection_status = ConnectionStatus::Disconnected;
|
||
}
|
||
}
|
||
self.health_tracker.record_failure(relay_url);
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_attempt(relay_url, false);
|
||
metrics.record_connection_status(relay_url, ConnectionStatus::Disconnected);
|
||
metrics.record_health_state(relay_url, self.health_tracker.get_state(relay_url));
|
||
}
|
||
}
|
||
|
||
/// Recompute sync actions for a specific relay
|
||
///
|
||
/// Uses derive_relay_targets and compute_actions to find new items
|
||
/// that need to be synced. Processes AddFilters actions for new items.
|
||
async fn recompute_new_sync_filters_for_relay(&mut self, relay_url: &str) {
|
||
use crate::sync::algorithms::compute_actions;
|
||
|
||
let relay_url = match canonical_relay_key(relay_url) {
|
||
Ok(relay_url) => relay_url,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
error = %error,
|
||
"Ignoring invalid dirty relay signal"
|
||
);
|
||
return;
|
||
}
|
||
};
|
||
|
||
// Get current state from indexes (need to collect to avoid holding locks)
|
||
let all_targets = self.derive_targets().await;
|
||
|
||
// Filter to only targets for this specific relay
|
||
let relay_target = match all_targets.get(&relay_url) {
|
||
Some(target) => target.clone(),
|
||
None => {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"No sync targets found for relay"
|
||
);
|
||
return;
|
||
}
|
||
};
|
||
|
||
// Build single-relay targets map for compute_actions
|
||
let mut single_relay_targets = std::collections::HashMap::new();
|
||
single_relay_targets.insert(relay_url.clone(), relay_target);
|
||
|
||
// Compute actions for new items
|
||
let actions = {
|
||
let pending_index = self.pending_sync_index.read().await;
|
||
let relay_index = self.relay_sync_index.read().await;
|
||
compute_actions(&single_relay_targets, &pending_index, &relay_index)
|
||
};
|
||
|
||
if actions.is_empty() {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"No new items to sync for relay"
|
||
);
|
||
return;
|
||
}
|
||
|
||
// Process each action
|
||
for action in actions {
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
new_full_repos = action.items.repos.len(),
|
||
new_state_only_repos = action.items.state_only_repos.len(),
|
||
new_root_events = action.items.root_events.len(),
|
||
filters = action.filters.len(),
|
||
"Processing AddFilters for new items"
|
||
);
|
||
self.handle_new_sync_filters(action).await;
|
||
}
|
||
}
|
||
|
||
/// Sync purgatory announcements into repo_sync_index as StateOnly entries.
|
||
///
|
||
/// Called periodically by the purgatory announcement sync timer (every 5s).
|
||
/// For each announcement currently in purgatory, ensures a `StateOnly` entry
|
||
/// exists in `repo_sync_index`. New entries are then picked up by
|
||
/// `handle_new_sync_filters` which connects to listed relay URLs and subscribes
|
||
/// to state events for that repo.
|
||
///
|
||
/// Idempotent: existing entries are not downgraded (a promoted Full entry stays Full).
|
||
async fn sync_purgatory_announcements_to_index(&mut self) {
|
||
use crate::sync::algorithms::compute_actions;
|
||
|
||
// Collect all purgatory announcements (snapshot - no async holds)
|
||
let announcements = self.purgatory.announcements_for_sync();
|
||
let announcement_events = self.purgatory.announcement_events_for_sync();
|
||
|
||
// Replace StateOnly relay ownership from the current purgatory
|
||
// snapshot and remove entries only after full expiry. Soft-expired
|
||
// announcements remain in the snapshot for their 24-hour revival
|
||
// window; promoted Full entries are never downgraded or removed here.
|
||
let dirty_relays = {
|
||
let mut index = self.repo_sync_index.write().await;
|
||
reconcile_purgatory_relay_ownership(&mut index, &announcements)
|
||
};
|
||
for relay_url in dirty_relays {
|
||
self.recompute_new_sync_filters_for_relay(&relay_url).await;
|
||
}
|
||
|
||
if announcements.is_empty() {
|
||
return;
|
||
}
|
||
|
||
// A maintainer announcement or state may have been fetched and rejected
|
||
// before the reciprocal owner announcement reached this relay. Retry those
|
||
// dependencies after connecting the owner's relay hints so cold-cache
|
||
// misses can be fetched directly by event ID.
|
||
let dependency_events = select_purgatory_dependency_events(
|
||
announcement_events,
|
||
&mut self.purgatory_dependency_attempts,
|
||
Instant::now(),
|
||
purgatory_dependency_retry_after(),
|
||
MAX_PURGATORY_DEPENDENCY_EVENTS_PER_TICK,
|
||
);
|
||
let mut dependency_refetch_batches = HashMap::new();
|
||
for event in dependency_events {
|
||
self.reprocess_purgatory_announcement_dependencies(
|
||
&event,
|
||
&mut dependency_refetch_batches,
|
||
)
|
||
.await;
|
||
}
|
||
self.spawn_batched_purgatory_dependency_refetch(dependency_refetch_batches);
|
||
|
||
// Recompute all outstanding actions on every pass. A bounded pass may
|
||
// intentionally defer some relays, so restricting this to newly seen
|
||
// URLs would lose that work on the next tick.
|
||
let all_targets = self.derive_targets().await;
|
||
let mut actions = {
|
||
let pending_index = self.pending_sync_index.read().await;
|
||
let relay_index = self.relay_sync_index.read().await;
|
||
compute_actions(&all_targets, &pending_index, &relay_index)
|
||
};
|
||
// Policy-rejected targets must not consume the bounded per-tick
|
||
// budget, or a stored event listing forbidden relays could starve
|
||
// legitimate ones. Compare canonical keys: the rejected set stores
|
||
// them, while actions still carry raw event-supplied URLs.
|
||
actions.retain(|action| match canonical_relay_key(&action.relay_url) {
|
||
Ok(key) => !self.rejected_relay_targets.contains(&key),
|
||
Err(_) => false,
|
||
});
|
||
actions.sort_by_key(|action| {
|
||
!self
|
||
.dependency_relay_deadlines
|
||
.contains_key(&action.relay_url)
|
||
});
|
||
|
||
for action in actions
|
||
.into_iter()
|
||
.take(MAX_PURGATORY_FILTER_ACTIONS_PER_TICK)
|
||
{
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
full_repos = action.items.repos.len(),
|
||
state_only_repos = action.items.state_only_repos.len(),
|
||
"Purgatory sync timer: processing one bounded relay action"
|
||
);
|
||
self.handle_new_sync_filters(action).await;
|
||
}
|
||
}
|
||
|
||
/// Retry events whose authorization depends on a newly admitted owner announcement.
|
||
///
|
||
/// This mirrors the accepted-announcement dependency handling in
|
||
/// `process_event_static`, but runs while the owner announcement is still in
|
||
/// purgatory. Announcement policy already treats purgatory announcements as
|
||
/// maintainer authority; doing the retry here makes arrival order irrelevant.
|
||
async fn reprocess_purgatory_announcement_dependencies(
|
||
&mut self,
|
||
event: &Event,
|
||
dependency_refetch_batches: &mut HashMap<String, DependencyRelayBatch>,
|
||
) {
|
||
let announcement =
|
||
match crate::nostr::events::RepositoryAnnouncement::from_event(event.clone()) {
|
||
Ok(announcement) => announcement,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
event_id = %event.id,
|
||
error = %error,
|
||
"Failed to parse purgatory announcement for dependency retry"
|
||
);
|
||
return;
|
||
}
|
||
};
|
||
|
||
let relay_url = format!("ws://{}", self.service_domain);
|
||
let mut dependency_refetch_ids = HashSet::new();
|
||
let mut hot_dependency_events = Vec::new();
|
||
let mut maintainer_queue: VecDeque<String> =
|
||
announcement.maintainers.iter().cloned().collect();
|
||
let mut visited_maintainers = HashSet::new();
|
||
let mut dependency_relay_urls: HashSet<String> =
|
||
announcement.relays.iter().cloned().collect();
|
||
|
||
while let Some(maintainer_hex) = maintainer_queue.pop_front() {
|
||
let maintainer_pubkey = match PublicKey::from_hex(&maintainer_hex) {
|
||
Ok(pubkey) => pubkey,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
maintainer_hex = %maintainer_hex,
|
||
error = %error,
|
||
"Invalid maintainer public key in purgatory announcement"
|
||
);
|
||
continue;
|
||
}
|
||
};
|
||
if !visited_maintainers.insert(maintainer_pubkey) {
|
||
continue;
|
||
}
|
||
|
||
for event_type in [
|
||
rejected_index::EventType::Announcement,
|
||
rejected_index::EventType::State,
|
||
] {
|
||
dependency_relay_urls.extend(self.rejected_events_index.dependency_relay_hints(
|
||
&maintainer_pubkey,
|
||
&announcement.identifier,
|
||
Some(event_type),
|
||
));
|
||
let (event_ids, hot_events) = self.rejected_events_index.dependency_candidates(
|
||
&maintainer_pubkey,
|
||
&announcement.identifier,
|
||
Some(event_type),
|
||
);
|
||
let hot_event_ids: HashSet<EventId> =
|
||
hot_events.iter().map(|candidate| candidate.id).collect();
|
||
|
||
dependency_refetch_ids.extend(
|
||
event_ids
|
||
.into_iter()
|
||
.filter(|event_id| !hot_event_ids.contains(event_id)),
|
||
);
|
||
hot_dependency_events.extend(
|
||
hot_events
|
||
.into_iter()
|
||
.map(|candidate| (relay_url.clone(), candidate)),
|
||
);
|
||
}
|
||
|
||
// Follow already accepted announcements so expired dependencies in
|
||
// a transitive maintainer chain are recovered on subsequent ticks.
|
||
match self
|
||
.database
|
||
.query(
|
||
Filter::new()
|
||
.kind(Kind::GitRepoAnnouncement)
|
||
.author(maintainer_pubkey)
|
||
.identifier(&announcement.identifier),
|
||
)
|
||
.await
|
||
{
|
||
Ok(events) => {
|
||
for accepted_event in events {
|
||
if let Ok(accepted_announcement) =
|
||
crate::nostr::events::RepositoryAnnouncement::from_event(accepted_event)
|
||
{
|
||
dependency_relay_urls.extend(accepted_announcement.relays);
|
||
maintainer_queue.extend(accepted_announcement.maintainers);
|
||
}
|
||
}
|
||
}
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
maintainer = %maintainer_hex,
|
||
identifier = %announcement.identifier,
|
||
error = %error,
|
||
"Failed to resolve accepted transitive maintainer announcements"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// The owner's state may also have arrived before their announcement.
|
||
dependency_relay_urls.extend(self.rejected_events_index.dependency_relay_hints(
|
||
&event.pubkey,
|
||
&announcement.identifier,
|
||
Some(rejected_index::EventType::State),
|
||
));
|
||
let (event_ids, hot_events) = self.rejected_events_index.dependency_candidates(
|
||
&event.pubkey,
|
||
&announcement.identifier,
|
||
Some(rejected_index::EventType::State),
|
||
);
|
||
let hot_event_ids: HashSet<EventId> =
|
||
hot_events.iter().map(|candidate| candidate.id).collect();
|
||
dependency_refetch_ids.extend(
|
||
event_ids
|
||
.into_iter()
|
||
.filter(|event_id| !hot_event_ids.contains(event_id)),
|
||
);
|
||
hot_dependency_events.extend(
|
||
hot_events
|
||
.into_iter()
|
||
.map(|candidate| (relay_url.clone(), candidate)),
|
||
);
|
||
|
||
let reserved_hot_ids = self.reserve_dependency_refetch_attempts(
|
||
hot_dependency_events
|
||
.iter()
|
||
.map(|(_, candidate)| candidate.id),
|
||
);
|
||
hot_dependency_events.retain(|(_, candidate)| reserved_hot_ids.contains(&candidate.id));
|
||
|
||
if !hot_dependency_events.is_empty() {
|
||
Self::process_purgatory_dependency_events(
|
||
hot_dependency_events,
|
||
&self.database,
|
||
&self.write_policy,
|
||
&self.local_relay,
|
||
&self.rejected_events_index,
|
||
&self.dependency_refetch_attempts,
|
||
)
|
||
.await;
|
||
}
|
||
|
||
if !dependency_refetch_ids.is_empty() {
|
||
let mut canonical_dependency_relays = Vec::new();
|
||
for relay_url in dependency_relay_urls {
|
||
let relay_url = match canonical_relay_key(&relay_url) {
|
||
Ok(relay_url) => relay_url,
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
error = %error,
|
||
"Ignoring invalid rejected dependency source relay"
|
||
);
|
||
continue;
|
||
}
|
||
};
|
||
self.dependency_relay_deadlines.insert(
|
||
relay_url.clone(),
|
||
Instant::now() + dependency_relay_retention(),
|
||
);
|
||
if !self.connections.contains_key(&relay_url) {
|
||
if !self.register_relay(relay_url.clone(), false, false).await {
|
||
continue;
|
||
}
|
||
self.schedule_connect_relay(&relay_url).await;
|
||
}
|
||
canonical_dependency_relays.push(relay_url);
|
||
}
|
||
|
||
let connected_dependency_relays: Vec<String> = {
|
||
let relay_index = self.relay_sync_index.read().await;
|
||
canonical_dependency_relays
|
||
.into_iter()
|
||
.filter(|relay_url| {
|
||
relay_index
|
||
.get(relay_url)
|
||
.is_some_and(|state| state.connection_status.is_live_sync_active())
|
||
})
|
||
.collect()
|
||
};
|
||
for relay_url in connected_dependency_relays {
|
||
let batch = dependency_refetch_batches.entry(relay_url).or_default();
|
||
batch
|
||
.event_ids
|
||
.extend(dependency_refetch_ids.iter().copied());
|
||
batch.identifiers.insert(announcement.identifier.clone());
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Fetch cold-cache dependency IDs without blocking the sync manager.
|
||
///
|
||
/// Candidate IDs stay in the rejected index until policy processing
|
||
/// succeeds. All due repositories are combined by relay before requests
|
||
/// are spawned, so the maintenance cadence consumes one query per bounded
|
||
/// ID chunk rather than one query per repository and relay.
|
||
fn spawn_batched_purgatory_dependency_refetch(
|
||
&self,
|
||
mut batches: HashMap<String, DependencyRelayBatch>,
|
||
) {
|
||
if batches.is_empty() {
|
||
return;
|
||
}
|
||
|
||
let due_event_ids = self.reserve_dependency_refetch_attempts(
|
||
batches
|
||
.values()
|
||
.flat_map(|batch| batch.event_ids.iter().copied()),
|
||
);
|
||
if due_event_ids.is_empty() {
|
||
return;
|
||
}
|
||
for batch in batches.values_mut() {
|
||
batch
|
||
.event_ids
|
||
.retain(|event_id| due_event_ids.contains(event_id));
|
||
}
|
||
batches.retain(|relay_url, batch| {
|
||
!batch.event_ids.is_empty() && self.connections.contains_key(relay_url)
|
||
});
|
||
|
||
let connections = self.connections.clone();
|
||
let database = self.database.clone();
|
||
let write_policy = self.write_policy.clone();
|
||
let local_relay = self.local_relay.clone();
|
||
let rejected_events_index = self.rejected_events_index.clone();
|
||
let dependency_refetch_attempts = self.dependency_refetch_attempts.clone();
|
||
|
||
tokio::spawn(async move {
|
||
let mut fetches = Vec::new();
|
||
for (relay_url, batch) in batches {
|
||
let Some(connection) = connections.get(&relay_url).cloned() else {
|
||
continue;
|
||
};
|
||
let event_ids: Vec<EventId> = batch.event_ids.into_iter().collect();
|
||
let repository_count = batch.identifiers.len();
|
||
for chunk in event_ids.chunks(MAX_PURGATORY_DEPENDENCY_IDS_PER_QUERY) {
|
||
let chunk = chunk.to_vec();
|
||
let connection = connection.clone();
|
||
let relay_url = relay_url.clone();
|
||
fetches.push(async move {
|
||
let result = connection
|
||
.fetch_events(
|
||
Filter::new().ids(chunk.iter().copied()).limit(chunk.len()),
|
||
Duration::from_secs(5),
|
||
)
|
||
.await;
|
||
(relay_url, repository_count, chunk.len(), result)
|
||
});
|
||
}
|
||
}
|
||
|
||
let mut fetched_events = Vec::new();
|
||
for (relay_url, repository_count, requested_count, result) in join_all(fetches).await {
|
||
match result {
|
||
Ok(events) => {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
repository_count,
|
||
requested_count,
|
||
fetched_count = events.len(),
|
||
"Fetched purgatory dependencies by exact event ID"
|
||
);
|
||
fetched_events.extend(
|
||
events
|
||
.into_iter()
|
||
.map(|candidate| (relay_url.clone(), candidate)),
|
||
);
|
||
}
|
||
Err(error) => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
repository_count,
|
||
event_count = requested_count,
|
||
error = %error,
|
||
"Failed to refetch batched purgatory dependencies"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
Self::process_purgatory_dependency_events(
|
||
fetched_events,
|
||
&database,
|
||
&write_policy,
|
||
&local_relay,
|
||
&rejected_events_index,
|
||
&dependency_refetch_attempts,
|
||
)
|
||
.await;
|
||
});
|
||
}
|
||
|
||
/// Reserve due dependency IDs so timer ticks cannot create retry storms.
|
||
fn reserve_dependency_refetch_attempts(
|
||
&self,
|
||
event_ids: impl IntoIterator<Item = EventId>,
|
||
) -> HashSet<EventId> {
|
||
let retry_after = if std::env::var("NGIT_TEST").as_deref() == Ok("1") {
|
||
Duration::from_secs(6)
|
||
} else {
|
||
Duration::from_secs(30)
|
||
};
|
||
let now = Instant::now();
|
||
let mut attempts = self.dependency_refetch_attempts.lock().unwrap();
|
||
attempts.retain(|_, attempted_at| {
|
||
now.duration_since(*attempted_at) < Duration::from_secs(3600)
|
||
});
|
||
|
||
event_ids
|
||
.into_iter()
|
||
.filter(|event_id| {
|
||
let eligible = attempts
|
||
.get(event_id)
|
||
.is_none_or(|attempted_at| now.duration_since(*attempted_at) >= retry_after);
|
||
if eligible {
|
||
attempts.insert(*event_id, now);
|
||
}
|
||
eligible
|
||
})
|
||
.collect()
|
||
}
|
||
|
||
/// Apply dependency events in authorization order and clear only successes.
|
||
#[allow(clippy::too_many_arguments)]
|
||
async fn process_purgatory_dependency_events(
|
||
mut events: Vec<(String, Event)>,
|
||
database: &SharedDatabase,
|
||
write_policy: &Nip34WritePolicy,
|
||
local_relay: &LocalRelay,
|
||
rejected_events_index: &Arc<RejectedEventsIndex>,
|
||
dependency_refetch_attempts: &Arc<std::sync::Mutex<HashMap<EventId, Instant>>>,
|
||
) {
|
||
// Authorization for a state can depend on the reciprocal announcement
|
||
// returned by the same request. Relays may return matches in any order.
|
||
events.sort_by_key(|(_, candidate)| candidate.kind != Kind::GitRepoAnnouncement);
|
||
|
||
let mut processed_event_ids = HashSet::new();
|
||
for (relay_url, event) in events {
|
||
if !processed_event_ids.insert(event.id) {
|
||
continue;
|
||
}
|
||
|
||
let result = Self::process_event_static(
|
||
&event,
|
||
&relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
|
||
if result.is_terminally_accounted() {
|
||
rejected_events_index.remove(&event.id);
|
||
dependency_refetch_attempts
|
||
.lock()
|
||
.unwrap()
|
||
.remove(&event.id);
|
||
}
|
||
|
||
if result == ProcessResult::Purgatory && event.kind == Kind::RepoState {
|
||
if let Some(identifier) = event
|
||
.tags
|
||
.iter()
|
||
.find(|tag| tag.kind() == "d")
|
||
.and_then(|tag| tag.content())
|
||
{
|
||
write_policy.purgatory().enqueue_sync_immediate(identifier);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Handle a relay disconnection
|
||
///
|
||
/// This method is called when the event loop terminates and sends a disconnect notification.
|
||
/// It handles two cases:
|
||
/// - Unexpected disconnects: Updates state to Disconnected, keeps RelayConnection for reconnect
|
||
/// - Intentional disconnects: Completes cleanup of Disconnecting relays (removes from indices)
|
||
async fn handle_disconnect(&mut self, relay_url: &str) {
|
||
// Learned page sizes and NIP-11 hints belong to the ended WebSocket session.
|
||
self.pagination_sessions.remove(relay_url);
|
||
// A connection can end after a descendant REQ was sent but before its
|
||
// EOSE. Restart the bounded full rotation so that page is not treated
|
||
// as complete on the next session.
|
||
self.descendant_sync_rotations.remove(relay_url);
|
||
self.descendant_live_coverage.remove(relay_url);
|
||
self.auth_required_attempts
|
||
.retain(|(attempt_relay, _)| attempt_relay != relay_url);
|
||
|
||
// Check if this was an intentional disconnect (Disconnecting status)
|
||
let was_intentional = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.get(relay_url)
|
||
.map(|s| s.connection_status == ConnectionStatus::Disconnecting)
|
||
.unwrap_or(false)
|
||
};
|
||
self.cancel_deferred_consolidation(relay_url, "relay disconnect");
|
||
|
||
if was_intentional {
|
||
tracing::info!(relay = %relay_url, "Event loop terminated for intentional disconnect, completing cleanup");
|
||
self.complete_ended_session(relay_url, true).await;
|
||
} else {
|
||
// Ownership changes are passive: do not disturb a healthy session.
|
||
// Once the connection ends naturally, however, reconcile its
|
||
// confirmed state with the latest index before deciding whether it
|
||
// should ever reconnect.
|
||
let desired = self.derive_targets().await.remove(relay_url);
|
||
let Some(desired) = desired else {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"Naturally ended relay session is no longer owned; retiring without reconnect"
|
||
);
|
||
self.complete_ended_session(relay_url, true).await;
|
||
return;
|
||
};
|
||
|
||
// Unexpected disconnect - update state but keep for reconnection
|
||
tracing::warn!(relay = %relay_url, "Unexpected relay disconnect detected");
|
||
|
||
// Update RelayState in relay_sync_index
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
if let Some(state) = index.get_mut(relay_url) {
|
||
state.repos = desired.repos;
|
||
state.state_only_repos = desired.state_only_repos;
|
||
state.root_events = desired.root_events;
|
||
state.connection_status = ConnectionStatus::Disconnected;
|
||
state.disconnected_at = Some(Timestamp::now());
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
full_repos_tracked = state.repos.len(),
|
||
state_only_repos_tracked = state.state_only_repos.len(),
|
||
"Relay state updated to disconnected"
|
||
);
|
||
} else {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"No RelayState found for disconnected relay"
|
||
);
|
||
return;
|
||
}
|
||
}
|
||
|
||
// Clear pending sync batches for this relay
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if pending.remove(relay_url).is_some() {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Cleared pending sync batches for disconnected relay"
|
||
);
|
||
}
|
||
}
|
||
|
||
// Keep RelayConnection in HashMap for reuse on reconnect
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Keeping RelayConnection in HashMap for reconnection"
|
||
);
|
||
|
||
// Record failure in health tracker
|
||
self.health_tracker.record_failure(relay_url);
|
||
|
||
// Update metrics
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_status(relay_url, ConnectionStatus::Disconnected);
|
||
metrics.dec_connected_count();
|
||
metrics.record_health_state(relay_url, self.health_tracker.get_state(relay_url));
|
||
}
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
health_state = %self.health_tracker.get_state(relay_url),
|
||
"Unexpected disconnect handling complete"
|
||
);
|
||
}
|
||
}
|
||
|
||
/// Remove all state owned by an ended session that must not reconnect.
|
||
///
|
||
/// Connected sessions reach this after their event loop terminates. An
|
||
/// already-disconnected session has no event loop left to notify us, so
|
||
/// `disconnect_relay` calls it directly without decrementing the connected
|
||
/// gauge a second time.
|
||
async fn complete_ended_session(&mut self, relay_url: &str, was_connected: bool) {
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_status(relay_url, ConnectionStatus::Disconnected);
|
||
}
|
||
|
||
self.relay_sync_index.write().await.remove(relay_url);
|
||
self.pending_sync_index.write().await.remove(relay_url);
|
||
self.connections.remove(relay_url);
|
||
self.nip65_discovery_only_relays.remove(relay_url);
|
||
self.missing_event_recovery
|
||
.lock()
|
||
.unwrap()
|
||
.clear_relay(relay_url);
|
||
self.health_tracker.forget_relay(relay_url);
|
||
|
||
if let Some(ref metrics) = self.metrics {
|
||
if was_connected {
|
||
metrics.dec_connected_count();
|
||
}
|
||
metrics.forget_relay(relay_url);
|
||
}
|
||
|
||
tracing::info!(relay = %relay_url, "Ended relay session cleanup complete");
|
||
|
||
let still_desired = self.derive_targets().await.contains_key(relay_url);
|
||
if still_desired
|
||
&& self
|
||
.register_relay(relay_url.to_string(), false, false)
|
||
.await
|
||
{
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"Relay remains shared by current repository announcements; starting a clean session"
|
||
);
|
||
self.schedule_connect_relay(relay_url).await;
|
||
}
|
||
}
|
||
|
||
/// Re-process events from hot cache after their dependencies become available
|
||
///
|
||
/// This helper consolidates the common pattern of re-processing rejected events
|
||
/// when their missing dependencies (owner announcements, git data, etc.) arrive.
|
||
///
|
||
/// # Arguments
|
||
/// * `events` - Events to re-process from hot cache
|
||
/// * `context` - Description for logging (e.g., "maintainer announcement", "state event")
|
||
/// * `pubkey` - Public key for logging context
|
||
/// * `identifier` - Repository identifier for logging context
|
||
/// * `relay_url` - Relay URL for process_event_static
|
||
/// * `database` - Shared database for event storage
|
||
/// * `write_policy` - Policy for validating events
|
||
/// * `local_relay` - Local relay for broadcasting events
|
||
/// * `rejected_events_index` - Index for tracking rejected events
|
||
///
|
||
/// # Returns
|
||
/// Statistics about re-processing outcomes
|
||
#[allow(clippy::too_many_arguments)]
|
||
async fn reprocess_events_from_hot_cache(
|
||
events: Vec<Event>,
|
||
context: &str,
|
||
pubkey: &PublicKey,
|
||
identifier: &str,
|
||
relay_url: &str,
|
||
database: &SharedDatabase,
|
||
write_policy: &Nip34WritePolicy,
|
||
local_relay: &LocalRelay,
|
||
rejected_events_index: &Arc<RejectedEventsIndex>,
|
||
) -> ReprocessingStats {
|
||
let mut stats = ReprocessingStats::default();
|
||
|
||
for event in events {
|
||
tracing::info!(
|
||
event_id = %event.id,
|
||
pubkey = %pubkey,
|
||
identifier = %identifier,
|
||
context = %context,
|
||
"Re-processing {} from hot cache",
|
||
context
|
||
);
|
||
|
||
// Recursive call to process_event_static
|
||
// This is safe because:
|
||
// 1. The caller has observed a newly available dependency
|
||
// 2. Second attempt uses new context (different code path)
|
||
// 3. A failed retry remains indexed for a later bounded attempt
|
||
// Use Box::pin to avoid infinitely sized future
|
||
let reprocess_result = Box::pin(Self::process_event_static(
|
||
&event,
|
||
relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
crate::nostr::persistence::SaveContext::HotCacheReprocess,
|
||
))
|
||
.await;
|
||
|
||
match reprocess_result {
|
||
ProcessResult::Saved => {
|
||
stats.saved += 1;
|
||
tracing::info!(
|
||
event_id = %event.id,
|
||
pubkey = %pubkey,
|
||
identifier = %identifier,
|
||
"{} accepted on re-processing",
|
||
context
|
||
);
|
||
}
|
||
ProcessResult::Duplicate => {
|
||
stats.duplicate += 1;
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
"{} already exists (duplicate)",
|
||
context
|
||
);
|
||
}
|
||
ProcessResult::Purgatory => {
|
||
stats.purgatory += 1;
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
"{} added to purgatory (waiting for git data)",
|
||
context
|
||
);
|
||
|
||
// Hot-cache retries are sync-originated events. Reset any
|
||
// existing backoff so newly available announcement/clone
|
||
// metadata is acted on immediately rather than inheriting
|
||
// the direct-submission three-minute delay.
|
||
let identifier = if event.kind == Kind::RepoState {
|
||
event
|
||
.tags
|
||
.iter()
|
||
.find(|tag| tag.kind() == "d")
|
||
.and_then(|tag| tag.content())
|
||
.map(str::to_owned)
|
||
} else if event.kind == Kind::GitPullRequest
|
||
|| event.kind == Kind::GitPullRequestUpdate
|
||
{
|
||
crate::git::sync::extract_identifier_from_pr_event(&event)
|
||
} else {
|
||
None
|
||
};
|
||
|
||
if let Some(identifier) = identifier {
|
||
write_policy.purgatory().enqueue_sync_immediate(&identifier);
|
||
}
|
||
}
|
||
ProcessResult::Tombstoned => {
|
||
stats.duplicate += 1;
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
"{} remains absent under deletion semantics",
|
||
context
|
||
);
|
||
}
|
||
ProcessResult::Rejected(_) | ProcessResult::PersistenceError => {
|
||
stats.rejected += 1;
|
||
tracing::warn!(
|
||
event_id = %event.id,
|
||
pubkey = %pubkey,
|
||
identifier = %identifier,
|
||
"{} still rejected on re-processing",
|
||
context
|
||
);
|
||
}
|
||
}
|
||
|
||
if reprocess_result.is_terminally_accounted() {
|
||
rejected_events_index.remove(&event.id);
|
||
}
|
||
}
|
||
|
||
stats
|
||
}
|
||
|
||
/// Process a single event from a relay (static version for spawned tasks)
|
||
///
|
||
/// Processes events with dedup, policy check, database save, and broadcast:
|
||
/// - Deduplication (skips if event already exists)
|
||
/// - Write policy validation
|
||
/// - Database save
|
||
/// - Broadcast to WebSocket subscribers via notify_event (enables recursive relay discovery)
|
||
///
|
||
/// Returns `ProcessResult` to indicate whether the event was saved, duplicate, or rejected.
|
||
fn classify_policy_result(
|
||
prefix: &nostr_sdk::prelude::MachineReadablePrefix,
|
||
message: &str,
|
||
status: bool,
|
||
) -> ProcessResult {
|
||
use nostr_sdk::prelude::MachineReadablePrefix;
|
||
|
||
if status {
|
||
return if prefix.as_str() == "purgatory" || message.contains("purgatory") {
|
||
ProcessResult::Purgatory
|
||
} else {
|
||
ProcessResult::Duplicate
|
||
};
|
||
}
|
||
|
||
if message == "this event is deleted" || message == "this pubkey has requested to vanish" {
|
||
return ProcessResult::Tombstoned;
|
||
}
|
||
|
||
let rejection = match prefix {
|
||
MachineReadablePrefix::Blocked => PolicyRejection::Blocked,
|
||
MachineReadablePrefix::Invalid => PolicyRejection::Invalid,
|
||
MachineReadablePrefix::Restricted => PolicyRejection::Restricted,
|
||
MachineReadablePrefix::Error => PolicyRejection::Error,
|
||
_ => PolicyRejection::Other,
|
||
};
|
||
ProcessResult::Rejected(rejection)
|
||
}
|
||
|
||
async fn process_event_static(
|
||
event: &Event,
|
||
relay_url: &str,
|
||
database: &SharedDatabase,
|
||
write_policy: &Nip34WritePolicy,
|
||
local_relay: &LocalRelay,
|
||
rejected_events_index: &Arc<RejectedEventsIndex>,
|
||
save_context: crate::nostr::persistence::SaveContext,
|
||
) -> ProcessResult {
|
||
Self::process_event_static_inner(
|
||
event,
|
||
relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
save_context,
|
||
true,
|
||
)
|
||
.await
|
||
}
|
||
|
||
#[allow(clippy::too_many_arguments)]
|
||
async fn process_event_static_inner(
|
||
event: &Event,
|
||
relay_url: &str,
|
||
database: &SharedDatabase,
|
||
write_policy: &Nip34WritePolicy,
|
||
local_relay: &LocalRelay,
|
||
rejected_events_index: &Arc<RejectedEventsIndex>,
|
||
save_context: crate::nostr::persistence::SaveContext,
|
||
resolve_related_dependencies: bool,
|
||
) -> ProcessResult {
|
||
use nostr_sdk::prelude::{WritePolicy, WritePolicyResult};
|
||
use std::net::{IpAddr, Ipv4Addr, SocketAddr};
|
||
// Check if event already exists
|
||
match database.event_by_id(&event.id).await {
|
||
Ok(Some(_)) => {
|
||
tracing::trace!(event_id = %event.id, "Event already exists, skipping");
|
||
return ProcessResult::Duplicate;
|
||
}
|
||
Err(e) => {
|
||
tracing::warn!(event_id = %event.id, error = %e, "Database error checking event");
|
||
return ProcessResult::PersistenceError;
|
||
}
|
||
Ok(None) => {} // Continue processing
|
||
}
|
||
|
||
// Apply write policy using a dummy address (sync events aren't from network clients)
|
||
let dummy_addr = SocketAddr::new(IpAddr::V4(Ipv4Addr::LOCALHOST), 0);
|
||
let result = write_policy.admit_event(event, &dummy_addr).await;
|
||
|
||
match result {
|
||
WritePolicyResult::Accept => {
|
||
// Save event to database
|
||
|
||
if let Err(e) = write_policy.save_accepted_event(event, save_context).await {
|
||
tracing::error!(
|
||
event_id = %event.id,
|
||
relay = %relay_url,
|
||
error = %e,
|
||
"Failed to save synced event"
|
||
);
|
||
return ProcessResult::PersistenceError;
|
||
}
|
||
|
||
// Broadcast to WebSocket subscribers (enables recursive relay discovery)
|
||
// This allows SelfSubscriber to receive synced 30617 announcements
|
||
let broadcast_success = local_relay.notify_event(event.clone());
|
||
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
relay = %relay_url,
|
||
kind = %event.kind.as_u16(),
|
||
broadcast = broadcast_success,
|
||
"Synced event saved and broadcast"
|
||
);
|
||
|
||
// When a repository announcement is accepted, re-process any rejected events:
|
||
// 1. Maintainer announcements that were rejected because the owner announcement didn't exist yet
|
||
// 2. State events that were rejected because no announcement existed
|
||
// This handles race conditions where events arrive before their dependencies during relay sync.
|
||
if event.kind == Kind::GitRepoAnnouncement {
|
||
use crate::nostr::events::RepositoryAnnouncement;
|
||
|
||
match RepositoryAnnouncement::from_event(event.clone()) {
|
||
Ok(announcement) => {
|
||
// Re-process rejected maintainer announcements
|
||
if !announcement.maintainers.is_empty() {
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
identifier = %announcement.identifier,
|
||
maintainer_count = announcement.maintainers.len(),
|
||
"Owner announcement accepted, checking for rejected maintainer announcements"
|
||
);
|
||
|
||
// For each maintainer, invalidate and get their events
|
||
for maintainer_hex in &announcement.maintainers {
|
||
// Parse maintainer public key
|
||
match PublicKey::from_hex(maintainer_hex) {
|
||
Ok(maintainer_pubkey) => {
|
||
let (event_ids, hot_events) = rejected_events_index
|
||
.dependency_candidates(
|
||
&maintainer_pubkey,
|
||
&announcement.identifier,
|
||
Some(rejected_index::EventType::Announcement),
|
||
);
|
||
|
||
if !event_ids.is_empty() {
|
||
tracing::info!(
|
||
maintainer = %maintainer_hex,
|
||
identifier = %announcement.identifier,
|
||
dependency_event_count = event_ids.len(),
|
||
hot_cache_events = hot_events.len(),
|
||
"Found rejected maintainer announcement dependencies"
|
||
);
|
||
}
|
||
|
||
// Re-process events from hot cache immediately
|
||
if !hot_events.is_empty() {
|
||
let _stats = Self::reprocess_events_from_hot_cache(
|
||
hot_events,
|
||
"maintainer announcement",
|
||
&maintainer_pubkey,
|
||
&announcement.identifier,
|
||
relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
)
|
||
.await;
|
||
}
|
||
}
|
||
Err(e) => {
|
||
tracing::warn!(
|
||
maintainer_hex = %maintainer_hex,
|
||
error = %e,
|
||
"Invalid maintainer public key in announcement"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// Re-process rejected state events for this announcement
|
||
let (event_ids, hot_events) = rejected_events_index
|
||
.dependency_candidates(
|
||
&event.pubkey,
|
||
&announcement.identifier,
|
||
Some(rejected_index::EventType::State),
|
||
);
|
||
|
||
if !event_ids.is_empty() {
|
||
tracing::info!(
|
||
pubkey = %event.pubkey,
|
||
identifier = %announcement.identifier,
|
||
dependency_event_count = event_ids.len(),
|
||
hot_cache_events = hot_events.len(),
|
||
"Found rejected state dependencies after announcement acceptance"
|
||
);
|
||
}
|
||
|
||
// Re-process state events from hot cache immediately
|
||
if !hot_events.is_empty() {
|
||
let _stats = Self::reprocess_events_from_hot_cache(
|
||
hot_events,
|
||
"state event",
|
||
&event.pubkey,
|
||
&announcement.identifier,
|
||
relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
)
|
||
.await;
|
||
}
|
||
}
|
||
Err(e) => {
|
||
tracing::warn!(
|
||
event_id = %event.id,
|
||
error = %e,
|
||
"Failed to parse repository announcement for rejected event invalidation"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// When a state event is accepted (git data arrived), re-process any other
|
||
// rejected state events for the same repository. This handles the case where
|
||
// multiple state events arrive but only one has git data initially.
|
||
// Events in the hot cache are re-processed immediately now that git data is available.
|
||
if event.kind == Kind::RepoState {
|
||
// Extract identifier from 'd' tag
|
||
if let Some(identifier) = event
|
||
.tags
|
||
.iter()
|
||
.find(|t| t.kind() == "d")
|
||
.and_then(|t| t.content())
|
||
{
|
||
// Get rejected state events for this pubkey + identifier
|
||
let (event_ids, hot_events) = rejected_events_index.dependency_candidates(
|
||
&event.pubkey,
|
||
identifier,
|
||
Some(rejected_index::EventType::State),
|
||
);
|
||
|
||
if !event_ids.is_empty() {
|
||
tracing::info!(
|
||
pubkey = %event.pubkey,
|
||
identifier = %identifier,
|
||
dependency_event_count = event_ids.len(),
|
||
hot_cache_events = hot_events.len(),
|
||
"Found rejected state dependencies after git data became available"
|
||
);
|
||
}
|
||
|
||
// Re-process events from hot cache immediately
|
||
if !hot_events.is_empty() {
|
||
let _stats = Self::reprocess_events_from_hot_cache(
|
||
hot_events,
|
||
"state event",
|
||
&event.pubkey,
|
||
identifier,
|
||
relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
)
|
||
.await;
|
||
}
|
||
}
|
||
}
|
||
|
||
if resolve_related_dependencies {
|
||
let mut queue: VecDeque<Event> = rejected_events_index
|
||
.related_candidates_resolved_by(event)
|
||
.into_iter()
|
||
.collect();
|
||
let mut attempted = HashSet::new();
|
||
let mut saved = 0usize;
|
||
while let Some(candidate) = queue.pop_front() {
|
||
if attempted.len() >= rejected_index::RELATED_RETRY_CLOSURE_LIMIT
|
||
|| !attempted.insert(candidate.id)
|
||
{
|
||
continue;
|
||
}
|
||
let outcome = Box::pin(Self::process_event_static_inner(
|
||
&candidate,
|
||
relay_url,
|
||
database,
|
||
write_policy,
|
||
local_relay,
|
||
rejected_events_index,
|
||
save_context,
|
||
false,
|
||
))
|
||
.await;
|
||
if outcome.is_terminally_accounted() {
|
||
rejected_events_index.remove(&candidate.id);
|
||
}
|
||
if outcome == ProcessResult::Saved {
|
||
saved += 1;
|
||
queue.extend(
|
||
rejected_events_index.related_candidates_resolved_by(&candidate),
|
||
);
|
||
}
|
||
}
|
||
if !attempted.is_empty() {
|
||
tracing::info!(
|
||
trigger_event = %event.id,
|
||
attempted = attempted.len(),
|
||
saved,
|
||
remaining = rejected_events_index.related_len(),
|
||
"Reprocessed dependency-sensitive related-event closure"
|
||
);
|
||
}
|
||
}
|
||
|
||
ProcessResult::Saved
|
||
}
|
||
WritePolicyResult::Reject {
|
||
prefix,
|
||
message,
|
||
status,
|
||
} => {
|
||
let process_result = Self::classify_policy_result(&prefix, &message, status);
|
||
if matches!(
|
||
process_result,
|
||
ProcessResult::Purgatory | ProcessResult::Duplicate
|
||
) {
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
kind = %event.kind.as_u16(),
|
||
outcome = process_result.hydration_outcome(),
|
||
reason = %message,
|
||
"Event accepted without main-database persistence"
|
||
);
|
||
// Note: git data sync for state events is triggered by the policy
|
||
// layer when adding to purgatory (via start_state_sync)
|
||
process_result
|
||
} else {
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
relay = %relay_url,
|
||
kind = %event.kind.as_u16(),
|
||
outcome = process_result.hydration_outcome(),
|
||
reason = %message,
|
||
"Event rejected by write policy"
|
||
);
|
||
|
||
// Track rejected announcement and state events to avoid re-fetching them
|
||
if event.kind == Kind::GitRepoAnnouncement || event.kind == Kind::RepoState {
|
||
// Extract identifier from 'd' tag
|
||
if let Some(identifier) = event
|
||
.tags
|
||
.iter()
|
||
.find(|t| t.kind() == "d")
|
||
.and_then(|t| t.content())
|
||
{
|
||
// Determine rejection reason based on message
|
||
let reason = if message.contains("doesn't list this service")
|
||
|| message.contains("Announcement must list service")
|
||
{
|
||
rejected_index::RejectionReason::DoesNotListService
|
||
} else if message.contains("maintainer")
|
||
|| message.contains("no announcement exists")
|
||
|| message.contains("not authorized")
|
||
{
|
||
rejected_index::RejectionReason::MaintainerNotYetValid
|
||
} else {
|
||
rejected_index::RejectionReason::Other
|
||
};
|
||
|
||
// Use appropriate method based on event kind
|
||
if event.kind == Kind::RepoState {
|
||
rejected_events_index.add_state_from_relay(
|
||
event.clone(),
|
||
event.pubkey,
|
||
identifier.to_string(),
|
||
reason,
|
||
Some(relay_url.to_string()),
|
||
);
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
kind = %event.kind.as_u16(),
|
||
identifier = %identifier,
|
||
"Added rejected state event to two-tier index"
|
||
);
|
||
} else {
|
||
rejected_events_index.add_announcement_from_relay(
|
||
event.clone(),
|
||
event.pubkey,
|
||
identifier.to_string(),
|
||
reason,
|
||
Some(relay_url.to_string()),
|
||
);
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
kind = %event.kind.as_u16(),
|
||
identifier = %identifier,
|
||
pubkey = %event.pubkey,
|
||
"Added rejected announcement to two-tier index"
|
||
);
|
||
}
|
||
} else {
|
||
// No 'd' tag: structurally malformed, permanently
|
||
// invalid, and without a pubkey+identifier key for
|
||
// the two-tier index. Remember the exact ID so
|
||
// historic sync stops re-downloading and
|
||
// revalidating it. Direct live submissions never
|
||
// reach this sync-only path; their rejections are
|
||
// logged by the write policy.
|
||
rejected_events_index.add_unrecoverable(event.id, event.kind.as_u16());
|
||
tracing::warn!(
|
||
event_id = %event.id,
|
||
kind = %event.kind.as_u16(),
|
||
relay = %relay_url,
|
||
"Synced event missing 'd' tag, tracked as unrecoverable by ID"
|
||
);
|
||
}
|
||
} else if process_result == ProcessResult::Rejected(PolicyRejection::Restricted)
|
||
&& message.contains(
|
||
"event must reference an accepted repository or accepted event",
|
||
)
|
||
{
|
||
let (addressable_refs, event_refs) =
|
||
crate::nostr::policy::RelatedEventPolicy::extract_reference_tags(event);
|
||
let retained = rejected_events_index.add_related_from_relay(
|
||
event.clone(),
|
||
event_refs.into_iter().collect(),
|
||
addressable_refs.into_iter().collect(),
|
||
Some(relay_url.to_string()),
|
||
);
|
||
tracing::debug!(
|
||
event_id = %event.id,
|
||
kind = %event.kind.as_u16(),
|
||
retained,
|
||
"Indexed dependency-sensitive related event for durable retry"
|
||
);
|
||
}
|
||
|
||
process_result
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
// =========================================================================
|
||
// Consolidation System
|
||
// =========================================================================
|
||
|
||
async fn has_pending_batches(&self, relay_url: &str) -> bool {
|
||
self.pending_sync_index
|
||
.read()
|
||
.await
|
||
.get(relay_url)
|
||
.is_some_and(|batches| !batches.is_empty())
|
||
}
|
||
|
||
async fn process_deferred_consolidation(&mut self, relay_url: &str) {
|
||
let has_pending_batches = self.has_pending_batches(relay_url).await;
|
||
if !self
|
||
.deferred_consolidations
|
||
.take_ready(relay_url, has_pending_batches)
|
||
{
|
||
return;
|
||
}
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"Pending batches drained; running deferred consolidation"
|
||
);
|
||
let _ = self.consolidate(relay_url).await;
|
||
// The AddFilters action that requested this consolidation was
|
||
// intentionally not admitted. Re-derive it now that the deferral has
|
||
// cleared so a quiet relay does not wait for an unrelated dirty signal.
|
||
self.recompute_new_sync_filters_for_relay(relay_url).await;
|
||
}
|
||
|
||
fn cancel_deferred_consolidation(&mut self, relay_url: &str, reason: &'static str) {
|
||
if self.deferred_consolidations.cancel(relay_url) {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
reason,
|
||
"Cancelled deferred consolidation"
|
||
);
|
||
}
|
||
}
|
||
|
||
/// Consolidate all subscriptions for a relay
|
||
///
|
||
/// The caller must ensure pending batches have drained so EOSE processing
|
||
/// never waits behind the sync actor lock.
|
||
///
|
||
async fn consolidate(&mut self, relay_url: &str) -> bool {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
"Starting consolidation"
|
||
);
|
||
|
||
// Descendant coverage is auxiliary and independently reconstructible.
|
||
// Retire it before capturing the core rollback set so it can neither
|
||
// displace newly required core coverage nor become part of that set.
|
||
if !self
|
||
.close_descendant_live_coverage(relay_url, "core consolidation")
|
||
.await
|
||
{
|
||
return false;
|
||
}
|
||
|
||
let now = Timestamp::now();
|
||
let since = Timestamp::from(now.as_secs().saturating_sub(QUICK_RECONNECT_WINDOW_SECS));
|
||
let complete_live = self.complete_live_filters(relay_url, Some(since)).await;
|
||
|
||
// Replace L1+L2+L3 as one reserved transaction. The connection-level
|
||
// opener rolls every successful group back if a later group fails.
|
||
let connection = match self.connections.get(relay_url) {
|
||
Some(conn) => conn.clone(),
|
||
None => {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"No connection found, skipping consolidation"
|
||
);
|
||
return false;
|
||
}
|
||
};
|
||
|
||
let complete_group_count = grouped_subscription_count(&complete_live);
|
||
if !connection.complete_live_set_fits(complete_group_count) {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
complete_group_count,
|
||
"Complete live coverage exceeds the connection budget; keeping existing coverage and deferring replacement"
|
||
);
|
||
return false;
|
||
}
|
||
|
||
let unbounded_groups = live_filter_groups(&complete_live, connection.max_filters_per_req());
|
||
let remote_limit = connection.remote_subscription_byte_limit();
|
||
let (complete_groups, overflow_groups) =
|
||
groups_within_subscription_byte_limit(unbounded_groups, remote_limit, 0);
|
||
if connection
|
||
.replace_live_filter_groups(complete_groups)
|
||
.await
|
||
.is_err()
|
||
{
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
"Complete live coverage could not be replaced; historic work deferred"
|
||
);
|
||
return false;
|
||
}
|
||
if overflow_groups > 0 {
|
||
self.byte_limited_live_relays.insert(
|
||
relay_url.to_string(),
|
||
Instant::now() + byte_limited_catchup_interval(),
|
||
);
|
||
let items = self.desired_items_for_relay(relay_url).await;
|
||
self.historic_sync(relay_url, complete_live, items, Some(since))
|
||
.await;
|
||
} else {
|
||
self.byte_limited_live_relays.remove(relay_url);
|
||
self.sync_generic_history(relay_url, Some(since)).await;
|
||
}
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
since = %since,
|
||
overflow_groups,
|
||
"Consolidation complete - filter count reset"
|
||
);
|
||
true
|
||
}
|
||
|
||
fn release_removed_descendant_batch(&mut self, relay_url: &str, batch: &PendingBatch) {
|
||
if batch.purpose == PendingBatchPurpose::Descendants {
|
||
if let Some(rotation) = self.descendant_sync_rotations.get_mut(relay_url) {
|
||
rotation.mark_completed(batch.batch_id, false);
|
||
}
|
||
}
|
||
}
|
||
|
||
async fn handle_subscription_closed(
|
||
&mut self,
|
||
relay_url: &str,
|
||
subscription_id: SubscriptionId,
|
||
reason: &str,
|
||
live_generation: Option<u64>,
|
||
live_filter_count: Option<usize>,
|
||
) {
|
||
let is_descendant_live = self
|
||
.descendant_live_coverage
|
||
.get(relay_url)
|
||
.is_some_and(|coverage| coverage.subscription_ids.contains(&subscription_id));
|
||
let policy_category = if is_auth_required_message(reason) {
|
||
let Some(connection) = self.connections.get(relay_url) else {
|
||
return;
|
||
};
|
||
if !connection.holds_subscription_permit(&subscription_id) {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
"Ignoring auth-required CLOSED for an unowned subscription"
|
||
);
|
||
return;
|
||
}
|
||
// A retry is only worth reserving when the SDK will actually
|
||
// answer the challenge and resubscribe; without an authenticator
|
||
// the subscription is already gone and a reserved retry would
|
||
// dangle until disconnect cleanup.
|
||
if connection.answers_auth_challenges()
|
||
&& reserve_authentication_retry(
|
||
&mut self.auth_required_attempts,
|
||
relay_url,
|
||
&subscription_id,
|
||
)
|
||
{
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
"Relay requested authentication; retaining subscription for one NIP-42 retry"
|
||
);
|
||
return;
|
||
}
|
||
connection
|
||
.retire_auth_refused_subscription(&subscription_id)
|
||
.await;
|
||
Some(PolicyRefusal::AuthenticationRequired)
|
||
} else {
|
||
policy_refusal(reason)
|
||
};
|
||
if let Some(category) =
|
||
policy_category.filter(|category| *category != PolicyRefusal::FilterIncompatible)
|
||
{
|
||
let removed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
take_batch_containing_subscription(&mut pending, relay_url, &subscription_id)
|
||
};
|
||
if let Some(batch) = &removed_batch {
|
||
self.release_removed_descendant_batch(relay_url, batch);
|
||
}
|
||
if is_descendant_live {
|
||
if let Some(coverage) = self.descendant_live_coverage.remove(relay_url) {
|
||
if let Some(connection) = self.connections.get(relay_url) {
|
||
let _ = connection
|
||
.close_live_subscriptions(&coverage.subscription_ids)
|
||
.await;
|
||
}
|
||
}
|
||
}
|
||
self.health_tracker.record_policy_refusal(relay_url);
|
||
if let Some(metrics) = &self.metrics {
|
||
metrics.record_policy_refusal(relay_url, category.label());
|
||
metrics.record_health_state(relay_url, self.health_tracker.get_state(relay_url));
|
||
}
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
category = category.label(),
|
||
reason,
|
||
retry_hours = 24,
|
||
pending_batch_removed = removed_batch.is_some(),
|
||
descendant_live = is_descendant_live,
|
||
"Relay policy refused subscription; preserving connection and deferring coverage probe"
|
||
);
|
||
return;
|
||
}
|
||
if is_descendant_live {
|
||
let coverage = self
|
||
.descendant_live_coverage
|
||
.remove(relay_url)
|
||
.expect("descendant subscription belonged to tracked coverage");
|
||
if let Some(connection) = self.connections.get(relay_url) {
|
||
let _ = connection
|
||
.close_live_subscriptions(&coverage.subscription_ids)
|
||
.await;
|
||
}
|
||
let max_filters = self
|
||
.connections
|
||
.get(relay_url)
|
||
.map(|connection| connection.max_filters_per_req())
|
||
.unwrap_or(MAX_FILTERS_PER_REQ);
|
||
self.descendant_sync_rotations
|
||
.entry(relay_url.to_string())
|
||
.or_default()
|
||
.refresh(coverage.fallback_filters, max_filters);
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
reason,
|
||
"Descendant live subscription closed; using historic fallback"
|
||
);
|
||
return;
|
||
}
|
||
|
||
if let Some(limit) = subscription_state_byte_limit(reason) {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
limit,
|
||
"Remote retained-subscription capacity exhausted; rebuilding bounded live coverage"
|
||
);
|
||
self.byte_limited_live_relays
|
||
.insert(relay_url.to_string(), Instant::now());
|
||
let removed_batch = if live_generation.is_none() {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
take_batch_containing_subscription(&mut pending, relay_url, &subscription_id)
|
||
} else {
|
||
None
|
||
};
|
||
if let Some(batch) = &removed_batch {
|
||
self.release_removed_descendant_batch(relay_url, batch);
|
||
}
|
||
let has_pending = self.has_pending_batches(relay_url).await;
|
||
if self.deferred_consolidations.request(relay_url, has_pending) {
|
||
let _ = self.consolidate(relay_url).await;
|
||
}
|
||
return;
|
||
}
|
||
if is_rate_limit_message(reason) && !is_filter_count_refusal(reason) {
|
||
let removed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
take_batch_containing_subscription(&mut pending, relay_url, &subscription_id)
|
||
};
|
||
|
||
if let Some(batch) = removed_batch {
|
||
self.release_removed_descendant_batch(relay_url, &batch);
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
batch_id = batch.batch_id,
|
||
"Rate-limited subscription aborted its pending batch; retry deferred until cooldown"
|
||
);
|
||
} else if live_generation.is_some() {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
"Rate-limited live subscription restoration deferred until cooldown"
|
||
);
|
||
}
|
||
return;
|
||
}
|
||
|
||
if policy_refusal(reason) == Some(PolicyRefusal::FilterIncompatible) {
|
||
let pending_filter_count = {
|
||
let pending = self.pending_sync_index.read().await;
|
||
pending.get(relay_url).and_then(|batches| {
|
||
batches.iter().find_map(|batch| {
|
||
batch
|
||
.pagination_state
|
||
.get(&subscription_id)
|
||
.map(|state| state.filters.len())
|
||
})
|
||
})
|
||
};
|
||
if let (Some(rejected_count), Some(connection)) = (
|
||
live_filter_count.or(pending_filter_count),
|
||
self.connections.get(relay_url).cloned(),
|
||
) {
|
||
if rejected_count > 1 {
|
||
let learned_limit = connection
|
||
.reduce_max_filters_per_req(rejected_count)
|
||
.unwrap_or_else(|| connection.max_filters_per_req());
|
||
if let Some(metrics) = &self.metrics {
|
||
metrics.record_policy_refusal(relay_url, "filter_count");
|
||
}
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
rejected_filter_count = rejected_count,
|
||
learned_filter_count = learned_limit,
|
||
reason,
|
||
"Relay rejected filter count; regrouping complete live coverage"
|
||
);
|
||
let removed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
take_batch_containing_subscription(
|
||
&mut pending,
|
||
relay_url,
|
||
&subscription_id,
|
||
)
|
||
};
|
||
if let Some(batch) = &removed_batch {
|
||
self.release_removed_descendant_batch(relay_url, batch);
|
||
self.recompute_new_sync_filters_for_relay(relay_url).await;
|
||
} else if let Some(generation) = live_generation {
|
||
self.restore_live_coverage_after_closed(relay_url, generation)
|
||
.await;
|
||
}
|
||
return;
|
||
}
|
||
}
|
||
}
|
||
|
||
if let Some(category) = policy_refusal(reason) {
|
||
let removed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
take_batch_containing_subscription(&mut pending, relay_url, &subscription_id)
|
||
};
|
||
if let Some(batch) = &removed_batch {
|
||
self.release_removed_descendant_batch(relay_url, batch);
|
||
}
|
||
self.health_tracker.record_policy_refusal(relay_url);
|
||
if let Some(metrics) = &self.metrics {
|
||
metrics.record_policy_refusal(relay_url, category.label());
|
||
metrics.record_health_state(relay_url, self.health_tracker.get_state(relay_url));
|
||
}
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
category = category.label(),
|
||
reason,
|
||
retry_hours = 24,
|
||
pending_batch_removed = removed_batch.is_some(),
|
||
"Relay policy refused subscription; preserving connection and deferring coverage probe"
|
||
);
|
||
return;
|
||
}
|
||
|
||
if let Some(generation) = live_generation {
|
||
self.health_tracker.record_rate_limit(relay_url);
|
||
if let Some(metrics) = &self.metrics {
|
||
metrics.record_health_state(relay_url, self.health_tracker.get_state(relay_url));
|
||
}
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
sub_id = %subscription_id,
|
||
generation,
|
||
reason,
|
||
"Unexpected live subscription closure; deferring one repair attempt until cooldown"
|
||
);
|
||
}
|
||
}
|
||
|
||
async fn restore_live_coverage_after_closed(&mut self, relay_url: &str, generation: u64) {
|
||
let Some(current_generation) = self
|
||
.connections
|
||
.get(relay_url)
|
||
.map(RelayConnection::current_subscription_generation)
|
||
else {
|
||
return;
|
||
};
|
||
if current_generation != generation {
|
||
tracing::debug!(relay = %relay_url, generation, "Ignoring stale live CLOSED from retired session");
|
||
return;
|
||
}
|
||
if !self
|
||
.close_descendant_live_coverage(relay_url, "core live restoration")
|
||
.await
|
||
{
|
||
return;
|
||
}
|
||
let now = Timestamp::now();
|
||
let since = Timestamp::from(now.as_secs().saturating_sub(QUICK_RECONNECT_WINDOW_SECS));
|
||
let filters = self.complete_live_filters(relay_url, Some(since)).await;
|
||
let Some(connection) = self.connections.get(relay_url) else {
|
||
return;
|
||
};
|
||
if connection.current_subscription_generation() != generation {
|
||
return;
|
||
}
|
||
if let Err(error) = connection
|
||
.replace_live_filter_groups(live_filter_groups(
|
||
&filters,
|
||
connection.max_filters_per_req(),
|
||
))
|
||
.await
|
||
{
|
||
tracing::warn!(relay = %relay_url, %error, "Could not restore complete live coverage after relay CLOSED");
|
||
} else {
|
||
tracing::info!(relay = %relay_url, "Restored complete live coverage after relay CLOSED");
|
||
}
|
||
}
|
||
|
||
async fn sync_due_byte_limited_relay(&mut self) {
|
||
let now = Instant::now();
|
||
let due = self
|
||
.byte_limited_live_relays
|
||
.iter()
|
||
.find_map(|(relay, deadline)| (*deadline <= now).then(|| relay.clone()));
|
||
let Some(relay_url) = due else {
|
||
return;
|
||
};
|
||
|
||
if self.has_pending_batches(&relay_url).await {
|
||
self.byte_limited_live_relays
|
||
.insert(relay_url, now + Duration::from_secs(10));
|
||
return;
|
||
}
|
||
|
||
let connected = self
|
||
.relay_sync_index
|
||
.read()
|
||
.await
|
||
.get(&relay_url)
|
||
.is_some_and(|state| state.connection_status.is_live_sync_active());
|
||
if !connected {
|
||
self.byte_limited_live_relays
|
||
.insert(relay_url, now + byte_limited_catchup_interval());
|
||
return;
|
||
}
|
||
|
||
self.byte_limited_live_relays
|
||
.insert(relay_url.clone(), now + byte_limited_catchup_interval());
|
||
let overlap = byte_limited_catchup_interval() + Duration::from_secs(60);
|
||
let since = Timestamp::from(Timestamp::now().as_secs().saturating_sub(overlap.as_secs()));
|
||
let filters = self.complete_live_filters(&relay_url, Some(since)).await;
|
||
let items = self.desired_items_for_relay(&relay_url).await;
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
since = %since,
|
||
"Starting paced catch-up for byte-limited persistent coverage"
|
||
);
|
||
self.historic_sync(&relay_url, filters, items, Some(since))
|
||
.await;
|
||
}
|
||
|
||
/// Check for relays that should be disconnected
|
||
///
|
||
/// This method is called periodically by run_disconnect_checker.
|
||
/// It identifies non-bootstrap relays that have no repos or root events
|
||
/// to sync and disconnects them to free up resources.
|
||
///
|
||
/// Bootstrap relays are NEVER disconnected, even if empty.
|
||
async fn check_disconnects(&mut self) {
|
||
let now = Instant::now();
|
||
self.dependency_relay_deadlines
|
||
.retain(|_, deadline| *deadline > now);
|
||
|
||
let mut desired_relays: HashSet<String> = self.derive_targets().await.into_keys().collect();
|
||
desired_relays.extend(self.dependency_relay_deadlines.keys().cloned());
|
||
|
||
// Collect relays to disconnect
|
||
let to_disconnect: Vec<String> = {
|
||
let pending = self.pending_sync_index.read().await;
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.iter()
|
||
.filter_map(|(relay_url, state)| {
|
||
let has_pending_batches = pending
|
||
.get(relay_url)
|
||
.is_some_and(|batches| !batches.is_empty());
|
||
|
||
if state.is_disconnect_candidate(
|
||
has_pending_batches,
|
||
desired_relays.contains(relay_url),
|
||
) {
|
||
Some(relay_url.clone())
|
||
} else {
|
||
None
|
||
}
|
||
})
|
||
.collect()
|
||
};
|
||
|
||
if to_disconnect.is_empty() {
|
||
tracing::trace!("No empty relays to disconnect");
|
||
return;
|
||
}
|
||
|
||
tracing::info!(
|
||
count = to_disconnect.len(),
|
||
relay_sample = ?to_disconnect.iter().take(LOG_COLLECTION_SAMPLE_SIZE).collect::<Vec<_>>(),
|
||
"Found empty non-bootstrap relays to disconnect"
|
||
);
|
||
|
||
// Disconnect empty relays
|
||
for relay_url in to_disconnect {
|
||
self.disconnect_relay(&relay_url).await;
|
||
}
|
||
}
|
||
|
||
/// Disconnect a relay and mark it for cleanup
|
||
///
|
||
/// This method:
|
||
/// - Marks the relay as Disconnecting in relay_sync_index
|
||
/// - Initiates the connection disconnect
|
||
/// - Final cleanup happens in handle_disconnect when event loop terminates
|
||
///
|
||
/// Used by check_disconnects for cleanup of empty relays.
|
||
async fn disconnect_relay(&mut self, relay_url: &str) {
|
||
let prior_status = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index.get(relay_url).map(|state| state.connection_status)
|
||
};
|
||
if prior_status == Some(ConnectionStatus::Disconnecting) {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Relay disconnect already in progress"
|
||
);
|
||
return;
|
||
}
|
||
|
||
tracing::info!(relay = %relay_url, "Initiating disconnect for empty relay");
|
||
|
||
// Mark relay as Disconnecting (keep state for event loop to drain)
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
if let Some(state) = index.get_mut(relay_url) {
|
||
state.connection_status = ConnectionStatus::Disconnecting;
|
||
state.disconnected_at = Some(Timestamp::now());
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Marked relay as Disconnecting"
|
||
);
|
||
}
|
||
}
|
||
|
||
// Initiate disconnect - event loop will drain and send disconnect notification
|
||
if let Some(connection) = self.connections.get(relay_url) {
|
||
connection.disconnect().await;
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"Initiated connection disconnect"
|
||
);
|
||
}
|
||
|
||
// Update metrics
|
||
if let Some(ref metrics) = self.metrics {
|
||
metrics.record_connection_status(relay_url, ConnectionStatus::Disconnecting);
|
||
}
|
||
|
||
// An unexpected disconnect has already ended the event loop. There is
|
||
// no future notification to finish an intentional retirement.
|
||
if prior_status == Some(ConnectionStatus::Disconnected) {
|
||
self.complete_ended_session(relay_url, false).await;
|
||
return;
|
||
}
|
||
|
||
tracing::info!(relay = %relay_url, "Disconnect initiated, waiting for event loop termination");
|
||
}
|
||
|
||
/// Retry disconnected relays that are ready for reconnection
|
||
///
|
||
/// This method is called periodically by run_disconnect_checker.
|
||
/// It identifies relays that:
|
||
/// - Are currently disconnected
|
||
/// - Have repos or root events to sync (not empty)
|
||
/// - Have passed the exponential backoff period (respects health tracker)
|
||
///
|
||
/// For each eligible relay, a reconnection is queued via schedule_connect_relay.
|
||
async fn retry_disconnected_relays(&mut self) {
|
||
let desired_relays: HashSet<String> = self.derive_targets().await.into_keys().collect();
|
||
|
||
// Collect relays to reconnect
|
||
let to_reconnect: Vec<String> = {
|
||
let index = self.relay_sync_index.read().await;
|
||
index
|
||
.iter()
|
||
.filter_map(|(relay_url, state)| {
|
||
// Only consider disconnected relays
|
||
if state.connection_status != ConnectionStatus::Disconnected {
|
||
return None;
|
||
}
|
||
|
||
// A source can have desired StateOnly invitation work before
|
||
// its first successful historic batch confirms anything.
|
||
if state.repos.is_empty()
|
||
&& state.state_only_repos.is_empty()
|
||
&& state.root_events.is_empty()
|
||
&& !desired_relays.contains(relay_url)
|
||
{
|
||
return None;
|
||
}
|
||
|
||
// Check if backoff period has elapsed
|
||
if self.health_tracker.should_attempt_connection(relay_url) {
|
||
Some(relay_url.clone())
|
||
} else {
|
||
None
|
||
}
|
||
})
|
||
.collect()
|
||
};
|
||
|
||
if to_reconnect.is_empty() {
|
||
tracing::trace!("No disconnected relays ready for reconnection");
|
||
return;
|
||
}
|
||
|
||
tracing::info!(
|
||
count = to_reconnect.len(),
|
||
relay_sample = ?to_reconnect.iter().take(LOG_COLLECTION_SAMPLE_SIZE).collect::<Vec<_>>(),
|
||
"Attempting reconnection for disconnected relays"
|
||
);
|
||
|
||
// Reconnect eligible relays
|
||
for relay_url in to_reconnect {
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
health_state = %self.health_tracker.get_state(&relay_url),
|
||
"Attempting reconnection"
|
||
);
|
||
self.schedule_connect_relay(&relay_url).await;
|
||
}
|
||
}
|
||
|
||
/// Check for rate-limited relays that have exceeded cooldown
|
||
///
|
||
/// This method is called by the health and metrics checker every 2 seconds.
|
||
/// For each relay in RateLimited state that has exceeded the 65-second cooldown:
|
||
/// 1. Clears the rate limit state (sets to Healthy)
|
||
/// 2. Recomputes required actions for that relay
|
||
/// 3. Submits those actions
|
||
async fn check_rate_limit_recovery(&mut self) {
|
||
use crate::sync::algorithms::compute_actions;
|
||
|
||
// Exit rate limiting for relays whose cooldown has expired
|
||
let mut relays_to_recover: Vec<String> = self.health_tracker.exit_expired_rate_limits();
|
||
let policy_relays = self.health_tracker.exit_expired_policy_refusals();
|
||
for relay in policy_relays {
|
||
if !relays_to_recover.contains(&relay) {
|
||
relays_to_recover.push(relay);
|
||
}
|
||
}
|
||
|
||
if relays_to_recover.is_empty() {
|
||
return;
|
||
}
|
||
|
||
// Recompute actions - could optimise by adding relays: Option<&[]> to derive_relay_targets
|
||
let targets = self.derive_targets().await;
|
||
|
||
for relay_url in relays_to_recover {
|
||
tracing::info!(relay = %relay_url, "Subscription pause expired; probing coverage");
|
||
|
||
// A rate-limited CLOSED deliberately leaves live coverage down so
|
||
// it cannot immediately recreate the rejected request burst.
|
||
// Restore persistent coverage first when the relay admits work again.
|
||
if let Some(generation) = self
|
||
.connections
|
||
.get(&relay_url)
|
||
.map(RelayConnection::current_subscription_generation)
|
||
{
|
||
self.restore_live_coverage_after_closed(&relay_url, generation)
|
||
.await;
|
||
}
|
||
|
||
// Only compute actions for this specific relay
|
||
if let Some(relay_needs) = targets.get(&relay_url) {
|
||
let mut single_relay_targets = std::collections::HashMap::new();
|
||
single_relay_targets.insert(relay_url.clone(), relay_needs.clone());
|
||
|
||
let pending = self.pending_sync_index.read().await;
|
||
let confirmed = self.relay_sync_index.read().await;
|
||
|
||
let actions = compute_actions(&single_relay_targets, &pending, &confirmed);
|
||
drop(pending);
|
||
drop(confirmed);
|
||
|
||
// Submit each action
|
||
for action in actions {
|
||
tracing::info!(
|
||
relay = %action.relay_url,
|
||
full_repo_count = action.items.repos.len(),
|
||
state_only_repo_count = action.items.state_only_repos.len(),
|
||
event_count = action.items.root_events.len(),
|
||
"Submitting recovered actions after subscription pause"
|
||
);
|
||
self.handle_new_sync_filters(action).await;
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
/// Subscribe to filters for live (ongoing) events - NOT tracked in PendingSyncIndex
|
||
///
|
||
/// This method applies limit(0) to all filters to receive ONLY new events.
|
||
/// Per NIP-01, limit 0 means "send no stored events, only future events", which
|
||
/// ensures EOSE is received immediately and all subsequent events are tagged as "live"
|
||
/// in metrics (not "startup").
|
||
///
|
||
/// **Important**: Callers pass the SAME filters to both sync_live() and historic_sync().
|
||
/// This method applies limit(0) to prevent fetching historic events.
|
||
///
|
||
/// Live subscriptions are NOT tracked in PendingSyncIndex because they don't have
|
||
/// a definite "completion" - they stay open indefinitely.
|
||
///
|
||
/// Used for:
|
||
/// - Layer 1 live subscription (new announcements after initial sync)
|
||
/// - Layer 2+3 live subscriptions (new events after initial sync)
|
||
///
|
||
/// # Arguments
|
||
/// * `relay_url` - The relay URL to subscribe on
|
||
/// * `filters` - Filters to subscribe to (limit(0) will be applied)
|
||
///
|
||
/// # Returns
|
||
/// Vec of subscription IDs for the live subscriptions, or empty if connection not found
|
||
async fn sync_live(
|
||
&mut self,
|
||
relay_url: &str,
|
||
filters: &[Filter],
|
||
) -> Result<Vec<SubscriptionId>, String> {
|
||
if filters.is_empty() {
|
||
return Ok(vec![]);
|
||
}
|
||
|
||
let connection = match self.connections.get(relay_url) {
|
||
Some(conn) => conn.clone(),
|
||
None => {
|
||
tracing::debug!(relay = %relay_url, "No connection found for live sync");
|
||
return Err(format!("No connection found for live sync on {relay_url}"));
|
||
}
|
||
};
|
||
|
||
let filter_groups = live_filter_groups(filters, connection.max_filters_per_req());
|
||
let remote_limit = connection.remote_subscription_byte_limit();
|
||
let (filter_groups, overflow_groups) = groups_within_subscription_byte_limit(
|
||
filter_groups,
|
||
remote_limit,
|
||
connection.live_subscription_bytes(),
|
||
);
|
||
if overflow_groups > 0 {
|
||
self.byte_limited_live_relays.insert(
|
||
relay_url.to_string(),
|
||
Instant::now() + byte_limited_catchup_interval(),
|
||
);
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
remote_limit,
|
||
overflow_groups,
|
||
"Persistent coverage bounded by remote subscription-state limit; overflow will use paced catch-up"
|
||
);
|
||
}
|
||
if filter_groups.is_empty() {
|
||
return Ok(Vec::new());
|
||
}
|
||
let protected_subscription_ids = self
|
||
.descendant_live_coverage
|
||
.get(relay_url)
|
||
.map(|coverage| coverage.subscription_ids.clone())
|
||
.unwrap_or_default();
|
||
connection
|
||
.extend_live_filter_groups_minimally(
|
||
filter_groups,
|
||
&protected_subscription_ids,
|
||
)
|
||
.await
|
||
.inspect_err(|error| {
|
||
tracing::error!(relay = %relay_url, error = %error, "Failed to create complete live subscription set");
|
||
})
|
||
}
|
||
|
||
/// Submit grouped REQ+EOSE filters through the connection's transient
|
||
/// queue and return the subscriptions that actually started.
|
||
///
|
||
/// Callers submit only the current group while the connection waits for a
|
||
/// shared-ledger permit, rather than pre-queuing an unbounded historic
|
||
/// batch. The connection owns each permit until EOSE/CLOSED (or watchdog
|
||
/// recovery).
|
||
async fn subscribe_historic_filter_groups(
|
||
&self,
|
||
relay_url: &str,
|
||
batch_id: u64,
|
||
filters: &[Filter],
|
||
) -> (
|
||
HashSet<SubscriptionId>,
|
||
HashMap<SubscriptionId, PaginationState>,
|
||
) {
|
||
let mut subscription_ids = HashSet::new();
|
||
let mut pagination_state = HashMap::new();
|
||
let max_filters = self
|
||
.connections
|
||
.get(relay_url)
|
||
.map(RelayConnection::max_filters_per_req)
|
||
.unwrap_or(MAX_FILTERS_PER_REQ);
|
||
|
||
for (group_idx, filter_group) in group_filters_for_req_with_max(filters, max_filters)
|
||
.into_iter()
|
||
.enumerate()
|
||
{
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
group_idx,
|
||
filter_count = filter_group.len(),
|
||
filters = ?filter_group,
|
||
"Subscribing to grouped filters in REQ+EOSE path"
|
||
);
|
||
|
||
if let Some(connection) = self.connections.get(relay_url) {
|
||
match connection
|
||
.subscribe_filters(filter_group.clone(), TransientRequestClass::HistoricPage)
|
||
.await
|
||
{
|
||
Ok(subscription_id) => {
|
||
subscription_ids.insert(subscription_id.clone());
|
||
pagination_state
|
||
.insert(subscription_id, PaginationState::new(filter_group));
|
||
}
|
||
Err(error) => {
|
||
tracing::error!(
|
||
relay = %relay_url,
|
||
batch_id,
|
||
group_idx,
|
||
error = %error,
|
||
"Failed to subscribe to filter in historic_sync"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
(subscription_ids, pagination_state)
|
||
}
|
||
|
||
/// Sync historical events and track in PendingSyncIndex
|
||
///
|
||
/// This method handles historical synchronization for a set of filters,
|
||
/// creating a PendingBatch to track completion. It dispatches to either
|
||
/// negentropy sync or traditional REQ+EOSE based on relay capability and config.
|
||
///
|
||
/// Used for:
|
||
/// - Initial sync (no since filter)
|
||
/// - Reconnect sync (with since filter)
|
||
/// - Daily sync (no since filter, full re-sync)
|
||
///
|
||
/// # Arguments
|
||
/// * `relay_url` - The relay URL to sync from
|
||
/// * `filters` - Filters to sync (will have `since` applied if provided)
|
||
/// * `items` - Items being synced (for tracking in PendingBatch)
|
||
/// * `since` - Optional timestamp for incremental sync
|
||
///
|
||
/// # Returns
|
||
/// * `Some(batch_id)` - Batch was created and sync initiated
|
||
/// * `None` - No connection or sync failed to start
|
||
async fn historic_sync(
|
||
&mut self,
|
||
relay_url: &str,
|
||
filters: Vec<Filter>,
|
||
items: PendingItems,
|
||
since: Option<Timestamp>,
|
||
) -> Option<u64> {
|
||
let filters = if items.repos.is_empty()
|
||
&& items.state_only_repos.is_empty()
|
||
&& items.root_events.is_empty()
|
||
{
|
||
filters
|
||
} else {
|
||
let frontier = self.descendant_thread_members(&items.root_events).await;
|
||
packed_historic_filters(&items, &frontier)
|
||
};
|
||
self.historic_sync_with_options(
|
||
relay_url,
|
||
filters,
|
||
items,
|
||
since,
|
||
PendingBatchPurpose::Core,
|
||
false,
|
||
)
|
||
.await
|
||
}
|
||
|
||
async fn historic_sync_with_options(
|
||
&mut self,
|
||
relay_url: &str,
|
||
filters: Vec<Filter>,
|
||
items: PendingItems,
|
||
since: Option<Timestamp>,
|
||
purpose: PendingBatchPurpose,
|
||
force_req_eose: bool,
|
||
) -> Option<u64> {
|
||
// DEBUG TRACING: Log all filters being passed to historic_sync
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
filter_count = filters.len(),
|
||
filters = ?filters,
|
||
repos_count = items.repos.len(),
|
||
root_events_count = items.root_events.len(),
|
||
since = ?since,
|
||
"historic_sync called"
|
||
);
|
||
|
||
if filters.is_empty() && items.repos.is_empty() && items.root_events.is_empty() {
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
"historic_sync called with empty filters and items, skipping"
|
||
);
|
||
return None;
|
||
}
|
||
|
||
// Check connection exists and clone for async usage
|
||
let connection = match self.connections.get(relay_url) {
|
||
Some(conn) => conn.clone(),
|
||
None => {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
"No connection found for historic_sync"
|
||
);
|
||
return None;
|
||
}
|
||
};
|
||
|
||
// Apply since filter if provided
|
||
let filters_with_since: Vec<Filter> = if let Some(ts) = since {
|
||
filters.into_iter().map(|f| f.since(ts)).collect()
|
||
} else {
|
||
filters
|
||
};
|
||
|
||
// Check if we should use negentropy
|
||
let use_negentropy = !force_req_eose
|
||
&& !self.config.sync_disable_negentropy
|
||
&& connection.supports_negentropy().await;
|
||
|
||
// Generate batch ID
|
||
let batch_id = self.next_batch_id();
|
||
|
||
// Track whether negentropy succeeded (for fallback logic)
|
||
let mut negentropy_succeeded = false;
|
||
|
||
if use_negentropy && !filters_with_since.is_empty() {
|
||
// NIP-77 negentropy path
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
filter_count = filters_with_since.len(),
|
||
repos = items.repos.len(),
|
||
root_events = items.root_events.len(),
|
||
"Starting historic_sync with negentropy"
|
||
);
|
||
|
||
// Create PendingBatch for negentropy (empty outstanding_subs and pagination_state)
|
||
let batch = PendingBatch {
|
||
batch_id,
|
||
purpose,
|
||
items: items.clone(),
|
||
outstanding_subs: HashSet::new(),
|
||
sync_method: SyncMethod::Negentropy,
|
||
pagination_state: HashMap::new(), // Negentropy doesn't use pagination
|
||
requested_event_ids: None, // Will be set after negentropy diff
|
||
received_event_ids: None, // Will be set after negentropy diff
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
|
||
// Add to pending_sync_index
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
pending
|
||
.entry(relay_url.to_string())
|
||
.or_insert_with(Vec::new)
|
||
.push(batch);
|
||
}
|
||
|
||
// Perform negentropy sync for all filters concurrently
|
||
// Note: We sync each filter separately because negentropy works on a single filter
|
||
let diff_futures: Vec<_> = filters_with_since
|
||
.iter()
|
||
.enumerate()
|
||
.map(|(idx, filter)| {
|
||
let filter = filter.clone();
|
||
let conn = connection.clone();
|
||
async move { (idx, conn.negentropy_sync_diff(filter).await) }
|
||
})
|
||
.collect();
|
||
|
||
let diff_results = futures_util::future::join_all(diff_futures).await;
|
||
|
||
// Process results - collect all event IDs we need to fetch
|
||
let mut all_remote_ids = Vec::new();
|
||
let mut failed_count = 0;
|
||
|
||
// Get event IDs to exclude: purgatory + rejected announcements
|
||
let purgatory_ids = self.purgatory.event_ids();
|
||
let rejected_ids = self.rejected_events_index.get_all_event_ids();
|
||
let excluded_ids: HashSet<EventId> =
|
||
purgatory_ids.union(&rejected_ids).cloned().collect();
|
||
|
||
for (idx, result) in diff_results {
|
||
match result {
|
||
Ok(reconciliation) => {
|
||
let remote_excluding_ids: HashSet<EventId> = reconciliation
|
||
.remote
|
||
.keys()
|
||
.filter(|id| !excluded_ids.contains(id))
|
||
.copied()
|
||
.collect();
|
||
let remote_count = remote_excluding_ids.len();
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
filter_idx = idx,
|
||
remote_count = remote_count,
|
||
local_count = reconciliation.local.len(),
|
||
"Negentropy diff completed for filter"
|
||
);
|
||
if remote_count > 0 {
|
||
all_remote_ids.extend(remote_excluding_ids);
|
||
}
|
||
}
|
||
Err(e) => {
|
||
failed_count += 1;
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
filter_idx = idx,
|
||
error = %e,
|
||
"Negentropy diff failed for filter in historic_sync"
|
||
);
|
||
}
|
||
}
|
||
}
|
||
|
||
// Require ALL filters to succeed to confirm the batch
|
||
if failed_count > 0 {
|
||
// Remove failed negentropy batch and fall back to REQ+EOSE
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(relay_url) {
|
||
let batch_idx = batches.iter().position(|b| b.batch_id == batch_id);
|
||
if let Some(idx) = batch_idx {
|
||
batches.remove(idx);
|
||
if batches.is_empty() {
|
||
pending.remove(relay_url);
|
||
}
|
||
}
|
||
}
|
||
}
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
failed_count = failed_count,
|
||
total_filters = filters_with_since.len(),
|
||
"historic_sync (negentropy) failed - falling back to REQ+EOSE"
|
||
);
|
||
|
||
// Fall through to REQ+EOSE path below
|
||
} else {
|
||
// Negentropy succeeded - mark success and process results
|
||
negentropy_succeeded = true;
|
||
|
||
if all_remote_ids.is_empty() {
|
||
// Remove batch from pending and confirm it (no items to download)
|
||
let completed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batches) = pending.get_mut(relay_url) {
|
||
let batch_idx = batches.iter().position(|b| b.batch_id == batch_id);
|
||
if let Some(idx) = batch_idx {
|
||
let batch = batches.remove(idx);
|
||
if batches.is_empty() {
|
||
pending.remove(relay_url);
|
||
}
|
||
Some(batch)
|
||
} else {
|
||
None
|
||
}
|
||
} else {
|
||
None
|
||
}
|
||
};
|
||
|
||
if let Some(batch) = completed_batch {
|
||
self.confirm_batch(relay_url, batch).await;
|
||
}
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
total_received = 0,
|
||
"historic_sync (negentropy) completed - already up-to-date"
|
||
);
|
||
|
||
// Batch already confirmed, nothing more to do
|
||
return Some(batch_id);
|
||
}
|
||
|
||
// launch subscriptions to fetch missing events by id
|
||
let ids_filters: Vec<_> = all_remote_ids
|
||
.chunks(300)
|
||
.map(|c| Filter::new().ids(c.iter().copied()))
|
||
.collect();
|
||
|
||
tracing::info!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
total_event_ids = all_remote_ids.len(),
|
||
filter_chunks = ids_filters.len(),
|
||
"Creating subscriptions to fetch missing events by ID"
|
||
);
|
||
|
||
let planned_subscriptions: Vec<_> = ids_filters
|
||
.into_iter()
|
||
.map(|filter| (SubscriptionId::generate(), filter))
|
||
.collect();
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(relay_batches) = pending.get_mut(relay_url) {
|
||
if let Some(batch) =
|
||
relay_batches.iter_mut().find(|b| b.batch_id == batch_id)
|
||
{
|
||
register_negentropy_hydration_attempt(
|
||
batch,
|
||
planned_subscriptions
|
||
.iter()
|
||
.map(|(subscription_id, _)| subscription_id.clone()),
|
||
all_remote_ids.iter().copied(),
|
||
false,
|
||
);
|
||
}
|
||
}
|
||
}
|
||
for (idx, (subscription_id, filter)) in planned_subscriptions.iter().enumerate() {
|
||
let result = if let Some(conn) = self.connections.get(relay_url) {
|
||
conn.subscribe_filter_with_id(
|
||
filter.clone(),
|
||
TransientRequestClass::NegentropyHydration,
|
||
subscription_id.clone(),
|
||
)
|
||
.await
|
||
} else {
|
||
Err("Relay connection disappeared before hydration REQ".to_string())
|
||
};
|
||
if let Err(e) = result {
|
||
tracing::error!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
chunk_idx = idx,
|
||
error = %e,
|
||
"Failed to subscribe to ID filter chunk"
|
||
);
|
||
let completed_batch = {
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
if let Some(batch) = pending.get_mut(relay_url).and_then(|batches| {
|
||
batches.iter_mut().find(|batch| batch.batch_id == batch_id)
|
||
}) {
|
||
batch.outstanding_subs.remove(subscription_id);
|
||
}
|
||
take_drained_batch_as_failed(&mut pending, relay_url, batch_id)
|
||
};
|
||
if let Some(batch) = completed_batch {
|
||
self.confirm_batch(relay_url, batch).await;
|
||
}
|
||
}
|
||
}
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
subscription_ids = planned_subscriptions.len(),
|
||
events = all_remote_ids.len(),
|
||
"historic_sync (Negentropy) created subscriptions to fetch missing events by id, awaiting EOSE"
|
||
);
|
||
}
|
||
}
|
||
|
||
// Use REQ+EOSE if negentropy was not attempted or failed
|
||
if !negentropy_succeeded {
|
||
// Traditional REQ+EOSE path
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
filter_count = filters_with_since.len(),
|
||
repos = items.repos.len(),
|
||
root_events = items.root_events.len(),
|
||
use_negentropy = use_negentropy,
|
||
"Starting historic_sync with REQ+EOSE"
|
||
);
|
||
|
||
// Keep several OR filters under each relay-visible subscription.
|
||
let (subscription_ids, pagination_state) = self
|
||
.subscribe_historic_filter_groups(relay_url, batch_id, &filters_with_since)
|
||
.await;
|
||
|
||
if subscription_ids.is_empty() && !filters_with_since.is_empty() {
|
||
tracing::warn!(
|
||
relay = %relay_url,
|
||
"All filter subscriptions failed in historic_sync"
|
||
);
|
||
return None;
|
||
}
|
||
|
||
// Create PendingBatch for REQ+EOSE
|
||
let batch = PendingBatch {
|
||
batch_id,
|
||
purpose,
|
||
items,
|
||
outstanding_subs: subscription_ids,
|
||
sync_method: SyncMethod::ReqEose,
|
||
pagination_state,
|
||
requested_event_ids: None, // Not used for REQ+EOSE
|
||
received_event_ids: None, // Not used for REQ+EOSE
|
||
initial_hydration_counts: None,
|
||
retry_count: 0, // Not used for REQ+EOSE
|
||
failed: false,
|
||
};
|
||
|
||
// Add to pending_sync_index
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
pending
|
||
.entry(relay_url.to_string())
|
||
.or_insert_with(Vec::new)
|
||
.push(batch);
|
||
}
|
||
|
||
tracing::debug!(
|
||
relay = %relay_url,
|
||
batch_id = batch_id,
|
||
"historic_sync (REQ+EOSE) batch created, awaiting EOSE"
|
||
);
|
||
}
|
||
|
||
Some(batch_id)
|
||
}
|
||
|
||
/// Gracefully shutdown the SyncManager
|
||
///
|
||
/// This method:
|
||
/// - Sends shutdown signal to all background tasks (daily timer, disconnect checker)
|
||
/// - Disconnects all relay connections
|
||
/// - Clears all indices (relay_sync_index, pending_sync_index)
|
||
///
|
||
/// After calling this method, the SyncManager is no longer usable.
|
||
pub async fn shutdown(&mut self) {
|
||
tracing::info!("Starting SyncManager shutdown");
|
||
|
||
// 1. Send shutdown signal to all background tasks
|
||
if let Some(tx) = &self.shutdown_tx {
|
||
let _ = tx.send(());
|
||
tracing::debug!("Sent shutdown signal to background tasks");
|
||
}
|
||
|
||
// 2. Disconnect all relay connections
|
||
let relay_urls: Vec<String> = self.connections.keys().cloned().collect();
|
||
for relay_url in relay_urls {
|
||
if let Some(connection) = self.connections.remove(&relay_url) {
|
||
tracing::debug!(relay = %relay_url, "Disconnecting relay");
|
||
connection.disconnect().await;
|
||
}
|
||
}
|
||
|
||
// 3. Clear all indices
|
||
{
|
||
let mut index = self.relay_sync_index.write().await;
|
||
let count = index.len();
|
||
index.clear();
|
||
tracing::debug!(count = count, "Cleared relay_sync_index");
|
||
}
|
||
|
||
{
|
||
let mut pending = self.pending_sync_index.write().await;
|
||
let count = pending.len();
|
||
pending.clear();
|
||
tracing::debug!(count = count, "Cleared pending_sync_index");
|
||
}
|
||
self.deferred_consolidations.clear();
|
||
|
||
tracing::info!("SyncManager shutdown complete");
|
||
}
|
||
}
|
||
|
||
#[cfg(test)]
|
||
mod tests {
|
||
use super::*;
|
||
|
||
#[tokio::test]
|
||
async fn accepted_dependency_reprocesses_a_synced_policy_orphan() {
|
||
let directory = tempfile::tempdir().expect("create test directory");
|
||
let git_data_path = directory.path().join("git");
|
||
let mut config = Config::for_testing();
|
||
config.git_data_path = git_data_path.to_string_lossy().into_owned();
|
||
config.relay_data_path = directory
|
||
.path()
|
||
.join("relay")
|
||
.to_string_lossy()
|
||
.into_owned();
|
||
let purgatory = Arc::new(crate::purgatory::Purgatory::new(git_data_path));
|
||
let runtime = crate::nostr::builder::create_relay(
|
||
&config,
|
||
purgatory,
|
||
crate::grasp06::receive::RepoInitLocks::default(),
|
||
None,
|
||
)
|
||
.await
|
||
.expect("create test relay runtime");
|
||
let rejected = Arc::new(RejectedEventsIndex::new(
|
||
Duration::from_secs(120),
|
||
Duration::from_secs(604800),
|
||
));
|
||
let keys = Keys::generate();
|
||
let parent = EventBuilder::new(Kind::GitUserGraspList, "dependency")
|
||
.finalize(&keys)
|
||
.expect("build accepted dependency");
|
||
let child = EventBuilder::new(Kind::TextNote, "dependent comment")
|
||
.tags([Tag::event(parent.id)])
|
||
.finalize(&keys)
|
||
.expect("build dependent event");
|
||
|
||
let orphan_result = SyncManager::process_event_static(
|
||
&child,
|
||
"wss://source.example",
|
||
&runtime.stores.database,
|
||
&runtime.write_policy,
|
||
&runtime.relay,
|
||
&rejected,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
assert_eq!(
|
||
orphan_result,
|
||
ProcessResult::Rejected(PolicyRejection::Restricted)
|
||
);
|
||
assert!(rejected.is_dependency_pending(&child.id));
|
||
assert!(runtime
|
||
.stores
|
||
.database
|
||
.event_by_id(&child.id)
|
||
.await
|
||
.unwrap()
|
||
.is_none());
|
||
|
||
let parent_result = SyncManager::process_event_static(
|
||
&parent,
|
||
"wss://source.example",
|
||
&runtime.stores.database,
|
||
&runtime.write_policy,
|
||
&runtime.relay,
|
||
&rejected,
|
||
crate::nostr::persistence::SaveContext::RelaySync,
|
||
)
|
||
.await;
|
||
assert_eq!(parent_result, ProcessResult::Saved);
|
||
assert!(runtime
|
||
.stores
|
||
.database
|
||
.event_by_id(&child.id)
|
||
.await
|
||
.unwrap()
|
||
.is_some());
|
||
assert!(!rejected.contains(&child.id));
|
||
}
|
||
|
||
#[test]
|
||
fn private_members_include_only_owners_of_accepted_relays() {
|
||
let configured = Keys::generate().public_key();
|
||
let accepted_owner = Keys::generate().public_key();
|
||
let unrelated_owner = Keys::generate().public_key();
|
||
let members = effective_private_members(
|
||
&HashSet::from([configured]),
|
||
&HashSet::from(["wss://accepted.example/".to_string()]),
|
||
&HashMap::from([
|
||
("wss://accepted.example/".to_string(), accepted_owner),
|
||
("wss://unrelated.example/".to_string(), unrelated_owner),
|
||
]),
|
||
);
|
||
|
||
assert_eq!(members, HashSet::from([configured, accepted_owner]));
|
||
}
|
||
|
||
#[test]
|
||
fn partial_nip65_batch_retries_missing_authors_early() {
|
||
let returned = Keys::generate().public_key();
|
||
let missing = Keys::generate().public_key();
|
||
let represented = HashSet::from([returned]);
|
||
|
||
assert_eq!(
|
||
nip65_author_retry_after(true, &represented, &returned),
|
||
nip65_discovery_refresh_interval()
|
||
);
|
||
assert_eq!(
|
||
nip65_author_retry_after(true, &represented, &missing),
|
||
nip65_discovery_retry_interval()
|
||
);
|
||
assert_eq!(
|
||
nip65_author_retry_after(false, &represented, &returned),
|
||
nip65_discovery_retry_interval()
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn retained_list_does_not_mask_an_empty_new_outbox_response() {
|
||
let author_with_retained_list = Keys::generate().public_key();
|
||
// `represented_by_source` is intentionally empty: a relay list held
|
||
// in the local database came from an earlier source, while this newly
|
||
// followed outbox returned no list and needs the short retry cadence.
|
||
let represented_by_source = HashSet::new();
|
||
|
||
assert_eq!(
|
||
nip65_author_retry_after(true, &represented_by_source, &author_with_retained_list,),
|
||
nip65_discovery_retry_interval()
|
||
);
|
||
let represented_by_source = HashSet::from([author_with_retained_list]);
|
||
assert_eq!(
|
||
nip65_author_retry_after(true, &represented_by_source, &author_with_retained_list,),
|
||
nip65_discovery_refresh_interval()
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn event_pipeline_window_separates_queue_and_processing_costs() {
|
||
let mut window = EventPipelineWindow::default();
|
||
window.record(
|
||
ProcessResult::Duplicate,
|
||
std::time::Duration::from_millis(12),
|
||
std::time::Duration::from_millis(3),
|
||
);
|
||
window.record(
|
||
ProcessResult::Saved,
|
||
std::time::Duration::from_millis(8),
|
||
std::time::Duration::from_millis(7),
|
||
);
|
||
|
||
assert_eq!(window.delivered, 2);
|
||
assert_eq!(window.saved, 1);
|
||
assert_eq!(window.duplicate, 1);
|
||
assert_eq!(window.queue_delay, std::time::Duration::from_millis(20));
|
||
assert_eq!(window.max_queue_delay, std::time::Duration::from_millis(12));
|
||
assert_eq!(window.processing_time, std::time::Duration::from_millis(10));
|
||
assert_eq!(
|
||
window.max_processing_time,
|
||
std::time::Duration::from_millis(7)
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn hydration_outcomes_separate_terminal_and_retryable_failures() {
|
||
use nostr::message::relay::SingleWord;
|
||
use nostr_sdk::prelude::MachineReadablePrefix;
|
||
|
||
let purgatory = MachineReadablePrefix::Custom(
|
||
SingleWord::from_static("purgatory").expect("valid custom prefix"),
|
||
);
|
||
assert_eq!(
|
||
SyncManager::classify_policy_result(
|
||
&purgatory,
|
||
"won't be served until git data arrives",
|
||
true,
|
||
),
|
||
ProcessResult::Purgatory
|
||
);
|
||
assert_eq!(
|
||
SyncManager::classify_policy_result(
|
||
&MachineReadablePrefix::Invalid,
|
||
"this event is deleted",
|
||
false,
|
||
),
|
||
ProcessResult::Tombstoned
|
||
);
|
||
|
||
let invalid = SyncManager::classify_policy_result(
|
||
&MachineReadablePrefix::Invalid,
|
||
"malformed event",
|
||
false,
|
||
);
|
||
let restricted = SyncManager::classify_policy_result(
|
||
&MachineReadablePrefix::Restricted,
|
||
"dependency not accepted yet",
|
||
false,
|
||
);
|
||
let persistence_error = ProcessResult::PersistenceError;
|
||
|
||
assert_eq!(invalid.hydration_outcome(), "rejected_invalid");
|
||
assert!(invalid.is_terminally_accounted());
|
||
assert_eq!(restricted.hydration_outcome(), "rejected_restricted");
|
||
assert!(!restricted.is_terminally_accounted());
|
||
assert!(!persistence_error.is_terminally_accounted());
|
||
}
|
||
|
||
#[test]
|
||
fn lifecycle_inbox_accepts_production_sized_terminal_burst() {
|
||
let (tx, mut rx) = lifecycle_notification_channel::<EoseNotification>();
|
||
for index in 0..169 {
|
||
tx.try_send(EoseNotification {
|
||
relay_url: "wss://relay.example".to_string(),
|
||
sub_id: SubscriptionId::new(format!("hydration-{index}")),
|
||
})
|
||
.expect("the actor-side receiver remains alive");
|
||
}
|
||
|
||
assert_eq!(rx.len(), 169);
|
||
for _ in 0..169 {
|
||
rx.try_recv().expect("every terminal remains queued");
|
||
}
|
||
assert!(rx.try_recv().is_err());
|
||
}
|
||
|
||
#[test]
|
||
fn descendant_frontier_derives_replaceable_and_addressable_coordinates() {
|
||
let keys = Keys::generate();
|
||
let addressable = EventBuilder::new(Kind::Custom(30_023), "addressable")
|
||
.tag(Tag::custom("d", ["article-name"]))
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
let replaceable = EventBuilder::new(Kind::Custom(10_000), "replaceable")
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
let regular = EventBuilder::new(Kind::TextNote, "regular")
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
let malformed_addressable = EventBuilder::new(Kind::Custom(30_023), "missing d")
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
|
||
assert_eq!(
|
||
descendant_event_coordinate(&addressable),
|
||
Some(format!("30023:{}:article-name", keys.public_key().to_hex()))
|
||
);
|
||
assert_eq!(
|
||
descendant_event_coordinate(&replaceable),
|
||
Some(format!("10000:{}:", keys.public_key().to_hex()))
|
||
);
|
||
assert_eq!(descendant_event_coordinate(®ular), None);
|
||
assert_eq!(descendant_event_coordinate(&malformed_addressable), None);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn descendant_frontier_recurses_through_event_and_coordinate_references() {
|
||
let keys = Keys::generate();
|
||
let participant = Keys::generate();
|
||
let reactor = Keys::generate();
|
||
let root = EventBuilder::new(Kind::GitIssue, "root")
|
||
.finalize(&keys)
|
||
.expect("build root");
|
||
let addressable = EventBuilder::new(Kind::Custom(30_023), "first descendant")
|
||
.tags([Tag::identifier("thread-member"), Tag::event(root.id)])
|
||
.finalize(&keys)
|
||
.expect("build addressable descendant");
|
||
let coordinate = descendant_event_coordinate(&addressable).unwrap();
|
||
let coordinate_child = EventBuilder::new(Kind::TextNote, "coordinate child")
|
||
.tag(Tag::custom("a", [coordinate.clone()]))
|
||
.finalize(&participant)
|
||
.expect("build coordinate child");
|
||
let grandchild = EventBuilder::new(Kind::TextNote, "grandchild")
|
||
.tag(Tag::event(coordinate_child.id))
|
||
.finalize(&reactor)
|
||
.expect("build grandchild");
|
||
let unrelated = EventBuilder::new(Kind::TextNote, "unrelated")
|
||
.finalize(&keys)
|
||
.expect("build unrelated event");
|
||
let database: SharedDatabase = Arc::new(nostr_memory::MemoryDatabase::unbounded());
|
||
for event in [
|
||
&root,
|
||
&addressable,
|
||
&coordinate_child,
|
||
&grandchild,
|
||
&unrelated,
|
||
] {
|
||
database.save_event(event).await.expect("save test event");
|
||
}
|
||
|
||
let frontier =
|
||
recursive_descendant_frontier(&database, &HashSet::from([root.id]), 500).await;
|
||
|
||
assert_eq!(
|
||
frontier.event_ids,
|
||
HashSet::from([addressable.id, coordinate_child.id, grandchild.id])
|
||
);
|
||
assert_eq!(frontier.coordinates, HashSet::from([coordinate]));
|
||
assert_eq!(
|
||
frontier.author_roots,
|
||
HashMap::from([
|
||
(keys.public_key(), HashSet::from([root.id])),
|
||
(participant.public_key(), HashSet::from([root.id])),
|
||
(reactor.public_key(), HashSet::from([root.id])),
|
||
])
|
||
);
|
||
assert!(!frontier.event_ids.contains(&unrelated.id));
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn descendant_frontier_preserves_exact_root_provenance() {
|
||
let owner = Keys::generate();
|
||
let first_participant = Keys::generate();
|
||
let second_participant = Keys::generate();
|
||
let shared_participant = Keys::generate();
|
||
let first_root = EventBuilder::new(Kind::GitIssue, "first root")
|
||
.finalize(&owner)
|
||
.expect("build first root");
|
||
let second_root = EventBuilder::new(Kind::GitIssue, "second root")
|
||
.finalize(&owner)
|
||
.expect("build second root");
|
||
let first_reply = EventBuilder::new(Kind::TextNote, "first reply")
|
||
.tag(Tag::event(first_root.id))
|
||
.finalize(&first_participant)
|
||
.expect("build first reply");
|
||
let second_reply = EventBuilder::new(Kind::TextNote, "second reply")
|
||
.tag(Tag::event(second_root.id))
|
||
.finalize(&second_participant)
|
||
.expect("build second reply");
|
||
let shared_reply = EventBuilder::new(Kind::TextNote, "shared reply")
|
||
.tags([Tag::event(first_reply.id), Tag::event(second_reply.id)])
|
||
.finalize(&shared_participant)
|
||
.expect("build shared reply");
|
||
let database: SharedDatabase = Arc::new(nostr_memory::MemoryDatabase::unbounded());
|
||
for event in [
|
||
&first_root,
|
||
&second_root,
|
||
&first_reply,
|
||
&second_reply,
|
||
&shared_reply,
|
||
] {
|
||
database.save_event(event).await.expect("save test event");
|
||
}
|
||
|
||
let frontier = recursive_descendant_frontier(
|
||
&database,
|
||
&HashSet::from([first_root.id, second_root.id]),
|
||
500,
|
||
)
|
||
.await;
|
||
|
||
assert_eq!(
|
||
frontier.author_roots[&first_participant.public_key()],
|
||
HashSet::from([first_root.id])
|
||
);
|
||
assert_eq!(
|
||
frontier.author_roots[&second_participant.public_key()],
|
||
HashSet::from([second_root.id])
|
||
);
|
||
assert_eq!(
|
||
frontier.author_roots[&shared_participant.public_key()],
|
||
HashSet::from([first_root.id, second_root.id])
|
||
);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn descendant_frontier_stops_at_the_depth_bound() {
|
||
let keys = Keys::generate();
|
||
let root = EventBuilder::new(Kind::GitIssue, "root")
|
||
.finalize(&keys)
|
||
.expect("build root");
|
||
let mut chain = Vec::new();
|
||
let mut parent = root.id;
|
||
for depth in 1..=MAX_DESCENDANT_FRONTIER_DEPTH + 1 {
|
||
let child = EventBuilder::new(Kind::TextNote, format!("depth {depth}"))
|
||
.tag(Tag::event(parent))
|
||
.finalize(&keys)
|
||
.expect("build descendant");
|
||
parent = child.id;
|
||
chain.push(child);
|
||
}
|
||
let database: SharedDatabase = Arc::new(nostr_memory::MemoryDatabase::unbounded());
|
||
database.save_event(&root).await.expect("save root");
|
||
for event in &chain {
|
||
database.save_event(event).await.expect("save descendant");
|
||
}
|
||
|
||
let frontier =
|
||
recursive_descendant_frontier(&database, &HashSet::from([root.id]), 500).await;
|
||
|
||
assert_eq!(frontier.event_ids.len(), MAX_DESCENDANT_FRONTIER_DEPTH);
|
||
assert!(chain[..MAX_DESCENDANT_FRONTIER_DEPTH]
|
||
.iter()
|
||
.all(|event| frontier.event_ids.contains(&event.id)));
|
||
assert!(!frontier
|
||
.event_ids
|
||
.contains(&chain[MAX_DESCENDANT_FRONTIER_DEPTH].id));
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn descendant_frontier_preserves_direct_members_and_bounds_recursive_events() {
|
||
let keys = Keys::generate();
|
||
let root = EventBuilder::new(Kind::GitIssue, "root")
|
||
.finalize(&keys)
|
||
.expect("build root");
|
||
let direct = EventBuilder::new(Kind::TextNote, "direct member")
|
||
.tag(Tag::event(root.id))
|
||
.custom_created_at(Timestamp::from_secs(1))
|
||
.finalize(&keys)
|
||
.expect("build direct member");
|
||
let direct_siblings: Vec<Event> = (0..3)
|
||
.map(|index| {
|
||
EventBuilder::new(Kind::TextNote, format!("direct sibling {index}"))
|
||
.tag(Tag::event(root.id))
|
||
.custom_created_at(Timestamp::from_secs(2 + index))
|
||
.finalize(&keys)
|
||
.expect("build direct sibling")
|
||
})
|
||
.collect();
|
||
let mut children = Vec::new();
|
||
for created_at in [30, 10, 20] {
|
||
children.push(
|
||
EventBuilder::new(Kind::TextNote, format!("created at {created_at}"))
|
||
.tag(Tag::event(direct.id))
|
||
.custom_created_at(Timestamp::from_secs(created_at))
|
||
.finalize(&keys)
|
||
.expect("build child"),
|
||
);
|
||
}
|
||
let database: SharedDatabase = Arc::new(nostr_memory::MemoryDatabase::unbounded());
|
||
database
|
||
.save_event(&direct)
|
||
.await
|
||
.expect("save direct member");
|
||
for event in &direct_siblings {
|
||
database
|
||
.save_event(event)
|
||
.await
|
||
.expect("save direct sibling");
|
||
}
|
||
for event in &children {
|
||
database.save_event(event).await.expect("save child");
|
||
}
|
||
|
||
let frontier = recursive_descendant_frontier_with_limits(
|
||
&database,
|
||
&HashSet::from([root.id]),
|
||
MAX_DESCENDANT_FRONTIER_DEPTH,
|
||
2,
|
||
)
|
||
.await;
|
||
|
||
assert_eq!(frontier.event_ids.len(), 3);
|
||
assert!(!frontier.event_ids.contains(&direct.id));
|
||
assert!(direct_siblings
|
||
.iter()
|
||
.all(|event| frontier.event_ids.contains(&event.id)));
|
||
assert!(children
|
||
.iter()
|
||
.all(|child| !frontier.event_ids.contains(&child.id)));
|
||
}
|
||
|
||
#[test]
|
||
fn descendant_frontier_queries_event_and_coordinate_tag_variants() {
|
||
let since = Timestamp::from_secs(1234);
|
||
let frontier = DescendantFrontier {
|
||
event_ids: HashSet::from([EventId::from_byte_array([7; 32])]),
|
||
coordinates: HashSet::from([format!("30023:{}:article", "a".repeat(64))]),
|
||
..DescendantFrontier::default()
|
||
};
|
||
let filters = descendant_frontier_filters(&frontier, Some(since));
|
||
|
||
assert_eq!(filters.len(), 6, "e/E/q and a/A/q must all be covered");
|
||
assert!(filters.iter().all(|filter| {
|
||
serde_json::to_value(filter).unwrap()["since"] == serde_json::json!(1234)
|
||
}));
|
||
let serialized = filters.iter().map(Filter::as_json).collect::<Vec<_>>();
|
||
for tag in ["#e", "#E", "#a", "#A"] {
|
||
assert!(
|
||
serialized.iter().any(|filter| filter.contains(tag)),
|
||
"missing {tag} descendant filter"
|
||
);
|
||
}
|
||
assert_eq!(
|
||
serialized
|
||
.iter()
|
||
.filter(|filter| filter.contains("#q"))
|
||
.count(),
|
||
2,
|
||
"event IDs and coordinates each require quote coverage"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_probe_filters_cover_repository_roots_and_descendants() {
|
||
let root = EventId::from_byte_array([21; 32]);
|
||
let descendant = EventId::from_byte_array([22; 32]);
|
||
let repository = "30617:owner:repo".to_string();
|
||
let coordinate = "30023:participant:thread".to_string();
|
||
let frontier = DescendantFrontier {
|
||
event_ids: HashSet::from([descendant]),
|
||
coordinates: HashSet::from([coordinate.clone()]),
|
||
..Default::default()
|
||
};
|
||
|
||
let filters = mailbox_probe_filters(
|
||
&HashSet::from([repository.clone()]),
|
||
&HashSet::from([root]),
|
||
&frontier,
|
||
);
|
||
let serialized = filters.iter().map(Filter::as_json).collect::<Vec<_>>();
|
||
|
||
assert_eq!(filters.len(), 6, "a/A/q and e/E/q must be probed");
|
||
for value in [repository, coordinate, root.to_hex(), descendant.to_hex()] {
|
||
assert!(serialized.iter().any(|filter| filter.contains(&value)));
|
||
}
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_probe_order_prioritizes_longest_waiting_due_source() {
|
||
let now = Instant::now();
|
||
let root = EventId::from_byte_array([23; 32]);
|
||
let roots = HashMap::from([
|
||
("wss://old.example".to_string(), HashSet::from([root])),
|
||
(
|
||
"wss://large.example".to_string(),
|
||
HashSet::from([
|
||
root,
|
||
EventId::from_byte_array([24; 32]),
|
||
EventId::from_byte_array([25; 32]),
|
||
]),
|
||
),
|
||
("wss://future.example".to_string(), HashSet::from([root])),
|
||
]);
|
||
let next_at = HashMap::from([
|
||
(
|
||
"wss://old.example".to_string(),
|
||
now - Duration::from_secs(10),
|
||
),
|
||
(
|
||
"wss://large.example".to_string(),
|
||
now - Duration::from_secs(1),
|
||
),
|
||
(
|
||
"wss://future.example".to_string(),
|
||
now + Duration::from_secs(1),
|
||
),
|
||
]);
|
||
|
||
assert_eq!(
|
||
due_mailbox_relays(&roots, &next_at, now),
|
||
vec![
|
||
"wss://old.example".to_string(),
|
||
"wss://large.example".to_string(),
|
||
]
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_probe_waits_for_committed_connection_lifecycle() {
|
||
assert!(!mailbox_probe_connection_ready(
|
||
Some(ConnectionStatus::Connecting),
|
||
true,
|
||
));
|
||
assert!(mailbox_probe_connection_ready(
|
||
Some(ConnectionStatus::Syncing),
|
||
true,
|
||
));
|
||
assert!(!mailbox_probe_connection_ready(
|
||
Some(ConnectionStatus::Connected),
|
||
false,
|
||
));
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_probe_prefers_ready_relay_over_older_unavailable_relay() {
|
||
let unavailable = "wss://unavailable.example".to_string();
|
||
let ready = "wss://ready.example".to_string();
|
||
let due = vec![unavailable.clone(), ready.clone()];
|
||
|
||
assert_eq!(
|
||
select_due_mailbox_relay(&due, &HashSet::from([ready.clone()])),
|
||
Some(ready),
|
||
);
|
||
assert_eq!(
|
||
select_due_mailbox_relay(&due, &HashSet::new()),
|
||
Some(unavailable),
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_cursor_survives_inventory_growth_and_retires_with_source() {
|
||
let relay = "wss://mailbox.example".to_string();
|
||
let first_root = EventId::from_byte_array([31; 32]);
|
||
let second_root = EventId::from_byte_array([32; 32]);
|
||
let mut discovery = Nip65DiscoveryState::default();
|
||
discovery
|
||
.mailbox_roots
|
||
.insert(relay.clone(), HashSet::from([first_root]));
|
||
discovery.mailbox_probe_next_filter.insert(relay.clone(), 3);
|
||
|
||
discovery.install_mailbox_overlay(
|
||
HashMap::from([(relay.clone(), HashSet::from([first_root, second_root]))]),
|
||
Instant::now(),
|
||
);
|
||
assert_eq!(discovery.mailbox_probe_next_filter[&relay], 3);
|
||
|
||
discovery.install_mailbox_overlay(HashMap::new(), Instant::now());
|
||
assert!(!discovery.mailbox_probe_next_filter.contains_key(&relay));
|
||
assert!(!discovery.mailbox_probe_next_at.contains_key(&relay));
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_completion_for_removed_relay_leaves_no_cursor_state() {
|
||
let relay = "wss://mailbox.example".to_string();
|
||
let root = EventId::from_byte_array([33; 32]);
|
||
let now = Instant::now();
|
||
let mut discovery = Nip65DiscoveryState::default();
|
||
discovery
|
||
.mailbox_roots
|
||
.insert(relay.clone(), HashSet::from([root]));
|
||
|
||
discovery.record_probe_completion(&relay, 1, Duration::from_secs(60), now);
|
||
assert_eq!(discovery.mailbox_probe_next_filter[&relay], 1);
|
||
assert_eq!(
|
||
discovery.mailbox_probe_next_at[&relay],
|
||
now + Duration::from_secs(60)
|
||
);
|
||
|
||
discovery.install_mailbox_overlay(HashMap::new(), now);
|
||
discovery.record_probe_completion(&relay, 2, Duration::from_secs(60), now);
|
||
assert!(!discovery.mailbox_probe_next_filter.contains_key(&relay));
|
||
assert!(!discovery.mailbox_probe_next_at.contains_key(&relay));
|
||
}
|
||
|
||
#[test]
|
||
fn mailbox_completion_advances_after_failure_without_ending_cycle() {
|
||
let (next_filter, completed_cycle, retry_after) = mailbox_probe_completion(0, 2, false);
|
||
assert_eq!(next_filter, 1);
|
||
assert!(!completed_cycle);
|
||
assert_eq!(retry_after, mailbox_probe_retry_interval());
|
||
|
||
let (next_filter, completed_cycle, refresh_after) = mailbox_probe_completion(1, 2, true);
|
||
assert_eq!(next_filter, 0);
|
||
assert!(completed_cycle);
|
||
assert_eq!(refresh_after, mailbox_probe_refresh_interval());
|
||
}
|
||
|
||
#[test]
|
||
fn own_relay_targets_are_excluded_from_sync_actions() {
|
||
assert!(is_own_sync_target("wss://gitnostr.com", "gitnostr.com"));
|
||
assert!(is_own_sync_target("ws://gitnostr.com", "gitnostr.com"));
|
||
assert!(is_own_sync_target("ws://127.0.0.1:7334", "127.0.0.1:7334"));
|
||
|
||
assert!(!is_own_sync_target(
|
||
"wss://gitnostr.com.attacker.example",
|
||
"gitnostr.com"
|
||
));
|
||
assert!(!is_own_sync_target("ws://127.0.0.1:7335", "127.0.0.1:7334"));
|
||
}
|
||
|
||
#[test]
|
||
fn semantic_fallback_requires_material_first_pass_incompatibility() {
|
||
for (requested, received) in [(118, 115), (203, 197), (380, 301), (1818, 1420)] {
|
||
assert!(
|
||
!should_use_semantic_fallback(requested, received),
|
||
"{received}/{requested} is a residual, not a hydration incompatibility"
|
||
);
|
||
}
|
||
|
||
assert!(should_use_semantic_fallback(20, 0));
|
||
assert!(should_use_semantic_fallback(20, 2));
|
||
assert!(!should_use_semantic_fallback(20, 3));
|
||
assert!(!should_use_semantic_fallback(19, 0));
|
||
}
|
||
|
||
#[test]
|
||
fn descendant_rotation_advances_cursor_only_after_successful_eose() {
|
||
let member = EventId::from_byte_array([7; 32]);
|
||
let frontier = DescendantFrontier {
|
||
event_ids: HashSet::from([member]),
|
||
coordinates: HashSet::new(),
|
||
..DescendantFrontier::default()
|
||
};
|
||
let now = Timestamp::from_secs(200_000);
|
||
let mut rotation = DescendantSyncRotation::default();
|
||
|
||
rotation.refresh(
|
||
descendant_frontier_filters(&frontier, None),
|
||
MAX_FILTERS_PER_REQ,
|
||
);
|
||
let (filter_index, first, until) = rotation.next_request(now).unwrap();
|
||
assert!(first
|
||
.iter()
|
||
.all(|filter| { serde_json::to_value(filter).unwrap().get("since").is_none() }));
|
||
rotation.mark_started(41, filter_index, until);
|
||
assert!(rotation.next_request(now).is_none());
|
||
|
||
assert!(rotation.mark_completed(41, false));
|
||
let (retry_index, retry, retry_until) = rotation.next_request(now).unwrap();
|
||
assert_eq!(retry_index, filter_index);
|
||
assert!(retry
|
||
.iter()
|
||
.all(|filter| { serde_json::to_value(filter).unwrap().get("since").is_none() }));
|
||
rotation.mark_started(42, retry_index, retry_until);
|
||
assert!(rotation.mark_completed(42, true));
|
||
|
||
// All three e/E/q variants share one relay-compatible REQ, so the
|
||
// next turn returns to that group with its successful overlap cursor.
|
||
let (_, recent, _) = rotation
|
||
.next_request(Timestamp::from_secs(201_000))
|
||
.unwrap();
|
||
assert!(recent.iter().all(|filter| {
|
||
serde_json::to_value(filter).unwrap()["since"]
|
||
== serde_json::json!(now.as_secs() - DESCENDANT_FALLBACK_OVERLAP_SECS)
|
||
}));
|
||
}
|
||
|
||
#[test]
|
||
fn tier_cutoff_keeps_only_complete_priority_tiers_live() {
|
||
use filters::{CoverageTier, TieredFilter};
|
||
|
||
let entries = vec![
|
||
TieredFilter {
|
||
tier: CoverageTier::RootUppercase,
|
||
filter: Filter::new().kind(Kind::Custom(31_001)),
|
||
},
|
||
TieredFilter {
|
||
tier: CoverageTier::CoreCompatibility,
|
||
filter: Filter::new().kind(Kind::Custom(31_002)),
|
||
},
|
||
TieredFilter {
|
||
tier: CoverageTier::CoreCompatibility,
|
||
filter: Filter::new().kind(Kind::Custom(31_003)),
|
||
},
|
||
TieredFilter {
|
||
tier: CoverageTier::DescendantCanonical,
|
||
filter: Filter::new().kind(Kind::Custom(31_004)),
|
||
},
|
||
TieredFilter {
|
||
tier: CoverageTier::HistoricOnly,
|
||
filter: Filter::new().kind(Kind::Custom(31_005)),
|
||
},
|
||
];
|
||
|
||
let (live, rotated) = split_live_tier_prefix(&entries, 2, |groups| groups.len() <= 1);
|
||
assert_eq!(
|
||
live.len(),
|
||
1,
|
||
"the two-filter compatibility tier must not be split"
|
||
);
|
||
assert_eq!(rotated.len(), 4);
|
||
assert!(rotated.iter().any(|filter| {
|
||
filter.as_json() == Filter::new().kind(Kind::Custom(31_005)).as_json()
|
||
}));
|
||
|
||
let fallback = complete_auxiliary_fallback(&entries);
|
||
assert_eq!(fallback.len(), entries.len());
|
||
assert!(live.iter().all(|live_filter| fallback
|
||
.iter()
|
||
.any(|filter| filter.as_json() == live_filter.as_json())));
|
||
}
|
||
|
||
#[test]
|
||
fn historic_packing_unions_core_and_descendant_reference_values() {
|
||
let root = EventId::from_byte_array([1; 32]);
|
||
let descendant = EventId::from_byte_array([2; 32]);
|
||
let items = PendingItems {
|
||
repos: HashSet::from(["30617:owner:repo".to_string()]),
|
||
state_only_repos: HashSet::new(),
|
||
root_events: HashSet::from([root]),
|
||
};
|
||
let frontier = DescendantFrontier {
|
||
event_ids: HashSet::from([descendant]),
|
||
coordinates: HashSet::from(["1621:author:patch".to_string()]),
|
||
..DescendantFrontier::default()
|
||
};
|
||
|
||
let packed = packed_historic_filters(&items, &frontier);
|
||
// One state filter plus one a/A/q and one e/E/q family. Appending
|
||
// descendants separately would produce thirteen filters here.
|
||
assert_eq!(packed.len(), 7);
|
||
let json = packed
|
||
.iter()
|
||
.map(Filter::as_json)
|
||
.collect::<Vec<_>>()
|
||
.join("\n");
|
||
assert!(json.contains(&root.to_hex()));
|
||
assert!(json.contains(&descendant.to_hex()));
|
||
assert!(json.contains("30617:owner:repo"));
|
||
assert!(json.contains("1621:author:patch"));
|
||
}
|
||
|
||
#[test]
|
||
fn large_descendant_frontier_splits_across_live_subscriptions() {
|
||
let members: HashSet<EventId> = (0..600u32)
|
||
.map(|index| {
|
||
let mut bytes = [0u8; 32];
|
||
bytes[..4].copy_from_slice(&index.to_be_bytes());
|
||
EventId::from_byte_array(bytes)
|
||
})
|
||
.collect();
|
||
let filters = descendant_frontier_filters(
|
||
&DescendantFrontier {
|
||
event_ids: members,
|
||
coordinates: HashSet::new(),
|
||
..DescendantFrontier::default()
|
||
},
|
||
None,
|
||
);
|
||
let groups = live_filter_groups(&filters, MAX_FILTERS_PER_REQ);
|
||
|
||
assert!(filters.len() > 3, "the frontier must span byte chunks");
|
||
assert!(
|
||
groups.len() > 1,
|
||
"large complete descendant coverage must reserve several subscriptions"
|
||
);
|
||
assert!(groups
|
||
.iter()
|
||
.all(|group| group.len() <= MAX_FILTERS_PER_REQ));
|
||
}
|
||
|
||
#[test]
|
||
fn group_filters_for_req_respects_count_and_byte_budgets() {
|
||
// Many small filters group by the count cap.
|
||
let small: Vec<Filter> = (0..25)
|
||
.map(|i| Filter::new().id(EventId::from_byte_array([0; 32])).limit(i))
|
||
.collect();
|
||
let groups = group_filters_for_req(&small);
|
||
assert_eq!(
|
||
groups.iter().map(Vec::len).collect::<Vec<_>>(),
|
||
vec![MAX_FILTERS_PER_REQ, MAX_FILTERS_PER_REQ, 5]
|
||
);
|
||
|
||
// Large filters group by the byte budget instead: every group stays
|
||
// within it (single oversized filters excepted) and below the cap.
|
||
let ids: Vec<EventId> = (0..600u32)
|
||
.map(|i| {
|
||
let mut bytes = [0u8; 32];
|
||
bytes[..4].copy_from_slice(&i.to_be_bytes());
|
||
EventId::from_byte_array(bytes)
|
||
})
|
||
.collect();
|
||
let root_events: std::collections::HashSet<EventId> = ids.into_iter().collect();
|
||
let large = filters::tagged_one_of_our_root_event_filters(&root_events, None);
|
||
assert!(large.len() > 3, "600 IDs should span multiple chunks");
|
||
|
||
let groups = group_filters_for_req(&large);
|
||
assert!(groups.len() > 1);
|
||
for group in &groups {
|
||
assert!(group.len() <= MAX_FILTERS_PER_REQ);
|
||
let bytes: usize = group.iter().map(|f| f.as_json().len()).sum();
|
||
assert!(
|
||
group.len() == 1 || bytes <= REQ_MESSAGE_BYTE_BUDGET,
|
||
"group of {} filters serializes to {} bytes",
|
||
group.len(),
|
||
bytes
|
||
);
|
||
}
|
||
// Grouping preserves every filter exactly once.
|
||
assert_eq!(groups.iter().map(Vec::len).sum::<usize>(), large.len());
|
||
}
|
||
|
||
#[test]
|
||
fn purgatory_dependency_budget_prioritizes_a_fresh_announcement() {
|
||
let keys = Keys::generate();
|
||
let now = Instant::now();
|
||
let retry_after = Duration::from_secs(30);
|
||
let mut attempts = HashMap::new();
|
||
let mut events = Vec::new();
|
||
|
||
for created_at in 1..=1000 {
|
||
let event = EventBuilder::new(Kind::GitRepoAnnouncement, "")
|
||
.tag(Tag::identifier(format!("old-{created_at}")))
|
||
.custom_created_at(Timestamp::from_secs(created_at))
|
||
.finalize(&keys)
|
||
.expect("Failed to create old purgatory announcement");
|
||
attempts.insert(event.id, now);
|
||
events.push(event);
|
||
}
|
||
|
||
let fresh = EventBuilder::new(Kind::GitRepoAnnouncement, "")
|
||
.tag(Tag::identifier("fresh"))
|
||
.custom_created_at(Timestamp::from_secs(1001))
|
||
.finalize(&keys)
|
||
.expect("Failed to create fresh purgatory announcement");
|
||
events.push(fresh.clone());
|
||
|
||
let selected = select_purgatory_dependency_events(
|
||
events,
|
||
&mut attempts,
|
||
now,
|
||
retry_after,
|
||
MAX_PURGATORY_DEPENDENCY_EVENTS_PER_TICK,
|
||
);
|
||
|
||
assert_eq!(selected.len(), 1);
|
||
assert_eq!(selected[0].id, fresh.id);
|
||
assert!(
|
||
select_purgatory_dependency_events(
|
||
vec![fresh],
|
||
&mut attempts,
|
||
now,
|
||
retry_after,
|
||
MAX_PURGATORY_DEPENDENCY_EVENTS_PER_TICK,
|
||
)
|
||
.is_empty(),
|
||
"an attempted announcement must wait for its bounded retry deadline"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn connect_attempt_tokens_deduplicate_and_reject_stale_results() {
|
||
let relay = "wss://relay.example";
|
||
let mut in_flight = HashMap::new();
|
||
let mut next_token = 0;
|
||
|
||
let first = reserve_connect_attempt(&mut in_flight, &mut next_token, relay)
|
||
.expect("first attempt should be reserved");
|
||
assert!(
|
||
reserve_connect_attempt(&mut in_flight, &mut next_token, relay).is_none(),
|
||
"a queued or running relay must not be scheduled twice"
|
||
);
|
||
assert!(
|
||
!take_connect_attempt(&mut in_flight, relay, ConnectAttemptToken(first.0 + 1)),
|
||
"a stale worker result must not consume the active attempt"
|
||
);
|
||
assert_eq!(in_flight.get(relay), Some(&first));
|
||
assert!(take_connect_attempt(&mut in_flight, relay, first));
|
||
|
||
let second = reserve_connect_attempt(&mut in_flight, &mut next_token, relay)
|
||
.expect("relay should be schedulable after completion");
|
||
assert_ne!(first, second, "attempt identities must not be reused");
|
||
}
|
||
|
||
#[test]
|
||
fn lifecycle_notifications_apply_fixed_backpressure_under_peer_repetition() {
|
||
let (sender, mut receiver) = lifecycle_notification_channel();
|
||
|
||
for sequence in 0..LIFECYCLE_NOTIFICATION_CAPACITY {
|
||
sender
|
||
.try_send(sequence)
|
||
.expect("the documented lifecycle capacity should be usable");
|
||
}
|
||
assert!(
|
||
matches!(
|
||
sender.try_send(LIFECYCLE_NOTIFICATION_CAPACITY),
|
||
Err(tokio::sync::mpsc::error::TrySendError::Full(_))
|
||
),
|
||
"repeated peer terminals must backpressure instead of retaining unbounded memory"
|
||
);
|
||
|
||
assert_eq!(receiver.try_recv().unwrap(), 0);
|
||
sender
|
||
.try_send(LIFECYCLE_NOTIFICATION_CAPACITY)
|
||
.expect("draining one terminal must release exactly one queue slot");
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn connect_attempt_semaphore_caps_parallel_workers() {
|
||
use std::sync::atomic::{AtomicUsize, Ordering};
|
||
|
||
const WORKERS: usize = MAX_CONCURRENT_CONNECT_ATTEMPTS * 3;
|
||
let semaphore = Arc::new(Semaphore::new(MAX_CONCURRENT_CONNECT_ATTEMPTS));
|
||
let release = Arc::new(Semaphore::new(0));
|
||
let active = Arc::new(AtomicUsize::new(0));
|
||
let maximum = Arc::new(AtomicUsize::new(0));
|
||
let (started_tx, mut started_rx) = tokio::sync::mpsc::unbounded_channel();
|
||
let mut workers = Vec::new();
|
||
|
||
for _ in 0..WORKERS {
|
||
let semaphore = Arc::clone(&semaphore);
|
||
let release = Arc::clone(&release);
|
||
let active = Arc::clone(&active);
|
||
let maximum = Arc::clone(&maximum);
|
||
let started_tx = started_tx.clone();
|
||
workers.push(tokio::spawn(async move {
|
||
let _permit = semaphore.acquire_owned().await.unwrap();
|
||
let now_active = active.fetch_add(1, Ordering::SeqCst) + 1;
|
||
maximum.fetch_max(now_active, Ordering::SeqCst);
|
||
started_tx.send(()).unwrap();
|
||
let _release = release.acquire().await.unwrap();
|
||
active.fetch_sub(1, Ordering::SeqCst);
|
||
}));
|
||
}
|
||
drop(started_tx);
|
||
|
||
for _ in 0..MAX_CONCURRENT_CONNECT_ATTEMPTS {
|
||
started_rx.recv().await.unwrap();
|
||
}
|
||
assert!(
|
||
tokio::time::timeout(Duration::from_millis(25), started_rx.recv())
|
||
.await
|
||
.is_err(),
|
||
"a ninth worker started while all eight permits were occupied"
|
||
);
|
||
|
||
release.add_permits(WORKERS);
|
||
for worker in workers {
|
||
worker.await.unwrap();
|
||
}
|
||
assert_eq!(
|
||
maximum.load(Ordering::SeqCst),
|
||
MAX_CONCURRENT_CONNECT_ATTEMPTS
|
||
);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn queued_connect_attempt_records_health_only_when_worker_starts() {
|
||
let relay = "wss://queued.example";
|
||
let semaphore = Arc::new(Semaphore::new(0));
|
||
let health_tracker = Arc::new(RelayHealthTracker::with_defaults());
|
||
let worker_semaphore = Arc::clone(&semaphore);
|
||
let worker_health = Arc::clone(&health_tracker);
|
||
|
||
let worker = tokio::spawn(async move {
|
||
begin_connect_attempt(worker_semaphore, worker_health, relay).await
|
||
});
|
||
tokio::task::yield_now().await;
|
||
assert!(
|
||
health_tracker.get_health(relay).is_none(),
|
||
"waiting for scheduler capacity must not start backoff"
|
||
);
|
||
|
||
semaphore.add_permits(1);
|
||
let _permit = worker
|
||
.await
|
||
.unwrap()
|
||
.expect("worker should start after a permit becomes available");
|
||
assert!(
|
||
health_tracker
|
||
.get_health(relay)
|
||
.is_some_and(|health| health.last_attempt_time.is_some()),
|
||
"health attempt time must be recorded when the worker starts"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn canonical_relay_keys_dedupe_only_the_root_slash() {
|
||
assert_eq!(
|
||
canonical_relay_key("wss://relay.example").unwrap(),
|
||
"wss://relay.example"
|
||
);
|
||
assert_eq!(
|
||
canonical_relay_key("wss://relay.example/").unwrap(),
|
||
"wss://relay.example"
|
||
);
|
||
assert_eq!(
|
||
canonical_relay_key("wss://relay.example/nostr/").unwrap(),
|
||
"wss://relay.example/nostr/"
|
||
);
|
||
assert_eq!(
|
||
canonical_relay_key("relay.example/").unwrap(),
|
||
"wss://relay.example"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn connection_lookup_deduplicates_relay_url_variants() {
|
||
let canonical = canonical_relay_key("wss://relay.example").unwrap();
|
||
let connections = HashMap::from([(canonical.clone(), 7_u8)]);
|
||
let candidates = vec![
|
||
"wss://relay.example".to_string(),
|
||
"wss://relay.example/".to_string(),
|
||
];
|
||
|
||
assert_eq!(
|
||
connections_for_relay_urls(&connections, &candidates),
|
||
vec![(canonical, 7)]
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn grouped_pagination_advances_only_filters_that_fill_a_page() {
|
||
let keys = Keys::generate();
|
||
let metadata_filter = Filter::new().kind(Kind::Metadata);
|
||
let note_filter = Filter::new().kind(Kind::TextNote);
|
||
let mut pagination =
|
||
PaginationState::new(vec![metadata_filter.clone(), note_filter.clone()]);
|
||
|
||
for created_at in 1..=100 {
|
||
let event = EventBuilder::new(Kind::Metadata, created_at.to_string())
|
||
.custom_created_at(Timestamp::from_secs(created_at as u64))
|
||
.finalize(&keys)
|
||
.expect("build metadata event");
|
||
pagination.record_event(&event);
|
||
}
|
||
let note = EventBuilder::new(Kind::TextNote, "one note")
|
||
.custom_created_at(Timestamp::from_secs(100))
|
||
.finalize(&keys)
|
||
.expect("build text note");
|
||
pagination.record_event(¬e);
|
||
|
||
let next_filters = pagination
|
||
.next_page(&mut RelayPaginationSession::default())
|
||
.expect("full observed page should paginate")
|
||
.filters();
|
||
assert_eq!(next_filters.len(), 1);
|
||
assert_eq!(next_filters[0].until, Some(Timestamp::from_secs(1)));
|
||
assert!(
|
||
next_filters[0]
|
||
.kinds
|
||
.as_ref()
|
||
.unwrap()
|
||
.contains(&Kind::Metadata),
|
||
"the full metadata filter should advance"
|
||
);
|
||
assert!(
|
||
!next_filters[0]
|
||
.kinds
|
||
.as_ref()
|
||
.unwrap()
|
||
.contains(&Kind::TextNote),
|
||
"the partial text-note filter should not advance"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn raw_pagination_counts_purgatory_and_rejected_deliveries() {
|
||
let keys = Keys::generate();
|
||
let filter = Filter::new().kinds([Kind::GitRepoAnnouncement, Kind::RepoState]);
|
||
let mut pagination = PaginationState::new(vec![filter]);
|
||
let deliveries = [
|
||
(
|
||
EventBuilder::new(Kind::GitRepoAnnouncement, "purgatory")
|
||
.custom_created_at(Timestamp::from_secs(20))
|
||
.finalize(&keys)
|
||
.expect("build purgatory-routed event"),
|
||
ProcessResult::Purgatory,
|
||
),
|
||
(
|
||
EventBuilder::new(Kind::RepoState, "rejected")
|
||
.custom_created_at(Timestamp::from_secs(10))
|
||
.finalize(&keys)
|
||
.expect("build rejected event"),
|
||
ProcessResult::Rejected(PolicyRejection::Restricted),
|
||
),
|
||
];
|
||
|
||
for (event, policy_result) in deliveries {
|
||
// The production handler records here, before it knows this result.
|
||
pagination.record_event(&event);
|
||
assert!(matches!(
|
||
policy_result,
|
||
ProcessResult::Purgatory | ProcessResult::Rejected(_)
|
||
));
|
||
}
|
||
|
||
assert_eq!(pagination.filters[0].event_count, 2);
|
||
assert_eq!(
|
||
pagination.filters[0].min_created_at,
|
||
Some(Timestamp::from_secs(10))
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn raw_pagination_counts_repeat_deliveries_at_the_boundary() {
|
||
let keys = Keys::generate();
|
||
let filter = Filter::new().kind(Kind::TextNote);
|
||
let mut pagination = PaginationState::new(vec![filter]);
|
||
let event = EventBuilder::new(Kind::TextNote, "same relay delivery")
|
||
.custom_created_at(Timestamp::from_secs(42))
|
||
.finalize(&keys)
|
||
.expect("build repeated event");
|
||
|
||
for _ in 0..PAGINATION_THRESHOLD_FLOOR {
|
||
// Repeat deliveries each consume a result slot even though the second and later
|
||
// process as Duplicate after the raw-delivery accounting point.
|
||
pagination.record_event(&event);
|
||
}
|
||
|
||
assert_eq!(
|
||
pagination.filters[0].event_count,
|
||
PAGINATION_THRESHOLD_FLOOR
|
||
);
|
||
assert!(pagination
|
||
.next_page(&mut RelayPaginationSession::default())
|
||
.is_some());
|
||
}
|
||
|
||
#[test]
|
||
fn raw_pagination_counts_an_event_for_each_overlapping_filter() {
|
||
let keys = Keys::generate();
|
||
let event = EventBuilder::new(Kind::TextNote, "overlap")
|
||
.custom_created_at(Timestamp::from_secs(42))
|
||
.finalize(&keys)
|
||
.expect("build overlapping event");
|
||
let mut pagination = PaginationState::new(vec![
|
||
Filter::new().kind(Kind::TextNote),
|
||
Filter::new().author(keys.public_key()),
|
||
]);
|
||
|
||
pagination.record_event(&event);
|
||
|
||
assert_eq!(pagination.filters[0].event_count, 1);
|
||
assert_eq!(pagination.filters[1].event_count, 1);
|
||
assert_eq!(
|
||
pagination.filters[0].min_created_at,
|
||
pagination.filters[1].min_created_at
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn adaptive_threshold_selects_from_observation_and_hint() {
|
||
let mut observed_only = RelayPaginationSession::new(None);
|
||
observed_only.observe_page(100);
|
||
assert_eq!(observed_only.pagination_threshold(), 90);
|
||
|
||
let hint_only = RelayPaginationSession::new(Some(500));
|
||
assert_eq!(hint_only.pagination_threshold(), 450);
|
||
|
||
let mut hint_vs_observed = RelayPaginationSession::new(Some(500));
|
||
hint_vs_observed.observe_page(600);
|
||
assert_eq!(hint_vs_observed.pagination_threshold(), 540);
|
||
|
||
let mut sub_floor = RelayPaginationSession::new(Some(50));
|
||
sub_floor.observe_page(80);
|
||
assert_eq!(sub_floor.pagination_threshold(), 90);
|
||
}
|
||
|
||
#[test]
|
||
fn productive_verification_page_discards_the_hint() {
|
||
let keys = Keys::generate();
|
||
let mut first_page = PaginationState::new(vec![Filter::new().kind(Kind::TextNote)]);
|
||
for created_at in 100..200 {
|
||
let event = EventBuilder::new(Kind::TextNote, created_at.to_string())
|
||
.custom_created_at(Timestamp::from_secs(created_at))
|
||
.finalize(&keys)
|
||
.expect("build first-page event");
|
||
first_page.record_event(&event);
|
||
}
|
||
let mut session = RelayPaginationSession::new(Some(1000));
|
||
let mut verification = first_page
|
||
.next_page(&mut session)
|
||
.expect("a suspiciously short hinted page needs verification");
|
||
assert_eq!(session.hint, PaginationHint::Verifying(1000));
|
||
assert_eq!(
|
||
verification.request_class(),
|
||
TransientRequestClass::PaginationVerification
|
||
);
|
||
|
||
let unseen = EventBuilder::new(Kind::TextNote, "older unseen event")
|
||
.custom_created_at(Timestamp::from_secs(99))
|
||
.finalize(&keys)
|
||
.expect("build productive verification event");
|
||
for _ in 0..100 {
|
||
verification.record_event(&unseen);
|
||
}
|
||
|
||
assert!(verification.next_page(&mut session).is_some());
|
||
assert_eq!(session.hint, PaginationHint::Discarded);
|
||
assert_eq!(session.pagination_threshold(), 90);
|
||
}
|
||
|
||
#[test]
|
||
fn empty_verification_page_confirms_the_hint_without_another_page() {
|
||
let keys = Keys::generate();
|
||
let mut first_page = PaginationState::new(vec![Filter::new().kind(Kind::TextNote)]);
|
||
for created_at in 100..200 {
|
||
let event = EventBuilder::new(Kind::TextNote, created_at.to_string())
|
||
.custom_created_at(Timestamp::from_secs(created_at))
|
||
.finalize(&keys)
|
||
.expect("build first-page event");
|
||
first_page.record_event(&event);
|
||
}
|
||
let mut session = RelayPaginationSession::new(Some(1000));
|
||
let verification = first_page
|
||
.next_page(&mut session)
|
||
.expect("a suspiciously short hinted page needs verification");
|
||
|
||
assert!(verification.next_page(&mut session).is_none());
|
||
assert_eq!(session.hint, PaginationHint::Verified(1000));
|
||
assert_eq!(session.pagination_threshold(), 900);
|
||
}
|
||
|
||
#[test]
|
||
fn deferred_consolidation_runs_only_after_final_batch_completion() {
|
||
let relay_url = "wss://relay.example";
|
||
let mut deferred = DeferredConsolidations::default();
|
||
|
||
assert!(
|
||
!deferred.request(relay_url, true),
|
||
"an in-flight batch must defer instead of waiting in the sync actor"
|
||
);
|
||
assert!(
|
||
!deferred.take_ready(relay_url, true),
|
||
"consolidation must remain queued while another batch is pending"
|
||
);
|
||
|
||
assert!(
|
||
deferred.take_ready(relay_url, false),
|
||
"the final actor-owned batch completion must make consolidation runnable"
|
||
);
|
||
assert!(
|
||
!deferred.take_ready(relay_url, false),
|
||
"taking deferred work must be idempotent"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn reset_or_disconnect_cancels_deferred_consolidation_and_stale_wakeup() {
|
||
for reason in ["reset", "disconnect"] {
|
||
let relay_url = format!("wss://{reason}.example");
|
||
let mut deferred = DeferredConsolidations::default();
|
||
|
||
assert!(!deferred.request(&relay_url, true));
|
||
assert!(deferred.cancel(&relay_url));
|
||
assert!(
|
||
!deferred.take_ready(&relay_url, false),
|
||
"a queued wakeup must not resurrect consolidation after {reason}"
|
||
);
|
||
}
|
||
}
|
||
|
||
#[test]
|
||
fn failed_pagination_completes_only_a_fully_drained_batch() {
|
||
let relay_url = "wss://pagination.example";
|
||
let still_pending = SubscriptionId::new("still-pending");
|
||
let make_batch = |batch_id, outstanding_subs| PendingBatch {
|
||
batch_id,
|
||
purpose: PendingBatchPurpose::Core,
|
||
items: PendingItems::default(),
|
||
outstanding_subs,
|
||
sync_method: SyncMethod::ReqEose,
|
||
pagination_state: HashMap::new(),
|
||
requested_event_ids: None,
|
||
received_event_ids: None,
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
let mut pending = HashMap::from([(
|
||
relay_url.to_string(),
|
||
vec![
|
||
make_batch(41, HashSet::new()),
|
||
make_batch(42, HashSet::from([still_pending])),
|
||
],
|
||
)]);
|
||
|
||
let completed = take_drained_batch_as_failed(&mut pending, relay_url, 41)
|
||
.expect("failed pagination must complete a drained batch");
|
||
assert!(completed.failed);
|
||
assert!(take_drained_batch_as_failed(&mut pending, relay_url, 42).is_none());
|
||
assert_eq!(pending[relay_url].len(), 1);
|
||
assert!(!pending[relay_url][0].failed);
|
||
}
|
||
|
||
#[test]
|
||
fn rate_limit_classifier_covers_notice_and_closed_reason_forms() {
|
||
for message in [
|
||
"rate-limited: too many queries",
|
||
"Rate limit exceeded",
|
||
"slow down please",
|
||
"subscription throttled",
|
||
] {
|
||
assert!(is_rate_limit_message(message), "missed: {message}");
|
||
}
|
||
assert!(!is_rate_limit_message("blocked: unsupported filter"));
|
||
}
|
||
|
||
#[test]
|
||
fn policy_refusal_classifier_uses_bounded_categories() {
|
||
assert_eq!(
|
||
policy_refusal("auth-required: this relay only serves private notes"),
|
||
None,
|
||
"authentication gets one SDK-owned retry before it becomes a policy refusal"
|
||
);
|
||
assert_eq!(
|
||
policy_refusal("restricted: this relay does not accept REQs"),
|
||
Some(PolicyRefusal::Restricted)
|
||
);
|
||
assert_eq!(
|
||
policy_refusal("restricted: you are not a member of this relay"),
|
||
Some(PolicyRefusal::MembershipRequired)
|
||
);
|
||
assert_eq!(
|
||
policy_refusal("blocked: Request rejected"),
|
||
Some(PolicyRefusal::Blocked)
|
||
);
|
||
assert_eq!(
|
||
policy_refusal("ERROR: filter validation failed: invalid number of filters: 8"),
|
||
Some(PolicyRefusal::FilterIncompatible)
|
||
);
|
||
assert_eq!(
|
||
policy_refusal("rate-limited: REQ exceeds max filter count 3"),
|
||
Some(PolicyRefusal::FilterIncompatible)
|
||
);
|
||
assert_eq!(policy_refusal("rate-limited: too many queries"), None);
|
||
assert_eq!(policy_refusal("error: temporary backend failure"), None);
|
||
}
|
||
|
||
#[test]
|
||
fn authentication_gets_one_same_subscription_retry() {
|
||
let relay = "wss://private.example";
|
||
let subscription_id = SubscriptionId::new("auth-retry");
|
||
let mut attempts = HashSet::new();
|
||
|
||
assert!(reserve_authentication_retry(
|
||
&mut attempts,
|
||
relay,
|
||
&subscription_id
|
||
));
|
||
assert!(!reserve_authentication_retry(
|
||
&mut attempts,
|
||
relay,
|
||
&subscription_id
|
||
));
|
||
assert!(
|
||
attempts.is_empty(),
|
||
"a refused retry must retire its marker"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn subscription_state_limit_parser_is_specific_and_extracts_bytes() {
|
||
assert_eq!(
|
||
subscription_state_byte_limit(
|
||
"rate-limited: active subscriptions exceed max size 1048576 bytes"
|
||
),
|
||
Some(1_048_576)
|
||
);
|
||
assert_eq!(
|
||
subscription_state_byte_limit("rate-limited: too many queries"),
|
||
None
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn learned_subscription_byte_limit_reserves_transient_capacity() {
|
||
let first = vec![Filter::new().kind(Kind::TextNote).limit(0)];
|
||
let second = vec![Filter::new().kind(Kind::Metadata).limit(0)];
|
||
let limit = SUBSCRIPTION_BYTE_RESERVED_MARGIN + req_message_size(&first);
|
||
|
||
let (admitted, overflow) =
|
||
groups_within_subscription_byte_limit(vec![first.clone(), second], Some(limit), 0);
|
||
|
||
assert_eq!(admitted, vec![first]);
|
||
assert_eq!(overflow, 1);
|
||
}
|
||
|
||
#[test]
|
||
fn rate_limited_closed_removes_only_its_pending_batch_for_retry() {
|
||
let relay_url = "wss://limited.example";
|
||
let rejected = SubscriptionId::new("rejected");
|
||
let accepted = SubscriptionId::new("accepted");
|
||
let make_batch = |batch_id, subscription_id| PendingBatch {
|
||
batch_id,
|
||
purpose: PendingBatchPurpose::Core,
|
||
items: PendingItems::default(),
|
||
outstanding_subs: HashSet::from([subscription_id]),
|
||
sync_method: SyncMethod::ReqEose,
|
||
pagination_state: HashMap::new(),
|
||
requested_event_ids: None,
|
||
received_event_ids: None,
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
let mut pending = HashMap::from([(
|
||
relay_url.to_string(),
|
||
vec![
|
||
make_batch(1, rejected.clone()),
|
||
make_batch(2, accepted.clone()),
|
||
],
|
||
)]);
|
||
|
||
let removed = take_batch_containing_subscription(&mut pending, relay_url, &rejected)
|
||
.expect("rejected subscription must release its batch for cooldown retry");
|
||
assert_eq!(removed.batch_id, 1);
|
||
assert_eq!(pending[relay_url].len(), 1);
|
||
assert!(pending[relay_url][0].outstanding_subs.contains(&accepted));
|
||
|
||
take_batch_containing_subscription(&mut pending, relay_url, &accepted)
|
||
.expect("last batch must also be removable");
|
||
assert!(!pending.contains_key(relay_url));
|
||
}
|
||
|
||
#[test]
|
||
fn rate_limited_pagination_keeps_batch_pending_without_actor_wait() {
|
||
let relay_url = "wss://pagination.example";
|
||
let completed_sub = SubscriptionId::new("completed-page");
|
||
let mut batch = PendingBatch {
|
||
batch_id: 73,
|
||
purpose: PendingBatchPurpose::Core,
|
||
items: PendingItems::default(),
|
||
outstanding_subs: HashSet::new(),
|
||
sync_method: SyncMethod::ReqEose,
|
||
pagination_state: HashMap::new(),
|
||
requested_event_ids: None,
|
||
received_event_ids: None,
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
|
||
let deferred = mark_deferred_pagination(&mut batch, &completed_sub);
|
||
assert!(batch.outstanding_subs.contains(&deferred));
|
||
|
||
let mut pending = HashMap::from([(relay_url.to_string(), vec![batch])]);
|
||
assert!(
|
||
take_drained_batch_as_failed(&mut pending, relay_url, 73).is_none(),
|
||
"the deferred exact page must prevent a generic historic batch from being confirmed early"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn relay_disconnect_waits_for_pending_and_historic_sync_work() {
|
||
let mut source = RelayState {
|
||
connection_status: ConnectionStatus::Syncing,
|
||
..RelayState::default()
|
||
};
|
||
|
||
assert!(
|
||
!source.is_disconnect_candidate(true, false),
|
||
"a source with missing-ID subscriptions in flight must stay connected"
|
||
);
|
||
assert!(
|
||
!source.is_disconnect_candidate(false, false),
|
||
"the historic-sync batch window must stay connected even between batches"
|
||
);
|
||
|
||
source.connection_status = ConnectionStatus::Connected;
|
||
assert!(
|
||
!source.is_disconnect_candidate(false, false),
|
||
"an active relay cannot be disconnected before historic sync settles"
|
||
);
|
||
|
||
source.historic_sync_completed = true;
|
||
assert!(
|
||
source.is_disconnect_candidate(false, false),
|
||
"an empty relay can be released after historic sync settles"
|
||
);
|
||
assert!(
|
||
!source.is_disconnect_candidate(true, false),
|
||
"a later pending batch still keeps an otherwise empty relay connected"
|
||
);
|
||
assert!(
|
||
!source.is_disconnect_candidate(false, true),
|
||
"unconfirmed desired invitation work keeps its source connected"
|
||
);
|
||
|
||
source
|
||
.repos
|
||
.insert("30617:maintainer:repository".to_string());
|
||
assert!(
|
||
!source.is_disconnect_candidate(false, false),
|
||
"confirmed repository work keeps the source connected"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn disconnected_empty_relay_can_still_be_cleaned_up() {
|
||
let mut source = RelayState::default();
|
||
|
||
assert!(source.is_disconnect_candidate(false, false));
|
||
assert!(!source.is_disconnect_candidate(false, true));
|
||
|
||
source.historic_sync_completed = true;
|
||
source.connection_status = ConnectionStatus::Connecting;
|
||
assert!(
|
||
!source.is_disconnect_candidate(false, false),
|
||
"cleanup must not race an in-flight connection worker"
|
||
);
|
||
}
|
||
|
||
#[test]
|
||
fn purgatory_snapshot_replaces_state_only_relays_and_prunes_full_expiry() {
|
||
let active = "30617:owner:active".to_string();
|
||
let expired = "30617:owner:expired".to_string();
|
||
let promoted = "30617:owner:promoted".to_string();
|
||
let old = "wss://old.example".to_string();
|
||
let new = "wss://new.example".to_string();
|
||
let expired_relay = "wss://expired.example".to_string();
|
||
let promoted_relay = "wss://promoted.example".to_string();
|
||
let mut index = HashMap::from([
|
||
(
|
||
active.clone(),
|
||
RepoSyncNeeds {
|
||
relays: HashSet::from([old.clone()]),
|
||
sync_level: SyncLevel::StateOnly,
|
||
..Default::default()
|
||
},
|
||
),
|
||
(
|
||
expired.clone(),
|
||
RepoSyncNeeds {
|
||
relays: HashSet::from([expired_relay.clone()]),
|
||
sync_level: SyncLevel::StateOnly,
|
||
..Default::default()
|
||
},
|
||
),
|
||
(
|
||
promoted.clone(),
|
||
RepoSyncNeeds {
|
||
relays: HashSet::from([promoted_relay.clone()]),
|
||
sync_level: SyncLevel::Full,
|
||
..Default::default()
|
||
},
|
||
),
|
||
]);
|
||
|
||
let dirty = reconcile_purgatory_relay_ownership(
|
||
&mut index,
|
||
&[(active.clone(), HashSet::from([new.clone()]))],
|
||
);
|
||
|
||
assert_eq!(index[&active].relays, HashSet::from([new.clone()]));
|
||
assert!(!index.contains_key(&expired));
|
||
assert_eq!(index[&promoted].relays, HashSet::from([promoted_relay]));
|
||
assert_eq!(dirty, HashSet::from([old, new, expired_relay]));
|
||
}
|
||
|
||
#[test]
|
||
fn soft_expired_purgatory_entry_remains_while_present_in_snapshot() {
|
||
let repo = "30617:owner:revivable".to_string();
|
||
let relay = "wss://owner.example".to_string();
|
||
let mut index = HashMap::from([(
|
||
repo.clone(),
|
||
RepoSyncNeeds {
|
||
relays: HashSet::from([relay.clone()]),
|
||
sync_level: SyncLevel::StateOnly,
|
||
..Default::default()
|
||
},
|
||
)]);
|
||
|
||
let dirty = reconcile_purgatory_relay_ownership(
|
||
&mut index,
|
||
&[(repo.clone(), HashSet::from([relay]))],
|
||
);
|
||
|
||
assert!(index.contains_key(&repo));
|
||
assert!(dirty.is_empty());
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_rejected_events_index_tracks_announcements() {
|
||
// Create a rejected events index with 2 minute hot cache, 7 day cold index
|
||
let rejected_index = Arc::new(RejectedEventsIndex::new(
|
||
Duration::from_secs(120),
|
||
Duration::from_secs(604800),
|
||
));
|
||
|
||
// Create test announcement event (kind 30617) with 'd' tag
|
||
let keys = Keys::generate();
|
||
let announcement = EventBuilder::new(Kind::GitRepoAnnouncement, "test content")
|
||
.tag(nostr_sdk::prelude::Tag::custom("d", vec!["test-repo"]))
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
|
||
// Verify index is empty
|
||
assert_eq!(rejected_index.hot_cache_len(), 0);
|
||
assert_eq!(rejected_index.cold_index_len(), 0);
|
||
|
||
// Simulate rejection by adding to index
|
||
rejected_index.add_announcement(
|
||
announcement.clone(),
|
||
announcement.pubkey,
|
||
"test-repo".to_string(),
|
||
rejected_index::RejectionReason::DoesNotListService,
|
||
);
|
||
|
||
// Verify event is tracked in both tiers
|
||
assert!(rejected_index.contains(&announcement.id));
|
||
assert_eq!(rejected_index.hot_cache_len(), 1);
|
||
assert_eq!(rejected_index.cold_index_len(), 1);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_rejected_events_excluded_from_negentropy() {
|
||
// Create indices
|
||
let purgatory_ids: HashSet<EventId> = HashSet::new();
|
||
let rejected_index =
|
||
RejectedEventsIndex::new(Duration::from_secs(120), Duration::from_secs(604800));
|
||
|
||
// Create test event IDs
|
||
let _rejected_id =
|
||
EventId::from_hex("0000000000000000000000000000000000000000000000000000000000000001")
|
||
.unwrap();
|
||
let valid_id =
|
||
EventId::from_hex("0000000000000000000000000000000000000000000000000000000000000002")
|
||
.unwrap();
|
||
|
||
// Add rejected event to index
|
||
let keys = Keys::generate();
|
||
let rejected_event = EventBuilder::new(Kind::GitRepoAnnouncement, "rejected")
|
||
.tag(nostr_sdk::prelude::Tag::custom("d", vec!["rejected-repo"]))
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
|
||
// Override the event ID for testing (we need a specific ID)
|
||
// Since we can't override the ID, let's use the actual event ID
|
||
let rejected_id = rejected_event.id;
|
||
rejected_index.add_announcement(
|
||
rejected_event,
|
||
keys.public_key(),
|
||
"rejected-repo".to_string(),
|
||
rejected_index::RejectionReason::DoesNotListService,
|
||
);
|
||
|
||
// Get rejected IDs from index
|
||
let rejected_ids = rejected_index.get_all_event_ids();
|
||
|
||
// Simulate negentropy reconciliation result
|
||
let mut remote_ids = HashSet::new();
|
||
remote_ids.insert(rejected_id);
|
||
remote_ids.insert(valid_id);
|
||
|
||
// Exclude rejected and purgatory events
|
||
let excluded_ids: HashSet<EventId> = purgatory_ids.union(&rejected_ids).cloned().collect();
|
||
let filtered_ids: HashSet<EventId> =
|
||
remote_ids.difference(&excluded_ids).cloned().collect();
|
||
|
||
// Verify rejected event is excluded
|
||
assert!(!filtered_ids.contains(&rejected_id));
|
||
assert!(filtered_ids.contains(&valid_id));
|
||
assert_eq!(filtered_ids.len(), 1);
|
||
}
|
||
|
||
#[tokio::test]
|
||
async fn test_missing_d_tag_event_excluded_from_refetch_after_rejection() {
|
||
let purgatory_ids: HashSet<EventId> = HashSet::new();
|
||
let rejected_index =
|
||
RejectedEventsIndex::new(Duration::from_secs(120), Duration::from_secs(604800));
|
||
|
||
// Announcement without a 'd' tag: structurally malformed, no
|
||
// repository identifier exists to key the two-tier index with
|
||
let keys = Keys::generate();
|
||
let malformed = EventBuilder::new(Kind::GitRepoAnnouncement, "no d tag")
|
||
.finalize(&keys)
|
||
.unwrap();
|
||
assert!(!malformed.tags.iter().any(|t| t.kind() == "d"));
|
||
|
||
// Rejection path tracks it by exact ID
|
||
rejected_index.add_unrecoverable(malformed.id, malformed.kind.as_u16());
|
||
|
||
// Live/REQ path consults contains() before processing
|
||
assert!(rejected_index.contains(&malformed.id));
|
||
|
||
// Historic negentropy path excludes it from re-download
|
||
let rejected_ids = rejected_index.get_all_event_ids();
|
||
let excluded_ids: HashSet<EventId> = purgatory_ids.union(&rejected_ids).cloned().collect();
|
||
assert!(excluded_ids.contains(&malformed.id));
|
||
}
|
||
|
||
#[test]
|
||
fn test_negentropy_missing_event_detection() {
|
||
// Simulate scenario where relay returns fewer events than requested
|
||
// This tests the core logic for detecting missing events
|
||
|
||
// Requested 5 events from negentropy diff
|
||
let mut requested: HashSet<EventId> = HashSet::new();
|
||
for i in 1u8..=5 {
|
||
let id = EventId::from_hex(&format!("{:0>64}", format!("{:x}", i))).unwrap();
|
||
requested.insert(id);
|
||
}
|
||
|
||
// Only received 3 events (simulating relay limit)
|
||
let mut received: HashSet<EventId> = HashSet::new();
|
||
for i in 1u8..=3 {
|
||
let id = EventId::from_hex(&format!("{:0>64}", format!("{:x}", i))).unwrap();
|
||
received.insert(id);
|
||
}
|
||
|
||
// Calculate missing events
|
||
let missing: Vec<EventId> = requested.difference(&received).cloned().collect();
|
||
|
||
// Should have 2 missing events (IDs 4 and 5)
|
||
assert_eq!(missing.len(), 2);
|
||
assert_eq!(requested.len(), 5);
|
||
assert_eq!(received.len(), 3);
|
||
|
||
// Verify the specific missing IDs
|
||
let id_4 = EventId::from_hex(&format!("{:0>64}", format!("{:x}", 4u8))).unwrap();
|
||
let id_5 = EventId::from_hex(&format!("{:0>64}", format!("{:x}", 5u8))).unwrap();
|
||
assert!(missing.contains(&id_4));
|
||
assert!(missing.contains(&id_5));
|
||
}
|
||
|
||
#[test]
|
||
fn test_negentropy_all_events_received() {
|
||
// Simulate scenario where all requested events are received
|
||
let mut requested: HashSet<EventId> = HashSet::new();
|
||
for i in 1u8..=3 {
|
||
let id = EventId::from_hex(&format!("{:0>64}", format!("{:x}", i))).unwrap();
|
||
requested.insert(id);
|
||
}
|
||
|
||
// Received all 3 events
|
||
let received = requested.clone();
|
||
|
||
// Calculate missing events
|
||
let missing: Vec<EventId> = requested.difference(&received).cloned().collect();
|
||
|
||
// Should have no missing events
|
||
assert!(missing.is_empty());
|
||
}
|
||
|
||
#[test]
|
||
fn test_pending_batch_negentropy_fields() {
|
||
// Test that PendingBatch properly tracks negentropy-specific fields
|
||
let batch = PendingBatch {
|
||
batch_id: 1,
|
||
purpose: PendingBatchPurpose::Core,
|
||
items: PendingItems::default(),
|
||
outstanding_subs: HashSet::new(),
|
||
sync_method: SyncMethod::Negentropy,
|
||
pagination_state: HashMap::new(),
|
||
requested_event_ids: Some(HashSet::new()),
|
||
received_event_ids: Some(HashSet::new()),
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
|
||
assert!(batch.requested_event_ids.is_some());
|
||
assert!(batch.received_event_ids.is_some());
|
||
assert_eq!(batch.sync_method, SyncMethod::Negentropy);
|
||
assert_eq!(batch.retry_count, 0);
|
||
assert!(!batch.failed);
|
||
}
|
||
|
||
#[test]
|
||
fn hydration_attempt_registers_every_chunk_before_immediate_delivery() {
|
||
let mut batch = PendingBatch {
|
||
batch_id: 7,
|
||
purpose: PendingBatchPurpose::Core,
|
||
items: PendingItems::default(),
|
||
outstanding_subs: HashSet::new(),
|
||
sync_method: SyncMethod::Negentropy,
|
||
pagination_state: HashMap::new(),
|
||
requested_event_ids: None,
|
||
received_event_ids: None,
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
let subscription_ids: Vec<_> = (0..169)
|
||
.map(|index| SubscriptionId::new(format!("hydration-{index}")))
|
||
.collect();
|
||
let event_ids = [
|
||
EventId::from_byte_array([1; 32]),
|
||
EventId::from_byte_array([2; 32]),
|
||
];
|
||
|
||
register_negentropy_hydration_attempt(
|
||
&mut batch,
|
||
subscription_ids.iter().cloned(),
|
||
event_ids,
|
||
false,
|
||
);
|
||
|
||
assert_eq!(batch.outstanding_subs.len(), 169);
|
||
assert_eq!(batch.requested_event_ids.as_ref().unwrap().len(), 2);
|
||
assert_eq!(batch.received_event_ids, Some(HashSet::new()));
|
||
for event_id in event_ids {
|
||
if batch
|
||
.requested_event_ids
|
||
.as_ref()
|
||
.unwrap()
|
||
.contains(&event_id)
|
||
{
|
||
batch.received_event_ids.as_mut().unwrap().insert(event_id);
|
||
}
|
||
}
|
||
assert_eq!(batch.received_event_ids.as_ref().unwrap().len(), 2);
|
||
for subscription_id in &subscription_ids {
|
||
assert!(
|
||
batch.outstanding_subs.remove(subscription_id),
|
||
"an immediate terminal signal must find its pre-registered chunk"
|
||
);
|
||
}
|
||
assert!(batch.outstanding_subs.is_empty());
|
||
}
|
||
|
||
#[test]
|
||
fn test_pending_batch_req_eose_fields() {
|
||
// Test that REQ+EOSE batches don't use negentropy fields
|
||
let batch = PendingBatch {
|
||
batch_id: 1,
|
||
purpose: PendingBatchPurpose::Core,
|
||
items: PendingItems::default(),
|
||
outstanding_subs: HashSet::new(),
|
||
sync_method: SyncMethod::ReqEose,
|
||
pagination_state: HashMap::new(),
|
||
requested_event_ids: None,
|
||
received_event_ids: None,
|
||
initial_hydration_counts: None,
|
||
retry_count: 0,
|
||
failed: false,
|
||
};
|
||
|
||
assert!(batch.requested_event_ids.is_none());
|
||
assert!(batch.received_event_ids.is_none());
|
||
assert_eq!(batch.sync_method, SyncMethod::ReqEose);
|
||
assert!(!batch.failed);
|
||
}
|
||
}
|