From dbdb6edf4a98b7de3f0eebfd4ba58090bf8eaf21 Mon Sep 17 00:00:00 2001 From: DanConwayDev Date: Thu, 24 Sep 2026 06:45:31 +0000 Subject: [PATCH] fix(sync): honor subscription pauses before reconnecting refused peers PolicyLimited returned reconnect-allowed unconditionally, hiding an active transport failure deadline. Production refused peers consequently redialed within two seconds of disconnect while their failure streak was increasing. Reject dial admission during active policy, rate, or request-failure pauses. After those expire, continue enforcing the existing transport deadline. Established live connections remain untouched and SDK auto-reconnect remains disabled, so the application is the reconnect admission authority. This assumes a new socket cannot repair a subscription refusal during its pause. It deliberately does not ban the peer or modify remote state. Validation: 932 library tests and 114 sync integration tests passed (one ignored), formatting and strict all-target Clippy passed. The new regression checks policy refusal plus disconnect, policy expiry with transport backoff still active, eventual admission, and a separate silent-request pause. The integration validation includes a separate follow-up fixture correction: its last-update log search missed already installed coverage for another relay. Assisted-by: GPT-6 --- CHANGELOG.md | 3 ++ docs/explanation/sync-scaling-constraints.md | 3 ++ src/sync/health.rs | 37 +++++++++++++++++--- 3 files changed, 38 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d78a90e..5da238c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- Honor policy and request-failure pauses when reconnecting, so refused peers + cannot bypass transport backoff through their policy-limited health state. + - Back off repeated rate-limit refusals and incomplete requests exponentially, while keeping live connections and unrelated relays available. diff --git a/docs/explanation/sync-scaling-constraints.md b/docs/explanation/sync-scaling-constraints.md index f383858..0a27344 100644 --- a/docs/explanation/sync-scaling-constraints.md +++ b/docs/explanation/sync-scaling-constraints.md @@ -526,6 +526,9 @@ So concurrency is not a free scaling axis; it is the residual of the ledger: Overlapping rate, policy and request pauses must all expire before recovery; required live repair is remembered across overlapping pauses. Timeout-only recovery does not rebuild successful live subscriptions. + If the transport disconnects, dialing also waits for every active policy, + rate, or request pause before applying the ordinary transport backoff. A + policy-limited health label must not override an outstanding dial deadline. - Transient REQ+EOSE subscriptions — historic sync groups, fallback filters, exact-ID fetches, retries, and pagination pages — retain a five-request class cap inside the shared ledger: a slot is acquired when diff --git a/src/sync/health.rs b/src/sync/health.rs index c16819e..5fa7db2 100644 --- a/src/sync/health.rs +++ b/src/sync/health.rs @@ -690,16 +690,19 @@ impl RelayHealthTracker { Some(entry) => { let health = entry.value(); - // Don't reconnect if currently rate-limited - if health.is_rate_limited_now() { + // A new socket cannot repair a refused subscription policy or + // a silent request. Preserve every active pause across dials. + if health.is_rate_limited_now() + || health.is_policy_limited_now() + || health.is_request_paused_now() + { return false; } // Check state-based logic match health.state() { - HealthState::Healthy - | HealthState::Disconnected - | HealthState::PolicyLimited => true, + HealthState::Healthy | HealthState::Disconnected => true, + HealthState::PolicyLimited => false, HealthState::Degraded | HealthState::Dead | HealthState::RateLimited => { // Check if backoff/cooldown period has elapsed match health.next_retry_at { @@ -1149,6 +1152,30 @@ mod tests { assert!(health.connected); } + #[test] + fn policy_refused_disconnect_does_not_bypass_transport_backoff() { + let tracker = RelayHealthTracker::with_defaults(); + let relay = "wss://refusing.example"; + tracker.record_success(relay); + tracker.record_policy_refusal(relay); + tracker.record_failure(relay); + assert_eq!(tracker.get_state(relay), HealthState::PolicyLimited); + assert!(!tracker.should_attempt_connection(relay)); + tracker.health.get_mut(relay).unwrap().policy_retry_at = Some(Instant::now()); + tracker.exit_expired_policy_refusals(); + assert!( + !tracker.should_attempt_connection(relay), + "transport backoff still applies" + ); + tracker.health.get_mut(relay).unwrap().next_retry_at = Some(Instant::now()); + assert!(tracker.should_attempt_connection(relay)); + tracker.record_request_failure(relay); + assert!( + !tracker.should_attempt_connection(relay), + "silent-peer pause survives disconnect" + ); + } + #[test] fn policy_refusal_keeps_connection_and_allows_one_daily_probe() { let tracker = RelayHealthTracker::with_defaults();