diff --git a/CHANGELOG.md b/CHANGELOG.md index d78a90e..5da238c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,6 +9,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Fixed +- Honor policy and request-failure pauses when reconnecting, so refused peers + cannot bypass transport backoff through their policy-limited health state. + - Back off repeated rate-limit refusals and incomplete requests exponentially, while keeping live connections and unrelated relays available. diff --git a/docs/explanation/sync-scaling-constraints.md b/docs/explanation/sync-scaling-constraints.md index f383858..0a27344 100644 --- a/docs/explanation/sync-scaling-constraints.md +++ b/docs/explanation/sync-scaling-constraints.md @@ -526,6 +526,9 @@ So concurrency is not a free scaling axis; it is the residual of the ledger: Overlapping rate, policy and request pauses must all expire before recovery; required live repair is remembered across overlapping pauses. Timeout-only recovery does not rebuild successful live subscriptions. + If the transport disconnects, dialing also waits for every active policy, + rate, or request pause before applying the ordinary transport backoff. A + policy-limited health label must not override an outstanding dial deadline. - Transient REQ+EOSE subscriptions — historic sync groups, fallback filters, exact-ID fetches, retries, and pagination pages — retain a five-request class cap inside the shared ledger: a slot is acquired when diff --git a/src/sync/health.rs b/src/sync/health.rs index c16819e..5fa7db2 100644 --- a/src/sync/health.rs +++ b/src/sync/health.rs @@ -690,16 +690,19 @@ impl RelayHealthTracker { Some(entry) => { let health = entry.value(); - // Don't reconnect if currently rate-limited - if health.is_rate_limited_now() { + // A new socket cannot repair a refused subscription policy or + // a silent request. Preserve every active pause across dials. + if health.is_rate_limited_now() + || health.is_policy_limited_now() + || health.is_request_paused_now() + { return false; } // Check state-based logic match health.state() { - HealthState::Healthy - | HealthState::Disconnected - | HealthState::PolicyLimited => true, + HealthState::Healthy | HealthState::Disconnected => true, + HealthState::PolicyLimited => false, HealthState::Degraded | HealthState::Dead | HealthState::RateLimited => { // Check if backoff/cooldown period has elapsed match health.next_retry_at { @@ -1149,6 +1152,30 @@ mod tests { assert!(health.connected); } + #[test] + fn policy_refused_disconnect_does_not_bypass_transport_backoff() { + let tracker = RelayHealthTracker::with_defaults(); + let relay = "wss://refusing.example"; + tracker.record_success(relay); + tracker.record_policy_refusal(relay); + tracker.record_failure(relay); + assert_eq!(tracker.get_state(relay), HealthState::PolicyLimited); + assert!(!tracker.should_attempt_connection(relay)); + tracker.health.get_mut(relay).unwrap().policy_retry_at = Some(Instant::now()); + tracker.exit_expired_policy_refusals(); + assert!( + !tracker.should_attempt_connection(relay), + "transport backoff still applies" + ); + tracker.health.get_mut(relay).unwrap().next_retry_at = Some(Instant::now()); + assert!(tracker.should_attempt_connection(relay)); + tracker.record_request_failure(relay); + assert!( + !tracker.should_attempt_connection(relay), + "silent-peer pause survives disconnect" + ); + } + #[test] fn policy_refusal_keeps_connection_and_allows_one_daily_probe() { let tracker = RelayHealthTracker::with_defaults();