diff --git a/grasp-audit/src/client.rs b/grasp-audit/src/client.rs index d841317..9e10c75 100644 --- a/grasp-audit/src/client.rs +++ b/grasp-audit/src/client.rs @@ -71,7 +71,10 @@ impl AuditClient { client.add_relay(relay_url).await?; client.connect().await; - // Wait for connection to establish (with retries) + // Wait up to ~5s for nostr-sdk's background relay task to finish the + // WebSocket handshake. Full audit suites can briefly starve the task + // scheduler under load, so the older ~2s window produced flakes even + // when the relay was already accepting connections. let mut attempts = 0; let mut connected = false; while attempts < 50 { @@ -141,7 +144,10 @@ impl AuditClient { client.add_relay(relay_url).await?; client.connect().await; - // Wait for connection to establish (with retries) + // Wait up to ~5s for nostr-sdk's background relay task to finish the + // WebSocket handshake. Full audit suites can briefly starve the task + // scheduler under load, so the older ~2s window produced flakes even + // when the relay was already accepting connections. let mut attempts = 0; let mut connected = false; while attempts < 50 { diff --git a/tests/common/relay.rs b/tests/common/relay.rs index 2f14d9c..a3cde6a 100644 --- a/tests/common/relay.rs +++ b/tests/common/relay.rs @@ -24,7 +24,7 @@ use crate::common::port::{self, PortReservation}; /// How long to wait for the spawned ngit-grasp subprocess to handle HTTP /// requests before giving up on a single attempt. const READY_TIMEOUT: Duration = Duration::from_secs(5); -/// How often to retry the TCP probe while waiting for readiness. +/// How often to retry the HTTP probe while waiting for readiness. const READY_POLL: Duration = Duration::from_millis(100); /// Extra grace after the HTTP service responds before declaring the relay /// ready. @@ -537,17 +537,18 @@ impl TestRelay { &self.relay_data_path } - /// Probe the listener with async TCP connects until it accepts, - /// while concurrently watching for the subprocess to exit early - /// (the signature of a lost port-bind race). Without the early-exit - /// check we'd burn the full [`READY_TIMEOUT`] on a process that's - /// already dead. + /// Probe the relay with a real HTTP request until it responds 200, + /// while concurrently watching for the subprocess to exit early (the + /// signature of a lost port-bind race). A raw TCP connect can succeed + /// before Hyper has installed the service that accepts test clients, + /// so HTTP readiness is the point where WebSocket clients may safely + /// begin racing the relay. async fn wait_for_ready_or_early_exit(&mut self) -> ReadyOutcome { let deadline = Instant::now() + READY_TIMEOUT; loop { - // Check whether the subprocess has already exited. If so the - // TCP probe will never succeed — bail immediately so the - // caller can retry with a fresh port. + // Check whether the subprocess has already exited. If so no + // readiness probe can succeed — bail immediately so the caller + // can retry with a fresh port. match self.process.try_wait() { Ok(Some(status)) => return ReadyOutcome::EarlyExit { status }, Ok(None) => { /* still running */ } @@ -579,6 +580,9 @@ impl TestRelay { async fn probe_http_ready(&self) -> std::io::Result<()> { let probe = async { let mut stream = tokio::net::TcpStream::connect(("127.0.0.1", self.port)).await?; + // Use the relay's HTTP handler rather than a bare TCP connect: + // this verifies the accept loop and Hyper service are both live, + // which is what the immediately-following WebSocket tests need. let request = format!( "GET / HTTP/1.1\r\nHost: 127.0.0.1:{}\r\nConnection: close\r\n\r\n", self.port diff --git a/tests/replaceable_history.rs b/tests/replaceable_history.rs index bebe755..fb12975 100644 --- a/tests/replaceable_history.rs +++ b/tests/replaceable_history.rs @@ -74,10 +74,11 @@ async fn newer_30618_supersedes_older_and_preserves_old_in_history() { .expect("create audit client"); let (_announcement, repo_id) = publish_served_repo(&client, "history-30618").await; - // The fixture published by `publish_served_repo` already saved a 30618 - // for this coordinate. Use a deterministic later timestamp so this test's - // "old" version is always a valid replacement instead of racing the - // fixture event within the same one-second Nostr timestamp bucket. + // `publish_served_repo` already saved a 30618 for this coordinate. Nostr + // replaceable ordering is only second-granular, so a same-second "old" + // event can be rejected instead of superseding the fixture event. Start the + // test sequence in a deterministic future second to make both replacements + // strictly newer. let base_ts = Timestamp::from_secs(Timestamp::now().as_secs() + 1); let old_state = build_state_version(&client, &repo_id, base_ts, "state-v1"); let new_state = build_state_version( @@ -102,6 +103,9 @@ async fn newer_30618_supersedes_older_and_preserves_old_in_history() { let records = history .superseded_records_for_coordinate_before( &coordinate, + // The deterministic future timestamps can be ahead of wall-clock + // `Timestamp::now()`, so query just after the newest test event + // rather than using "now" as the history cutoff. Timestamp::from_secs(new_state.created_at.as_secs() + 1), ) .await; @@ -251,8 +255,9 @@ async fn serving_behavior_unchanged_latest_version_is_still_served() { .expect("create audit client"); let (_announcement, repo_id) = publish_served_repo(&client, "history-serving").await; - // The helper publishes an initial 30618 for this coordinate. Ensure the - // state versions created here are strictly newer than that fixture event. + // `publish_served_repo` already saved a 30618 for this coordinate. Keep the + // explicit state sequence in future seconds so the relay's second-granular + // replaceable ordering cannot tie it with the fixture event. let base_ts = Timestamp::from_secs(Timestamp::now().as_secs() + 1); let old_state = build_state_version(&client, &repo_id, base_ts, "state-old"); let new_state = build_state_version(