From ff79bc89de11177e91808d516eedb369db262d1a Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 14:27:02 +0000 Subject: [PATCH 01/30] perf(graperank): evict connect-silent hosts by authority after repeated timeouts MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit classifyDrainFailure deliberately treats every timeout — connect timeout or park idle-cut — as "busy, retry" and never dead, so a relay that connects but never answers a REQ gets re-routed through every straggler's outbox, every round, each visit burning the full timeout + park window for zero data. The outbox model makes this worse: one dead server (e.g. filter.nostr.wine) is advertised as hundreds of distinct per-user path URLs, so a per-URL counter never reaches a threshold on any single one. Count unproductive-timeout strikes per relay AUTHORITY (host[:port]) and evict the whole host after Config.timeoutEvictStrikes (default 3; CLI --timeout-evict, 0 disables). Any clean EOSE or delivered event clears the authority, so only never-productive hosts are evicted; a multi-path relay where some paths are slow but others deliver stays live. Authority is host-only and never folds a filter. subdomain into its parent, so an open bare host is untouched when its sibling filter host is shed. Purely behavior-driven — no NIP-11. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 1 + .../graperank/GrapeRankDataCrawler.kt | 133 +++++++++++++++++- .../graperank/GrapeRankAuthorityTest.kt | 66 +++++++++ 3 files changed, 193 insertions(+), 7 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index f3f5b21e68..6a1824aee8 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -372,6 +372,7 @@ object GrapeRankCommand { diagnose = args.bool("diagnose"), insertBatchSize = args.intFlag("insert-batch", 500), drainConcurrency = args.intFlag("drain-concurrency", 24), + timeoutEvictStrikes = args.intFlag("timeout-evict", 3), ), log = { System.err.println(it) }, ) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 682c633066..cb7d1e7c47 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -134,6 +134,13 @@ class GrapeRankDataCrawler( * moderate: a higher global fan-out re-floods busy hubs faster than demotion * catches up (an A/B at 64 ran ~2x slower with more dead relays), so 24 is the * validated default and raising it is a probe, not a speedup. + * @param timeoutEvictStrikes evict a relay after this many drains that timed out + * (connect timeout or park idle-cut) having delivered NOTHING. Unlike + * [classifyDrainFailure] — which never marks a timeout dead, since one slow + * answer shouldn't drop a relay — this catches the connect-but-silent / dead + * endpoints that are otherwise re-tried through every straggler's outbox for the + * rest of the crawl. A clean EOSE or any delivered event clears a relay's count, + * so only never-productive relays are evicted. `<= 0` disables it. */ class Config( val relayListDiscoveryRelays: Set, @@ -145,6 +152,7 @@ class GrapeRankDataCrawler( val diagnose: Boolean = false, val insertBatchSize: Int = 500, val drainConcurrency: Int = 24, + val timeoutEvictStrikes: Int = 3, ) /** What the crawl fetched — the counters the caller reports and the graph is built from. */ @@ -208,6 +216,24 @@ class GrapeRankDataCrawler( val deadRelays = ConcurrentSet() val relayStrikes = ConcurrentMap() + // Unproductive-TIMEOUT strikes, keyed by relay AUTHORITY (host[:port]), not the + // full URL. [classifyDrainFailure] deliberately treats every timeout — a connect + // timeout OR a park idle-cut — as "busy, retry" and never dead, because one slow + // answer shouldn't evict a relay. But in a crawl the same unresponsive server is + // routed through every straggler's outbox, every round, each visit burning the + // full timeout + park window for zero data. Keying by authority is what defeats + // the outbox-model's per-user path fragmentation: a paid/dead host like + // `filter.nostr.wine` is advertised as hundreds of distinct per-user URLs + // (`filter.nostr.wine/npubA?broadcast=true`, …), so a per-URL counter never + // reaches the threshold on any single one — but they are one server, and it + // times out on all of them. We strike the authority and, past + // [Config.timeoutEvictStrikes], mark it dead in [deadHosts] so every URL under + // it is skipped. A clean EOSE or any delivered event clears the authority (see + // [clearTimeoutStrikes]), so a host that ever produces is never evicted — only + // the connect-but-silent / dead-endpoint class is. + val deadTimeoutStrikes = ConcurrentMap() + val deadHosts = ConcurrentSet() + // Crawl-wide dedup of event ids, shared across all concurrent drains and // every round. The outbox model mirrors the SAME event (especially kind:10002 // relay lists) across many relays, indexers, and rounds; a per-drain set only @@ -271,11 +297,47 @@ class GrapeRankDataCrawler( } } + /** + * A relay's drain unit timed out (connect timeout or park idle-cut) having + * delivered nothing. Count the strike against its AUTHORITY and, once it + * reaches [Config.timeoutEvictStrikes], give up on the whole host — a + * connect-but-silent or dead endpoint that would otherwise be re-tried through + * every straggler's outbox for the rest of the crawl. Disabled when the + * threshold is <= 0. + */ + fun strikeUnproductiveTimeout(relay: NormalizedRelayUrl) { + val limit = config.timeoutEvictStrikes + if (limit <= 0) return + val authority = authorityOf(relay.url) + if (authority in deadHosts) return + if (deadTimeoutStrikes.merge(authority, 1) { a, b -> a + b } >= limit) deadHosts.add(authority) + } + + /** + * A relay just proved its host can produce — a clean EOSE or an actual event — + * so wipe any timeout strikes the authority accrued. Prevents an occasionally- + * slow but useful host (a busy backbone hub, or a multi-path relay where some + * paths are slow) from accumulating its way to eviction across a long crawl. + */ + fun clearTimeoutStrikes(relay: NormalizedRelayUrl) { + // ConcurrentMap exposes no remove; reset the count to 0 atomically (0 is + // below any positive eviction threshold, so it reads as "unstruck"). Guard + // on a prior entry so we don't insert a 0 for every host that ever answers. + val authority = authorityOf(relay.url) + if (deadTimeoutStrikes[authority] != null) deadTimeoutStrikes.merge(authority, 0) { _, _ -> 0 } + } + + /** + * A relay is out of the routing pool if it hard/transient-failed (per-URL + * [deadRelays]) or its whole authority was timeout-evicted ([deadHosts]). + */ + fun isDead(relay: NormalizedRelayUrl): Boolean = relay in deadRelays || authorityOf(relay.url) in deadHosts + /** The busiest live relays we've learned, excluding the dead ones. */ fun topLiveRelays(cap: Int): List = writeRelayFreq.entries .asSequence() - .filter { it.key in liveRelays && it.key !in deadRelays } + .filter { it.key in liveRelays && !isDead(it.key) } .sortedByDescending { it.value } .take(cap) .map { it.key } @@ -463,7 +525,7 @@ class GrapeRankDataCrawler( val perRelayAuthors = HashMap>() for (author in idsByAuthor.keys) { val write = relaysOf(author)?.writeRelaysNorm()?.takeIf { it.isNotEmpty() } ?: backbone - for (relay in write) if (relay !in deadRelays) perRelayAuthors.getOrPut(relay) { HashSet() }.add(author) + for (relay in write) if (!isDead(relay)) perRelayAuthors.getOrPut(relay) { HashSet() }.add(author) } if (perRelayAuthors.isEmpty()) return @@ -516,7 +578,7 @@ class GrapeRankDataCrawler( // (re-querying them for this user is guaranteed-empty waste). val emptied = askedEmpty[pk] for (relay in relays) { - if (relay in deadRelays) continue + if (isDead(relay)) continue if (emptied != null && relay in emptied) continue perRelay.getOrPut(relay) { HashSet() }.add(pk) } @@ -808,7 +870,17 @@ class GrapeRankDataCrawler( if (elapsedMs > SLOW_DRAIN_LOG_MS) logSlow(subRelay, reason, elapsedMs, groupFilters) unitEvents.close() client.unsubscribe(subId) - persist(buildList { for (e in unitEvents) add(e) }) + val drained = buildList { for (e in unitEvents) add(e) } + val persisted = persist(drained) + // Alive if it EOSE'd or handed us anything; a connect-timeout that + // gave nothing (classifyDrainFailure leaves it retryable forever) + // earns a strike toward eviction instead. + if (reason == "eose" || drained.isNotEmpty()) { + clearTimeoutStrikes(subRelay) + } else if (isTimeoutReason(reason)) { + strikeUnproductiveTimeout(subRelay) + } + persisted } else { // Still streaming — hand off and let the round move on. notAnswered.add(subRelay) @@ -829,7 +901,16 @@ class GrapeRankDataCrawler( classify(late, subRelay, lateDead) recordDead(lateDead.snapshot()) unitEvents.close() - for (pair in persist(buildList { for (e in unitEvents) add(e) })) lateHarvest.trySend(pair) + val drainedLate = buildList { for (e in unitEvents) add(e) } + for (pair in persist(drainedLate)) lateHarvest.trySend(pair) + // Same liveness rule as the fast path: a park that ended + // in a clean EOSE or delivered anything clears the relay; + // one that idle-cut ("timeout") with nothing strikes it. + if (late == "eose" || drainedLate.isNotEmpty()) { + clearTimeoutStrikes(subRelay) + } else if (late == "timeout" || isTimeoutReason(late)) { + strikeUnproductiveTimeout(subRelay) + } } finally { client.unsubscribe(subId) parkedInFlight.addAndFetch(-1) @@ -839,6 +920,9 @@ class GrapeRankDataCrawler( logSlow(subRelay, "timeout", mark.elapsedNow().inWholeMilliseconds, groupFilters) unitEvents.close() client.unsubscribe(subId) + // Parking disabled: a fast timeout with nothing delivered is + // the same unproductive-timeout signal, so strike it here too. + if (unitEvents.tryReceive().isFailure) strikeUnproductiveTimeout(subRelay) } emptyList() } @@ -920,7 +1004,7 @@ class GrapeRankDataCrawler( val backbone = topLiveRelays(BACKBONE_SIZE).toSet() // Snapshot of every relay we've seen work, for the wide Tier-2 // sweep (taken now, before the Phase-B workers mutate liveRelays). - val allLive = liveRelays.filterTo(HashSet()) { it !in deadRelays } + val allLive = liveRelays.filterTo(HashSet()) { !isDead(it) } ensureRelayLists(stragglers.toSet(), allLive, scope) // Continuous worker pool instead of chunked awaitAll barriers, so @@ -1035,7 +1119,8 @@ class GrapeRankDataCrawler( val stored = eventsStored.load() log( "[graperank] crawl complete: ${hopOf.size} discovered, $contactListsFed contact lists fed, " + - "${relaysContacted.size} relays contacted, ${deadRelays.size()} dead, $rounds rounds in $downloadMs ms; " + + "${relaysContacted.size} relays contacted, ${deadRelays.size()} dead + ${deadHosts.size()} timeout-evicted hosts, " + + "$rounds rounds in $downloadMs ms; " + "by hop: " + hopHistogram.entries.joinToString(" ") { "${it.key}=${it.value}" }, ) log( @@ -1093,6 +1178,40 @@ class GrapeRankDataCrawler( // timeout) is logged with its relay + filter, so slow relays can be replayed. private const val SLOW_DRAIN_LOG_MS = 4000L + /** + * Is a drain terminal reason a connect/read TIMEOUT — the class + * [classifyDrainFailure] leaves retryable forever? The reason shape is + * `cannot:` and the message now carries the exception class name + * (see BasicRelayClient), so a SocketTimeoutException surfaces as "timed + * out"/"timeout". Used to drive unproductive-timeout eviction. + */ + private fun isTimeoutReason(reason: String): Boolean { + if (!reason.startsWith("cannot")) return false + val m = reason.removePrefix("cannot:").lowercase() + return "timeout" in m || "timed out" in m + } + + /** + * The authority (host[:port]) of a normalized relay URL — the segment between + * the `wss://` / `ws://` scheme and the first `/`. This is the key the + * timeout-eviction counts on, so the many per-user path URLs the outbox model + * mints for one server (`filter.nostr.wine/npubA`, `filter.nostr.wine/npubB`, …) + * collapse to a single evictable host. A bare host is its own authority, so this + * is a no-op for the common no-path relay. Deliberately host-only: it must NOT + * fold `filter.nostr.wine` into `nostr.wine` — those are different servers with + * different behaviour (the bare host may read fine while the filter host stalls). + */ + fun authorityOf(url: String): String { + val afterScheme = + when { + url.startsWith("wss://") -> url.substring(6) + url.startsWith("ws://") -> url.substring(5) + else -> url + } + val slash = afterScheme.indexOf('/') + return if (slash >= 0) afterScheme.substring(0, slash) else afterScheme + } + // Once the frontier is empty but parked relays are still streaming, how long // to block waiting for one of them to deliver before re-checking convergence. private const val PARK_POLL_MS = 2000L diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt new file mode 100644 index 0000000000..bd83401d39 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt @@ -0,0 +1,66 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.experimental.graperank + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNotEquals + +/** + * [GrapeRankDataCrawler.authorityOf] is the key the crawl's timeout-eviction counts + * on. It must collapse the many per-user path URLs the outbox model mints for one + * server into a single host, WITHOUT folding a distinct sibling host (e.g. a + * `filter.` subdomain) into its parent. + */ +class GrapeRankAuthorityTest { + private fun auth(url: String) = GrapeRankDataCrawler.authorityOf(url) + + @Test + fun bareHostIsItsOwnAuthority() { + assertEquals("relay.damus.io", auth("wss://relay.damus.io")) + assertEquals("relay.damus.io", auth("wss://relay.damus.io/")) + assertEquals("nos.lol", auth("ws://nos.lol")) + } + + @Test + fun perUserPathUrlsOnOneHostCollapseToOneAuthority() { + val a = auth("wss://filter.nostr.wine/npub1aaaa?broadcast=true") + val b = auth("wss://filter.nostr.wine/npub1bbbb?broadcast=true&global=all") + val c = auth("wss://filter.nostr.wine/?global=all") + assertEquals("filter.nostr.wine", a) + assertEquals(a, b) + assertEquals(a, c) + } + + @Test + fun filterSubdomainIsNotFoldedIntoBareHost() { + // nostr.wine reads are open; filter.nostr.wine is a different server that may + // stall — evicting one must never take out the other. + assertNotEquals(auth("wss://filter.nostr.wine/npub1x"), auth("wss://nostr.wine")) + } + + @Test + fun portIsPartOfTheAuthority() { + assertEquals("relay.veganostr.com:443", auth("wss://relay.veganostr.com:443/npub1z")) + assertEquals("81.68.170.122:7114", auth("ws://81.68.170.122:7114/")) + assertNotEquals(auth("wss://example.com:443"), auth("wss://example.com:8080")) + } +} From 610f0c7355e342ab398fc168c85c651221939678 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 16:10:15 +0000 Subject: [PATCH 02/30] feat(graperank): recover straggler kind:3 from aggregator indexers MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The outbox model fetches a user's kind:3 only from their own kind:10002 write relays (and the write-frequency backbone). But a large tail of reachable users have no kind:3 on their own advertised outbox at all — it's dead, or they never published one there — while a network-wide aggregator (user.kindpag.es, …) that scrapes the whole network holds it. Those aggregators were queried only for kind:10002 relay lists in ensureRelayLists, never for kind:3 content, so the crawl structurally could not find these lists no matter how many rounds it ran. Add Config.contentAggregatorRelays and fold it into routeByOutbox for stragglers — users whose own outbox already failed (attempts > 0) or is unknown. The CLI wires the profile indexers (kindpag/purplepag/coracle/yabu/nostr1) plus the ActivityPub bridges (ditto/momostr/mostr, which host bridged users' lists); --no-aggregators disables it. Measured offline on observer 460c25e6 (max-hops 3): of ~2.2k users the crawl left without a contact list, querying the aggregators for kind:3 recovers ~500 (user.kindpag.es alone ~180) — lifting coverage from ~89% toward ~92%. The remainder have no kind:3 retrievable on any relay we know (bridged / inactive / never-published) — a data-absence floor, not a crawl deficiency. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 18 +++++++++++++++ .../graperank/GrapeRankDataCrawler.kt | 22 +++++++++++++++---- 2 files changed, 36 insertions(+), 4 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 0e51d070cd..3f819af485 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -104,6 +104,20 @@ object GrapeRankCommand { "wss://eden.nostr.land", ).mapNotNull { RelayUrlNormalizer.normalizeOrNull(it) }.toSet() + // Network-wide aggregators that scrape and hold kind:3 for users whose own + // outbox lacks it. The crawler queries these for a straggler's CONTENT (kind:3), + // not just their kind:10002 relay list. Measured on observer 460c25e6: querying + // user.kindpag.es for kind:3 alone recovers ~180 stragglers the pure outbox model + // never finds. The profile indexers (kindpag/purplepag/coracle/yabu/nostr1) plus + // the ActivityPub bridges (ditto/momostr/mostr, which host bridged users' lists). + private val CONTENT_AGGREGATOR_RELAYS: Set = + DefaultIndexerRelayList + + listOf( + "wss://relay.ditto.pub", + "wss://relay.momostr.pink", + "wss://relay.mostr.pub", + ).mapNotNull { RelayUrlNormalizer.normalizeOrNull(it) }.toSet() + suspend fun dispatch( dataDir: DataDir, tail: Array, @@ -358,6 +372,9 @@ object GrapeRankCommand { val discoveryRelays = ctx.bootstrapRelays() + Constants.eventFinderRelays + DefaultIndexerRelayList + EXTRA_DISCOVERY_RELAYS val contentFallback = ctx.bootstrapRelays() + Constants.eventFinderRelays + // Aggregator kind:3 recovery for stragglers is on by default; --no-aggregators + // disables it for A/B comparison. + val aggregators = if (args.bool("no-aggregators")) emptySet() else CONTENT_AGGREGATOR_RELAYS return GrapeRankDataCrawler( client = ctx.client, store = ctx.store, @@ -366,6 +383,7 @@ object GrapeRankCommand { GrapeRankDataCrawler.Config( relayListDiscoveryRelays = discoveryRelays, contentFallbackRelays = contentFallback, + contentAggregatorRelays = aggregators, maxRounds = args.intFlag("max-rounds", Int.MAX_VALUE), maxHops = args.intFlag("max-hops", Int.MAX_VALUE), timeoutMs = args.longFlag("timeout", 10L) * 1000, diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 996465664a..cb18b1cd92 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -111,6 +111,14 @@ class GrapeRankDataCrawler( * defaults that carry kind:10002 for most of the network. * @param contentFallbackRelays best-effort general relays that *might* hold a * user's kind:3/10000/1984 when their outbox is unknown or unreachable. + * @param contentAggregatorRelays index/aggregator relays queried for a STRAGGLER's + * kind:3 content (not just their kind:10002). The outbox model asks "where does + * this user write?" — but a large tail of users have no kind:3 on their own + * advertised outbox (or it's dead), while a network-wide aggregator (kindpag.es, + * …) scraped and holds it. Those aggregators are queried only for kind:10002 in + * [ensureRelayLists]; folding them in here, for users whose outbox already + * failed ([attempts] > 0) or is unknown, recovers contact lists the pure outbox + * model structurally cannot. Empty disables the behaviour. * @param maxRounds safety backstop on freshness passes (default: run to convergence). * @param maxHops follow-graph distance from the observer to crawl (Brainstorm uses 8). * @param timeoutMs the FAST per-drain timeout that gates a round's progression. @@ -145,6 +153,7 @@ class GrapeRankDataCrawler( class Config( val relayListDiscoveryRelays: Set, val contentFallbackRelays: Set, + val contentAggregatorRelays: Set = emptySet(), val maxRounds: Int = Int.MAX_VALUE, val maxHops: Int = Int.MAX_VALUE, val timeoutMs: Long = 10_000, @@ -590,8 +599,12 @@ class GrapeRankDataCrawler( * Group [pubkeys] by the relays we should query for their events: * - first try: the user's own kind:10002 write relays (the outbox model); * - a retry (`attempts[pk] > 0`, its outbox already failed): outbox + - * [backbone] — the known-good relays other people write to; - * - no outbox at all: harvested hints + backbone + the general fallback. + * [backbone] — the known-good relays other people write to — PLUS the + * content aggregators (kindpag.es, …), because a large tail of users have + * no kind:3 on their own outbox and only a network-wide aggregator holds it; + * - no outbox at all: harvested hints + backbone + the general fallback + + * the aggregators (same reason — their outbox is unknown, so the aggregator + * that scraped their kind:3 is often the only place to find it). * * Also tallies each user's write relays into [writeRelayFreq] so the * backbone can be learned from the crawl. Authors are chunked per relay. @@ -601,6 +614,7 @@ class GrapeRankDataCrawler( backbone: Set, ): Map> { val fallback = config.contentFallbackRelays + val aggregators = config.contentAggregatorRelays val perRelay = HashMap>() for (pk in pubkeys) { @@ -608,8 +622,8 @@ class GrapeRankDataCrawler( write?.forEach { writeRelayFreq[it] = (writeRelayFreq[it] ?: 0) + 1 } val relays = when { - write == null -> relayHints[pk]?.snapshot().orEmpty() + backbone + fallback - (attempts[pk] ?: 0) > 0 -> write + backbone + write == null -> relayHints[pk]?.snapshot().orEmpty() + backbone + fallback + aggregators + (attempts[pk] ?: 0) > 0 -> write + backbone + aggregators else -> write } // Skip relays proven dead (routing to them only burns the drain From 7598a157dd1022b9025aa8ec70e2bf4e664cb23d Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 16:59:14 +0000 Subject: [PATCH 03/30] fix(graperank): recover aggregator kind:3 for evicted hosts in the patient pass The dedicated straggler-recovery pass was skipping any content aggregator the main crawl had timeout-evicted, so it recovered ~1 contact list instead of the hundreds those indexers actually hold. Root cause: during the competitive crawl an indexer like user.kindpag.es is only ever asked for kind:10002 in bulk and kind:[3,10000,1984,10002] one author at a time. The latter parks and times out (60-80s each), striking the host until its authority is timeout-evicted. It is never asked for a clean bulk kind:3 -- the one thing it serves fast (~19 lists per 300 authors in seconds; ~369 of the run's missing authors live there). So by the time recovery runs, kindpag.es is dead and dropped from the aggregator set (8 configured -> 6 used), and the biggest single source of missing lists is never queried. Fix: the recovery pass now queries every configured aggregator regardless of eviction (drainGated doesn't re-check isDead, and a genuinely dead endpoint only costs one shared park window since units run concurrently), and clears any timeout strikes first so a partially-struck host starts clean. Also stops folding aggregators into routeByOutbox's multi-kind fan-out (they time out there) and asks them kind:3-only, matching what they serve. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 119 +++++++++++++++--- 1 file changed, 102 insertions(+), 17 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index cb18b1cd92..5154f2d617 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -111,14 +111,15 @@ class GrapeRankDataCrawler( * defaults that carry kind:10002 for most of the network. * @param contentFallbackRelays best-effort general relays that *might* hold a * user's kind:3/10000/1984 when their outbox is unknown or unreachable. - * @param contentAggregatorRelays index/aggregator relays queried for a STRAGGLER's - * kind:3 content (not just their kind:10002). The outbox model asks "where does - * this user write?" — but a large tail of users have no kind:3 on their own - * advertised outbox (or it's dead), while a network-wide aggregator (kindpag.es, - * …) scraped and holds it. Those aggregators are queried only for kind:10002 in - * [ensureRelayLists]; folding them in here, for users whose outbox already - * failed ([attempts] > 0) or is unknown, recovers contact lists the pure outbox - * model structurally cannot. Empty disables the behaviour. + * @param contentAggregatorRelays index/aggregator relays that hold a network-wide + * copy of kind:3, mined for stragglers by [recoverStragglersFromAggregators] + * after the crawl converges. The outbox model asks "where does this user write?" + * — but a large tail of users have no kind:3 on their own advertised outbox (it's + * dead, or they never published one there), while a network-wide aggregator + * (kindpag.es, …) scraped and holds it. Those aggregators are queried only for + * kind:10002 in [ensureRelayLists]; the dedicated recovery pass asks them for the + * kind:3 itself — patiently and kind:3-only, since a multi-kind filter makes the + * big aggregators time out. Empty disables the pass. * @param maxRounds safety backstop on freshness passes (default: run to convergence). * @param maxHops follow-graph distance from the observer to crawl (Brainstorm uses 8). * @param timeoutMs the FAST per-drain timeout that gates a round's progression. @@ -599,12 +600,13 @@ class GrapeRankDataCrawler( * Group [pubkeys] by the relays we should query for their events: * - first try: the user's own kind:10002 write relays (the outbox model); * - a retry (`attempts[pk] > 0`, its outbox already failed): outbox + - * [backbone] — the known-good relays other people write to — PLUS the - * content aggregators (kindpag.es, …), because a large tail of users have - * no kind:3 on their own outbox and only a network-wide aggregator holds it; - * - no outbox at all: harvested hints + backbone + the general fallback + - * the aggregators (same reason — their outbox is unknown, so the aggregator - * that scraped their kind:3 is often the only place to find it). + * [backbone] — the known-good relays other people write to; + * - no outbox at all: harvested hints + backbone + the general fallback. + * + * The content aggregators are deliberately NOT mixed in here: they only serve + * kind:3 to a kind:3-only filter and time out on this path's multi-kind + * [FETCH_KINDS] query, so recovering from them is done separately, once and + * patiently, in [recoverStragglersFromAggregators]. * * Also tallies each user's write relays into [writeRelayFreq] so the * backbone can be learned from the crawl. Authors are chunked per relay. @@ -614,7 +616,6 @@ class GrapeRankDataCrawler( backbone: Set, ): Map> { val fallback = config.contentFallbackRelays - val aggregators = config.contentAggregatorRelays val perRelay = HashMap>() for (pk in pubkeys) { @@ -622,8 +623,8 @@ class GrapeRankDataCrawler( write?.forEach { writeRelayFreq[it] = (writeRelayFreq[it] ?: 0) + 1 } val relays = when { - write == null -> relayHints[pk]?.snapshot().orEmpty() + backbone + fallback + aggregators - (attempts[pk] ?: 0) > 0 -> write + backbone + aggregators + write == null -> relayHints[pk]?.snapshot().orEmpty() + backbone + fallback + (attempts[pk] ?: 0) > 0 -> write + backbone else -> write } // Skip relays proven dead (routing to them only burns the drain @@ -644,6 +645,84 @@ class GrapeRankDataCrawler( } } + /** + * Final patient pass for the stragglers the outbox model couldn't resolve. + * A large tail of reachable users have no kind:3 on their own advertised + * outbox — it's dead, or they never published one there — while a + * network-wide aggregator ([Config.contentAggregatorRelays], e.g. + * kindpag.es) scraped and holds it. Mixing those aggregators into the + * competitive Phase-B fan-out doesn't work: there they'd be asked for the + * multi-kind [FETCH_KINDS] filter (which times them out) and would race + * thousands of outbox sockets, getting cut before a big aggregator finishes. + * So once the frontier is drained we ask the aggregators for the remaining + * stragglers' kind:3 ALONE: a handful of relays drained kind:3-only with the + * patient park window, not competing with the fan-out. Recovered contact + * lists are folded into the graph and persisted for a later `score`. + */ + private suspend fun recoverStragglersFromAggregators() { + // Query EVERY configured aggregator, even ones the main crawl evicted. + // During the competitive crawl an indexer like user.kindpag.es is only ever + // asked for kind:10002 in bulk and kind:[3,10000,1984,10002] one author at a + // time; the latter parks and times out (60–80s each), striking the host until + // it's timeout-evicted (isDead). It is never asked for a clean bulk kind:3 — + // the one thing it actually serves fast (≈19 lists per 300 authors in a few + // seconds). This deliberate patient pass IS that clean query, so eviction from + // the fan-out must not disqualify it here. [drainGated] subscribes to whatever + // filter map we hand it (it does not re-check isDead), and a genuinely dead + // endpoint just costs one shared park window since the units run concurrently. + val aggregators = config.contentAggregatorRelays.toHashSet() + if (aggregators.isEmpty()) return + // Wipe any timeout strikes the fan-out accrued so a partially-struck host + // starts this pass clean and a fast EOSE here keeps it healthy. + for (agg in aggregators) clearTimeoutStrikes(agg) + // Stragglers = crawled users we still have no kind:3 for. Most are already + // in `done` (their outbox attempts were exhausted), which is exactly why + // [harvest]/[ingestLate] can't be reused — they skip `done` users — so we + // fold these directly. + val stragglers = hopOf.keys.filterTo(HashSet()) { (hopOf[it] ?: 0) < config.maxHops && contactsOf(it) == null } + if (stragglers.isEmpty()) return + val before = contactListsFed + log("[graperank] aggregator recovery: ${stragglers.size} stragglers via ${aggregators.size} aggregators") + + // Build the query against the full straggler set BEFORE any folding (the + // filter lists are materialized here, so later mutation of `stragglers` is + // safe). Ask ONLY for kind:3 — the contact list we're missing. A multi-kind + // filter breaks the big aggregators: user.kindpag.es serves kind:3 in a few + // seconds when asked for it alone, but times out returning nothing when the + // same authors are requested with kinds=[3,10000,1984,10002]. Mutes/reports + // still come from the outbox model; the aggregator's job here is the lists. + val filters = + aggregators.associateWith { + stragglers.chunked(AUTHORS_PER_FILTER).map { chunk -> Filter(kinds = listOf(ContactListEvent.KIND), authors = chunk) } + } + + // Fold one delivered contact list per straggler, exactly once. + suspend fun foldAgg(events: List>) { + for ((relay, ev) in events) { + liveRelays.add(relay) + if (ev !is ContactListEvent) continue + val pk = ev.pubKey + if (pk !in stragglers) continue + val contacts = contactsOf(pk) ?: continue + stragglers.remove(pk) + done += pk + ingest(pk, contacts) + } + } + + relaysContacted += aggregators + // Fast deliveries fold immediately; a slow aggregator parks and its late + // kind:3 arrives on [lateHarvest], which we drain until the parked units + // finish — so a big aggregator that can't answer within the fast window is + // still fully harvested here instead of being abandoned. + foldAgg(drainGated(filters, null)) + while (parkedInFlight.load() > 0L) { + withTimeoutOrNull(PARK_POLL_MS) { lateHarvest.receive() }?.let { foldAgg(listOf(it)) } + } + while (true) foldAgg(listOf(lateHarvest.tryReceive().getOrNull() ?: break)) + log("[graperank] aggregator recovery: +${contactListsFed - before} contact lists") + } + /** * Dedup (crawl-wide [seenIds]), verify, and group-commit a unit's events, * returning the newly-stored ones tagged by relay. Safe to call concurrently @@ -1206,6 +1285,12 @@ class GrapeRankDataCrawler( // Crawl done — drop the warm pool. client.unsubscribe(WARM_SUB_ID) + // Patient final pass: recover the stragglers the outbox model couldn't + // resolve by asking the content aggregators for their kind:3 ALONE, no + // longer racing the full fan-out (which cut the aggregators short during + // the rounds). Runs before [scope] is cancelled so slow aggregators park. + recoverStragglersFromAggregators() + // Reports can be retracted. Ask each reporter's outbox for NIP-09 kind:5 // deletions that cite the reports we gathered (#e-filtered to our report // ids). The events land in the store; the caller decides which reports From eef2832bf4ed5a926ff3c521109de608b2b3cef0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 17:02:18 +0000 Subject: [PATCH 04/30] feat(graperank): add nostr.oxtr.dev and nos.lol to the content-aggregator set Per-relay attribution on observer 460c25e6 showed two big general relays hold kind:3 for a chunk of the missing authors that no profile indexer has: nostr.oxtr.dev (76 distinct) and nos.lol (72). Add both to the aggregator set so the patient kind:3-only recovery pass sweeps them alongside the indexers. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 12 ++++++++---- 1 file changed, 8 insertions(+), 4 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 3f819af485..fd9a64dfc3 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -106,16 +106,20 @@ object GrapeRankCommand { // Network-wide aggregators that scrape and hold kind:3 for users whose own // outbox lacks it. The crawler queries these for a straggler's CONTENT (kind:3), - // not just their kind:10002 relay list. Measured on observer 460c25e6: querying - // user.kindpag.es for kind:3 alone recovers ~180 stragglers the pure outbox model - // never finds. The profile indexers (kindpag/purplepag/coracle/yabu/nostr1) plus - // the ActivityPub bridges (ditto/momostr/mostr, which host bridged users' lists). + // not just their kind:10002 relay list. Measured on observer 460c25e6, the distinct + // missing authors whose kind:3 each holds: kindpag.es 369, yabu 126, oxtr.dev 76, + // nos.lol 72, ditto 56, nostr1 29, momostr 11, mostr 3. So beyond the profile + // indexers (kindpag/purplepag/coracle/yabu/nostr1) and the ActivityPub bridges + // (ditto/momostr/mostr, which host bridged users' lists), two big general relays -- + // nostr.oxtr.dev and nos.lol -- carry ~150 more that no indexer has. private val CONTENT_AGGREGATOR_RELAYS: Set = DefaultIndexerRelayList + listOf( "wss://relay.ditto.pub", "wss://relay.momostr.pink", "wss://relay.mostr.pub", + "wss://nostr.oxtr.dev", + "wss://nos.lol", ).mapNotNull { RelayUrlNormalizer.normalizeOrNull(it) }.toSet() suspend fun dispatch( From 041e6c83b879b9a54e58c6d2b22ee3aa4f9d664d Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 17:12:33 +0000 Subject: [PATCH 05/30] docs(graperank): correct why aggregator recovery is kind:3-only The comments said a multi-kind filter makes the big indexers "time out returning nothing." Reproduced against user.kindpag.es, the real mechanism is a per-REQ result cap: it returns ~100 events regardless of the requested limit, and a kinds=[3,10000,1984,10002] query fills that cap entirely with the far more abundant kind:10002, returning 0 kind:3. Asked kind:3-only it returns the contact lists in a few seconds. Same conclusion (query kind:3 alone), accurate reason. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 20 +++++++++++-------- 1 file changed, 12 insertions(+), 8 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 5154f2d617..29dc9600d8 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -603,10 +603,11 @@ class GrapeRankDataCrawler( * [backbone] — the known-good relays other people write to; * - no outbox at all: harvested hints + backbone + the general fallback. * - * The content aggregators are deliberately NOT mixed in here: they only serve - * kind:3 to a kind:3-only filter and time out on this path's multi-kind - * [FETCH_KINDS] query, so recovering from them is done separately, once and - * patiently, in [recoverStragglersFromAggregators]. + * The content aggregators are deliberately NOT mixed in here: this path's + * multi-kind [FETCH_KINDS] query loses their kind:3 to their per-REQ result + * cap (a big indexer fills the response with the abundant kind:10002 and + * returns no kind:3), so recovering from them is done separately — kind:3-only, + * once and patiently — in [recoverStragglersFromAggregators]. * * Also tallies each user's write relays into [writeRelayFreq] so the * backbone can be learned from the crawl. Authors are chunked per relay. @@ -687,10 +688,13 @@ class GrapeRankDataCrawler( // Build the query against the full straggler set BEFORE any folding (the // filter lists are materialized here, so later mutation of `stragglers` is // safe). Ask ONLY for kind:3 — the contact list we're missing. A multi-kind - // filter breaks the big aggregators: user.kindpag.es serves kind:3 in a few - // seconds when asked for it alone, but times out returning nothing when the - // same authors are requested with kinds=[3,10000,1984,10002]. Mutes/reports - // still come from the outbox model; the aggregator's job here is the lists. + // filter is useless against the big indexers: user.kindpag.es caps its + // response at ~100 events per REQ (it ignores our limit), so a + // kinds=[3,10000,1984,10002] query comes back 100× kind:10002 and 0× + // kind:3 — the abundant relay lists crowd the contact lists out entirely. + // Asked for kind:3 alone it returns them in a few seconds. Their kind:10002 + // is already fetched in bulk by [ensureRelayLists]; mutes/reports still come + // from the outbox model. The aggregator's job here is only the lists. val filters = aggregators.associateWith { stragglers.chunked(AUTHORS_PER_FILTER).map { chunk -> Filter(kinds = listOf(ContactListEvent.KIND), authors = chunk) } From f19b8052b0eab841909052223fdab019dc0f11f8 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 17:22:11 +0000 Subject: [PATCH 06/30] fix(graperank): paginate aggregator recovery so the page cap can't truncate it The recovery pass drained each aggregator with the crawl's single-shot path (drainGated: one REQ, collect until EOSE). Against an indexer that caps a page at ~100 events and ignores our limit, every straggler beyond the newest 100 was silently dropped -- and drainGated additionally merged all chunks into one giant REQ, which the big indexers answer with nothing at all. Query each aggregator with fetchAllPages instead, walking `until` cursors to exhaustion, one AUTHORS_PER_FILTER chunk per request so no request carries the whole straggler set. Relays paginate concurrently; each relay's chunks run sequentially to keep one subscription live per connection, gated by the same limiter. Delivered events land on a channel off the reader threads, then are verified/persisted and folded once. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 57 ++++++++++++------- 1 file changed, 36 insertions(+), 21 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 29dc9600d8..5cc896eef0 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -27,6 +27,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.AdaptiveRelayLimiter import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.DrainFailure import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.classifyDrainFailure +import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter @@ -685,19 +686,18 @@ class GrapeRankDataCrawler( val before = contactListsFed log("[graperank] aggregator recovery: ${stragglers.size} stragglers via ${aggregators.size} aggregators") - // Build the query against the full straggler set BEFORE any folding (the - // filter lists are materialized here, so later mutation of `stragglers` is - // safe). Ask ONLY for kind:3 — the contact list we're missing. A multi-kind - // filter is useless against the big indexers: user.kindpag.es caps its - // response at ~100 events per REQ (it ignores our limit), so a - // kinds=[3,10000,1984,10002] query comes back 100× kind:10002 and 0× - // kind:3 — the abundant relay lists crowd the contact lists out entirely. - // Asked for kind:3 alone it returns them in a few seconds. Their kind:10002 - // is already fetched in bulk by [ensureRelayLists]; mutes/reports still come - // from the outbox model. The aggregator's job here is only the lists. - val filters = - aggregators.associateWith { - stragglers.chunked(AUTHORS_PER_FILTER).map { chunk -> Filter(kinds = listOf(ContactListEvent.KIND), authors = chunk) } + // Chunk the stragglers once. Each chunk is queried on its OWN request: + // the big indexers return nothing for a filter carrying the whole set, so + // AUTHORS_PER_FILTER-sized chunks keep every request answerable. Ask ONLY + // for kind:3 — the contact list we're missing. A multi-kind filter is + // useless against these indexers: user.kindpag.es caps its response at ~100 + // events per page (it ignores our limit), so a kinds=[3,10000,1984,10002] + // query comes back 100× kind:10002 and 0× kind:3 — the abundant relay lists + // crowd the contact lists out. Their kind:10002 is already fetched in bulk + // by [ensureRelayLists]; mutes/reports still come from the outbox model. + val chunks = + stragglers.chunked(AUTHORS_PER_FILTER).map { chunk -> + Filter(kinds = listOf(ContactListEvent.KIND), authors = chunk) } // Fold one delivered contact list per straggler, exactly once. @@ -715,15 +715,30 @@ class GrapeRankDataCrawler( } relaysContacted += aggregators - // Fast deliveries fold immediately; a slow aggregator parks and its late - // kind:3 arrives on [lateHarvest], which we drain until the parked units - // finish — so a big aggregator that can't answer within the fast window is - // still fully harvested here instead of being abandoned. - foldAgg(drainGated(filters, null)) - while (parkedInFlight.load() > 0L) { - withTimeoutOrNull(PARK_POLL_MS) { lateHarvest.receive() }?.let { foldAgg(listOf(it)) } + // Paginate every aggregator with `until` cursors instead of the crawl's + // single-shot [drainGated]: that grabs one page and stops, so against a + // relay that caps a page at ~100 events any straggler beyond the newest 100 + // is silently dropped. [fetchAllPages] walks the whole filter to exhaustion. + // Relays run concurrently; each relay's chunks run sequentially so only one + // subscription is live per connection (staying under per-relay sub limits), + // gated by the [limiter] like every other query. Events land on a channel + // off the reader threads, then are verified/persisted and folded once. + val sink = Channel>(Channel.UNLIMITED) + coroutineScope { + for (agg in aggregators) { + launch { + for (chunk in chunks) { + limiter.withPermit(agg) { + client.fetchAllPages(agg, listOf(chunk), config.parkTimeoutMs) { ev -> + sink.trySend(agg to ev) + } + } + } + } + } } - while (true) foldAgg(listOf(lateHarvest.tryReceive().getOrNull() ?: break)) + sink.close() + foldAgg(persist(buildList { for (e in sink) add(e) })) log("[graperank] aggregator recovery: +${contactListsFed - before} contact lists") } From 22b8089a96171c0fed06110f73d707329567093d Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 17:35:52 +0000 Subject: [PATCH 07/30] Revert "fix(graperank): paginate aggregator recovery so the page cap can't truncate it" This reverts commit f19b8052b0eab841909052223fdab019dc0f11f8. --- .../graperank/GrapeRankDataCrawler.kt | 57 +++++++------------ 1 file changed, 21 insertions(+), 36 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 5cc896eef0..29dc9600d8 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -27,7 +27,6 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.AdaptiveRelayLimiter import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.DrainFailure import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.classifyDrainFailure -import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter @@ -686,18 +685,19 @@ class GrapeRankDataCrawler( val before = contactListsFed log("[graperank] aggregator recovery: ${stragglers.size} stragglers via ${aggregators.size} aggregators") - // Chunk the stragglers once. Each chunk is queried on its OWN request: - // the big indexers return nothing for a filter carrying the whole set, so - // AUTHORS_PER_FILTER-sized chunks keep every request answerable. Ask ONLY - // for kind:3 — the contact list we're missing. A multi-kind filter is - // useless against these indexers: user.kindpag.es caps its response at ~100 - // events per page (it ignores our limit), so a kinds=[3,10000,1984,10002] - // query comes back 100× kind:10002 and 0× kind:3 — the abundant relay lists - // crowd the contact lists out. Their kind:10002 is already fetched in bulk - // by [ensureRelayLists]; mutes/reports still come from the outbox model. - val chunks = - stragglers.chunked(AUTHORS_PER_FILTER).map { chunk -> - Filter(kinds = listOf(ContactListEvent.KIND), authors = chunk) + // Build the query against the full straggler set BEFORE any folding (the + // filter lists are materialized here, so later mutation of `stragglers` is + // safe). Ask ONLY for kind:3 — the contact list we're missing. A multi-kind + // filter is useless against the big indexers: user.kindpag.es caps its + // response at ~100 events per REQ (it ignores our limit), so a + // kinds=[3,10000,1984,10002] query comes back 100× kind:10002 and 0× + // kind:3 — the abundant relay lists crowd the contact lists out entirely. + // Asked for kind:3 alone it returns them in a few seconds. Their kind:10002 + // is already fetched in bulk by [ensureRelayLists]; mutes/reports still come + // from the outbox model. The aggregator's job here is only the lists. + val filters = + aggregators.associateWith { + stragglers.chunked(AUTHORS_PER_FILTER).map { chunk -> Filter(kinds = listOf(ContactListEvent.KIND), authors = chunk) } } // Fold one delivered contact list per straggler, exactly once. @@ -715,30 +715,15 @@ class GrapeRankDataCrawler( } relaysContacted += aggregators - // Paginate every aggregator with `until` cursors instead of the crawl's - // single-shot [drainGated]: that grabs one page and stops, so against a - // relay that caps a page at ~100 events any straggler beyond the newest 100 - // is silently dropped. [fetchAllPages] walks the whole filter to exhaustion. - // Relays run concurrently; each relay's chunks run sequentially so only one - // subscription is live per connection (staying under per-relay sub limits), - // gated by the [limiter] like every other query. Events land on a channel - // off the reader threads, then are verified/persisted and folded once. - val sink = Channel>(Channel.UNLIMITED) - coroutineScope { - for (agg in aggregators) { - launch { - for (chunk in chunks) { - limiter.withPermit(agg) { - client.fetchAllPages(agg, listOf(chunk), config.parkTimeoutMs) { ev -> - sink.trySend(agg to ev) - } - } - } - } - } + // Fast deliveries fold immediately; a slow aggregator parks and its late + // kind:3 arrives on [lateHarvest], which we drain until the parked units + // finish — so a big aggregator that can't answer within the fast window is + // still fully harvested here instead of being abandoned. + foldAgg(drainGated(filters, null)) + while (parkedInFlight.load() > 0L) { + withTimeoutOrNull(PARK_POLL_MS) { lateHarvest.receive() }?.let { foldAgg(listOf(it)) } } - sink.close() - foldAgg(persist(buildList { for (e in sink) add(e) })) + while (true) foldAgg(listOf(lateHarvest.tryReceive().getOrNull() ?: break)) log("[graperank] aggregator recovery: +${contactListsFed - before} contact lists") } From f6fa2620177395a3fcadcec1f2e6ff30be4dfeb0 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 17:48:04 +0000 Subject: [PATCH 08/30] fix(graperank): paginate capped relay pages across the whole crawl MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A single REQ can match up to authors×kinds events; a relay that caps its response below that silently drops the tail. Measured: user.kindpag.es returns at most ~100 events per REQ and ignores our limit, so a dense chunk -- 300 authors × the 4 FETCH_KINDS, or a popular-author kind:3 sweep -- loses everything past the newest 100 on the first (and only) page drainGated fetched. On a dense set kindpag returned 100 events single-shot vs 238 paginated; nos.lol and damus (higher caps) matched at 246 and 127. drainGated never paginated -- it took one page and moved on -- so this bit every sweep and outbox query, not just the aggregator recovery. Truncated users became stragglers that the multi-round retry mostly (not always) recovered elsewhere, which is why it stayed hidden. Now any page that comes back at FULL_PAGE_THRESHOLD (100, the smallest cap observed) is treated as possibly-capped and its remainder is drained in the background with fetchAllPages `until` cursors, streamed to lateHarvest exactly like a parked slow relay (tracked by parkedInFlight so the round waits for it, gated by the limiter). The boundary second is re-fetched and de-duplicated by persist's crawl-wide seen-set, so nothing double-counts. Only dense pages pay the extra REQs; the common under-cap page is untouched. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 50 +++++++++++++++++++ 1 file changed, 50 insertions(+) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 29dc9600d8..02aa4b6d3d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -27,6 +27,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.AdaptiveRelayLimiter import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.DrainFailure import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.classifyDrainFailure +import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter @@ -869,6 +870,40 @@ class GrapeRankDataCrawler( } } + /** + * A drain unit's page came back at the [FULL_PAGE_THRESHOLD] — it may have been + * truncated by the relay's per-REQ cap. Continue the SAME query in the + * background with `until` cursors ([fetchAllPages], starting at the page's + * oldest event, inclusive) to drain whatever the cap hid, streaming the extra + * events to [lateHarvest] just like a parked slow relay. Tracked by + * [parkedInFlight] so the round waits for it; gated by [limiter] and dropped if + * we have no [bgScope]. The boundary second is re-fetched and its already-seen + * events are dropped by [persist]'s crawl-wide dedup, so nothing double-counts. + * A no-op unless the page hit the threshold, so only dense units pay for it. + */ + private fun paginateIfCapped( + relay: NormalizedRelayUrl, + groupFilters: List, + page: List>, + ) { + if (page.size < FULL_PAGE_THRESHOLD) return + val scope = bgScope ?: return + val oldest = page.minOf { it.second.createdAt } + val contFilters = groupFilters.map { it.copy(until = oldest) } + parkedInFlight.addAndFetch(1) + scope.launch { + try { + val more = ArrayList>() + limiter.withPermit(relay) { + client.fetchAllPages(relay, contFilters, config.parkTimeoutMs) { ev -> more.add(relay to ev) } + } + for (pair in persist(more)) lateHarvest.trySend(pair) + } finally { + parkedInFlight.addAndFetch(-1) + } + } + } + /** * Subscribe each relay to its filters behind [limiter] and drain them. A relay * that reaches a terminal (EOSE/CLOSED/cannot-connect) within the FAST @@ -1009,6 +1044,9 @@ class GrapeRankDataCrawler( val drained = buildList { for (e in unitEvents) add(e) } telemetry.record(subRelay, RelayTelemetry.outcomeOf(reason, parked = false), elapsedMs, authorsIn(groupFilters), drained.size) val persisted = persist(drained) + // A full page from a clean EOSE may be the relay's cap, not the + // whole answer — background-paginate the remainder into lateHarvest. + if (reason == "eose") paginateIfCapped(subRelay, groupFilters, drained) // Alive if it EOSE'd or handed us anything; a connect-timeout that // gave nothing (classifyDrainFailure leaves it retryable forever) // earns a strike toward eviction instead. @@ -1042,6 +1080,9 @@ class GrapeRankDataCrawler( val lateDrained = buildList { for (e in unitEvents) add(e) } telemetry.record(subRelay, RelayTelemetry.outcomeOf(late, parked = true), lateMs, authorsIn(groupFilters), lateDrained.size) for (pair in persist(lateDrained)) lateHarvest.trySend(pair) + // A full parked page from a clean EOSE may also be capped — + // paginate its remainder in the background, same as the fast path. + if (late == "eose") paginateIfCapped(subRelay, groupFilters, lateDrained) // Same liveness rule as the fast path: a park that ended // in a clean EOSE or delivered anything clears the relay; // one that idle-cut ("timeout") with nothing strikes it. @@ -1542,6 +1583,15 @@ class GrapeRankDataCrawler( // Authors per REQ filter — keeps individual subscriptions within relay limits. private const val AUTHORS_PER_FILTER = 300 + // A single REQ can match up to authors×kinds events; a relay that caps its + // response below that silently drops the tail (measured: user.kindpag.es + // returns at most ~100 events per REQ and ignores our limit). Any page that + // comes back with at least this many events is treated as possibly-capped and + // paginated with `until` cursors to drain the rest. Set at the smallest page + // cap we've observed, so it catches every relay that caps at or above it while + // sparing the common under-cap page an extra REQ. + private const val FULL_PAGE_THRESHOLD = 100 + // Max total "entries" (authors + ids + tag values) in a single REQ frame. // Each entry is a ~67-byte hex string, so 2500 ≈ 167KB — under the 256KB // message cap most relays enforce. drainGated groups filters to stay within. From 5312b61164ca0cbfc65f5d3eaa7ceb47a547eb3d Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 19:01:09 +0000 Subject: [PATCH 09/30] fix(relay): evict connection-establishment failures on the first strike MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A connect failure and a read timeout were both treated as "busy, retry" and took three strikes to drop. Re-probing hop-8's failed relays fresh, outside the crawl, showed the two are not alike: relays that failed to ESTABLISH a connection (connect timed out, refused, unroutable, or the proxy couldn't tunnel the CONNECT) were 0/30 reachable — genuinely dead — while relays that hit a READ timeout were 12/18 (67%) reachable, alive but overloaded by the crawl's fan-out (user.kindpag.es among them). So classifyDrainFailure now returns HARD for connection-establishment failures (one strike drops them instead of burning two more dials on a dead host), while a read/generic timeout still returns null and stays on the patient, clear-on-success timeout-strike path so live-but-slow relays we need are not wrongly evicted. Mid-stream resets stay TRANSIENT. Adds DrainFailureTest, which the classifier previously had none of. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../relay/client/accessories/DrainFailure.kt | 39 ++++++++-- .../client/accessories/DrainFailureTest.kt | 77 +++++++++++++++++++ 2 files changed, 108 insertions(+), 8 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt index b2d605d044..94816d4f4f 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt @@ -27,13 +27,18 @@ package com.vitorpamplona.quartz.nip01Core.relay.client.accessories * - [HARD]: the relay answered wrong, or cannot exist. A bad HTTP upgrade (not a * websocket / dead status code), an unresolvable domain, or a TLS misconfig. * This will not fix itself, so one strike is enough to drop it. - * - [TRANSIENT]: a failure that might clear — connection refused / reset, host - * unreachable, or a temporary 429/5xx on the upgrade. Struck a few times - * before we give up. + * - [TRANSIENT]: a failure that might clear — a connection reset mid-stream or a + * temporary 429/5xx on the upgrade. Struck a few times before we give up. * - * A pure connect **timeout** is neither. The relay is most likely just busy, so - * we retry it and never mark it dead — [classifyDrainFailure] returns null for - * it (and for any non-failure terminal reason). + * The split between the two connect failures is drawn on measured reachability. On + * a hop-8 crawl, relays that failed to ESTABLISH a connection (connect timed out, + * refused, unroutable, or the proxy couldn't tunnel the CONNECT) were 0/30 reachable + * when re-probed fresh outside the crawl — genuinely dead, so they are [HARD] and one + * strike drops them. But relays that hit a *read* timeout (handshake accepted, slow + * to serve) were 12/18 (67%) reachable fresh — alive, only overloaded by the crawl's + * fan-out. Those must NOT be marked dead: [classifyDrainFailure] returns null for a + * read/generic timeout (and any non-failure terminal reason), and the crawler's + * per-authority timeout strikes, which CLEAR on any success, shed only the truly gone. */ enum class DrainFailure { HARD, TRANSIENT } @@ -48,8 +53,26 @@ fun classifyDrainFailure(reason: String): DrainFailure? { val m = reason.removePrefix("cannot:").lowercase() // The message now carries the exception class name (see BasicRelayClient), so // we can key on the stable *type* rather than localized message text. - // Busy, not dead: a connect/read timeout means the handshake just didn't - // finish in time. Retry it — the relay is probably fine, only slow or loaded. + // Couldn't even open the socket: the connect timed out, was refused, the host is + // unroutable, or the proxy couldn't tunnel the CONNECT. Measured 0/30 such relays + // reachable when re-probed fresh outside the crawl — dead, so one strike is enough. + // Checked BEFORE the timeout branch so "connect timed out" lands here and is not + // mistaken for the alive-but-slow *read* timeout below. + if ("connect timed out" in m || + "unexpected response code for connect" in m || // proxy couldn't CONNECT-tunnel + "connection refused" in m || + "econnrefused" in m || + "failed to connect" in m || + "no route to host" in m || + "network is unreachable" in m || + "network is down" in m + ) { + return DrainFailure.HARD + } + // Busy, not dead: a READ timeout means the relay accepted the handshake but was + // slow to serve — measured 12/18 (67%) reachable fresh outside the crawl, only + // overloaded by its fan-out. Retry (never mark dead); per-authority timeout + // strikes that clear on success shed the truly gone. if ("timeout" in m || "timed out" in m) return null // SocketTimeoutException, etc. // Cannot ever work: unresolvable domain (DNS) or a TLS misconfiguration. // Dead for good — one strike is enough. diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt new file mode 100644 index 0000000000..2b3a5f2f9e --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt @@ -0,0 +1,77 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.client.accessories + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNull + +class DrainFailureTest { + // Non-failure and non-"cannot" terminals are never dead signals. + @Test + fun nonFailureTerminalsAreNull() { + assertNull(classifyDrainFailure("eose")) + assertNull(classifyDrainFailure("closed:duplicate: sub")) + assertNull(classifyDrainFailure("timeout")) + } + + // A READ timeout (or generic post-handshake timeout) is alive-but-slow: never + // dead. Measured 67% of these relays were reachable when re-probed fresh. + @Test + fun readTimeoutsStayRetryable() { + assertNull(classifyDrainFailure("cannot:Read timed out (SocketTimeoutException)")) + assertNull(classifyDrainFailure("cannot:timeout (SocketTimeoutException)")) + } + + // Failing to ESTABLISH the connection is a strong dead signal (0/30 reachable + // fresh): HARD, so one strike drops it. "connect timed out" must be caught here + // and NOT fall through to the alive-but-slow read-timeout branch. + @Test + fun connectEstablishmentFailuresAreHard() { + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Connect timed out (SocketTimeoutException)")) + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Unexpected response code for CONNECT: (IOException)")) + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Connection refused (ConnectException)")) + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Failed to connect to /1.2.3.4:443")) + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:No route to host (NoRouteToHostException)")) + } + + // DNS and TLS misconfig can never work: HARD. + @Test + fun dnsAndTlsAreHard() { + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Unable to resolve host (UnknownHostException)")) + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Received fatal alert: unrecognized_name (SSLHandshakeException)")) + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:PKIX path building failed: certificate (CertificateException)")) + } + + // A bad HTTP upgrade is HARD unless the status is a retryable 429/5xx. + @Test + fun httpUpgradeSplitsOnStatus() { + assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Server Misconfigured. not a websocket")) + assertEquals(DrainFailure.TRANSIENT, classifyDrainFailure("cannot:Server Misconfigured. Response: 503 (ProtocolException)")) + } + + // A mid-stream reset (connection already established) might clear: TRANSIENT. + @Test + fun midStreamResetIsTransient() { + assertEquals(DrainFailure.TRANSIENT, classifyDrainFailure("cannot:Connection reset (SocketException)")) + assertEquals(DrainFailure.TRANSIENT, classifyDrainFailure("cannot:Broken pipe (SocketException)")) + } +} From 32c309e86f7535727061afe83e7dfedc5aea8690 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 19:33:27 +0000 Subject: [PATCH 10/30] =?UTF-8?q?refactor(relay):=20collapse=20TRANSIENT?= =?UTF-8?q?=20into=20DEAD=20=E2=80=94=20a=20failed=20relay=20is=20not=20re?= =?UTF-8?q?tried=20this=20run?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The drain classifier had two "act on it" verdicts, HARD (drop now) and TRANSIENT (strike a few times, might clear). Re-probing hop-8's failed relays fresh showed the TRANSIENT bucket almost never clears: 503 Service Unavailable 0/12 reachable, 502 Bad Gateway 3/15, connection-establishment failures 0/30; the codes that were alive (402/403) are gated and will never serve us, and 200 isn't a relay. So the extra dials TRANSIENT bought were spent on hosts that stay dead for the run. Collapse to a single DEAD verdict, dropped on the first strike, and carve out the only two connect failures that genuinely recover so they stay retryable (null): a READ timeout (relay answered the handshake, slow — 67% reachable fresh, kept on the clear-on-success authority-strike path) and an HTTP 429 rate-limit (alive, 4/4 reachable — retrying spaced by the limiter is how we get its data). Removes the now-unused relayStrikes map, MAX_DEAD_STRIKES, and the HARD/TRANSIENT merge. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 38 ++----- .../relay/client/accessories/DrainFailure.kt | 104 ++++++------------ .../client/accessories/DrainFailureTest.kt | 53 +++++---- 3 files changed, 79 insertions(+), 116 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 02aa4b6d3d..9607d68ba2 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -226,7 +226,7 @@ class GrapeRankDataCrawler( * — so those stay plain collections. The frontier IS [hopOf]'s key set: a user * is "discovered" iff it has a hop stamp. Only the state genuinely shared across * the producer / consumer / drain-worker coroutines is concurrent: relayHints, - * attempts, deadRelays, relayStrikes. + * attempts, deadRelays. */ private inner class CrawlRun( val observer: HexKey, @@ -249,12 +249,12 @@ class GrapeRankDataCrawler( val relayHints = ConcurrentMap>() val attempts = ConcurrentMap() val deadRelays = ConcurrentSet() - val relayStrikes = ConcurrentMap() // Unproductive-TIMEOUT strikes, keyed by relay AUTHORITY (host[:port]), not the - // full URL. [classifyDrainFailure] deliberately treats every timeout — a connect - // timeout OR a park idle-cut — as "busy, retry" and never dead, because one slow - // answer shouldn't evict a relay. But in a crawl the same unresponsive server is + // full URL. [classifyDrainFailure] treats a READ timeout or a park idle-cut — the + // relay answered the handshake but is slow — as "busy, retry" and never dead, + // because one slow answer shouldn't evict a relay. But in a crawl the same + // unresponsive server is // routed through every straggler's outbox, every round, each visit burning the // full timeout + park window for zero data. Keying by authority is what defeats // the outbox-model's per-user path fragmentation: a paid/dead host like @@ -324,20 +324,14 @@ class GrapeRankDataCrawler( var progConverging = false /** - * A relay that HARD-failed (bad domain, TLS misconfig, dead HTTP code) is - * dropped on the first strike: it will not fix itself. A TRANSIENT failure - * (refused/reset/unreachable, or a 429/5xx) might clear, so it takes - * MAX_DEAD_STRIKES before we give up. Pure timeouts never reach here — the - * drain treats them as busy-retry and does not report them dead at all. + * A relay [classifyDrainFailure] flagged [DrainFailure.DEAD] won't serve us + * this run (bad domain, TLS misconfig, dead/gated HTTP code, refused/reset, + * connect that never opened), so it is dropped on the first strike. Read + * timeouts and alive 429 rate-limits never reach here — the drain treats them + * as busy-retry and does not report them dead at all. */ fun recordDead(failed: Map) { - for ((r, kind) in failed) { - when (kind) { - DrainFailure.HARD -> deadRelays.add(r) - DrainFailure.TRANSIENT -> - if (relayStrikes.merge(r, 1) { a, b -> a + b } >= MAX_DEAD_STRIKES) deadRelays.add(r) - } - } + for ((r, _) in failed) deadRelays.add(r) } /** @@ -960,11 +954,7 @@ class GrapeRankDataCrawler( relay: NormalizedRelayUrl, into: ConcurrentMap, ) { - classifyDrainFailure(reason)?.let { kind -> - into.merge(relay, kind) { a, b -> - if (a == DrainFailure.HARD || b == DrainFailure.HARD) DrainFailure.HARD else DrainFailure.TRANSIENT - } - } + classifyDrainFailure(reason)?.let { kind -> into[relay] = kind } } fun logSlow( @@ -1678,10 +1668,6 @@ class GrapeRankDataCrawler( // kind:3 is often mirrored on a busy relay ranked below the top 10. private const val BROADCAST_RELAYS = 60 - // A relay that fails to CONNECT this many times is treated as dead. Kept - // above 1 so a single transient connect blip doesn't evict a relay. - private const val MAX_DEAD_STRIKES = 3 - // Most-used write relays kept as the known-good backbone for retrying users. private const val BACKBONE_SIZE = 30 diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt index 94816d4f4f..1786ac04c2 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailure.kt @@ -21,80 +21,48 @@ package com.vitorpamplona.quartz.nip01Core.relay.client.accessories /** - * Why a relay could not be used for a one-shot drain — when the reason is worth - * acting on (dropping the relay from further routing). + * A drain per-relay failure worth acting on: the relay will not serve us THIS run, + * so drop it from further routing on the first occurrence. There is only one such + * verdict — [DEAD] — because re-probing hop-8's failed relays fresh, outside the + * crawl, showed the old "might clear, retry a few times" (TRANSIENT) bucket almost + * never clears: 503 Service Unavailable was 0/12 reachable, 502 Bad Gateway 3/15, + * connection-establishment failures 0/30, and the codes that WERE alive (403/402) + * are gated and will never hand us events. Spending extra dials on them was waste. * - * - [HARD]: the relay answered wrong, or cannot exist. A bad HTTP upgrade (not a - * websocket / dead status code), an unresolvable domain, or a TLS misconfig. - * This will not fix itself, so one strike is enough to drop it. - * - [TRANSIENT]: a failure that might clear — a connection reset mid-stream or a - * temporary 429/5xx on the upgrade. Struck a few times before we give up. - * - * The split between the two connect failures is drawn on measured reachability. On - * a hop-8 crawl, relays that failed to ESTABLISH a connection (connect timed out, - * refused, unroutable, or the proxy couldn't tunnel the CONNECT) were 0/30 reachable - * when re-probed fresh outside the crawl — genuinely dead, so they are [HARD] and one - * strike drops them. But relays that hit a *read* timeout (handshake accepted, slow - * to serve) were 12/18 (67%) reachable fresh — alive, only overloaded by the crawl's - * fan-out. Those must NOT be marked dead: [classifyDrainFailure] returns null for a - * read/generic timeout (and any non-failure terminal reason), and the crawler's - * per-authority timeout strikes, which CLEAR on any success, shed only the truly gone. + * The only two connect failures that genuinely recover are kept OUT of this verdict + * by [classifyDrainFailure] returning null (retry, never dead): + * - a **read** timeout — the relay accepted the handshake but is slow to serve; + * 12/18 (67%) were reachable fresh, only overloaded by the crawl's fan-out. The + * crawler's per-authority timeout strikes, which CLEAR on success, shed the gone. + * - an HTTP **429 / too many requests** — alive and rate-limiting; 4/4 reachable + * fresh. Retrying (spaced by the rate limiter) is how we eventually get its data. */ -enum class DrainFailure { HARD, TRANSIENT } +enum class DrainFailure { DEAD, } /** - * Classify a drain per-relay terminal reason. Returns null when the relay should - * simply be retried (a timeout, or a non-failure like eose/closed). The reason - * shape is `cannot:` for a connect failure (see - * `BasicRelayClient.onCannotConnect`), or `eose` / `closed:…` / `timeout`. + * Classify a drain per-relay terminal reason. Returns null when the relay should be + * retried rather than dropped — a read/generic timeout, an alive 429 rate-limit, or + * a non-failure like eose/closed. Any other `cannot:` (see + * `BasicRelayClient.onCannotConnect`) is [DrainFailure.DEAD]: it will not serve us + * this run, so drop it now instead of paying repeated connect attempts. */ fun classifyDrainFailure(reason: String): DrainFailure? { if (!reason.startsWith("cannot")) return null val m = reason.removePrefix("cannot:").lowercase() - // The message now carries the exception class name (see BasicRelayClient), so - // we can key on the stable *type* rather than localized message text. - // Couldn't even open the socket: the connect timed out, was refused, the host is - // unroutable, or the proxy couldn't tunnel the CONNECT. Measured 0/30 such relays - // reachable when re-probed fresh outside the crawl — dead, so one strike is enough. - // Checked BEFORE the timeout branch so "connect timed out" lands here and is not - // mistaken for the alive-but-slow *read* timeout below. - if ("connect timed out" in m || - "unexpected response code for connect" in m || // proxy couldn't CONNECT-tunnel - "connection refused" in m || - "econnrefused" in m || - "failed to connect" in m || - "no route to host" in m || - "network is unreachable" in m || - "network is down" in m - ) { - return DrainFailure.HARD - } - // Busy, not dead: a READ timeout means the relay accepted the handshake but was - // slow to serve — measured 12/18 (67%) reachable fresh outside the crawl, only - // overloaded by its fan-out. Retry (never mark dead); per-authority timeout - // strikes that clear on success shed the truly gone. - if ("timeout" in m || "timed out" in m) return null // SocketTimeoutException, etc. - // Cannot ever work: unresolvable domain (DNS) or a TLS misconfiguration. - // Dead for good — one strike is enough. - if ("unknownhost" in m || // UnknownHostException - "unable to resolve host" in m || - "no address associated" in m || - "nodename nor servname" in m || - "sslhandshake" in m || // SSLHandshakeException - "sslpeerunverified" in m || - "sslexception" in m || - "certificate" in m || // CertificateException - "trust anchor" in m || - "certpath" in m - ) { - return DrainFailure.HARD - } - // Wrong HTTP upgrade. Usually a misconfigured endpoint (not a relay), but - // 429 / 5xx mean "busy, come back later", so those stay transient. - if ("server misconfigured" in m || "not a websocket" in m || "expected http 101" in m) { - val transientCode = Regex("response: (429|500|502|503|504)").containsMatchIn(m) - return if (transientCode) DrainFailure.TRANSIENT else DrainFailure.HARD - } - // Refused / reset / unreachable / anything else: might clear — retry a few times. - return DrainFailure.TRANSIENT + // Alive, only asking us to slow down: an HTTP 429 / "too many requests" reliably + // clears — 4/4 such relays were reachable when re-probed fresh. Retry it (the + // rate limiter spaces our opens); never drop it. + if ("429" in m || "too many requests" in m) return null + // A READ timeout means the relay accepted the handshake but is slow to serve — + // 12/18 (67%) reachable fresh, alive but overloaded by the fan-out. Retry; the + // crawler's per-authority timeout strikes, which clear on success, shed the gone. + // A *connect* timeout is the opposite (the socket never opened, 0/30 reachable), + // so it is excluded here and falls through to DEAD with every other failure. + if (("timeout" in m || "timed out" in m) && "connect timed out" !in m) return null + // Everything else won't serve us this run: connect refused / unroutable / the + // proxy couldn't tunnel the CONNECT, a DNS or TLS failure, a dead-or-not-a-relay + // HTTP upgrade (502/503/500/504/410/404/200/…), or a mid-stream reset. Measured + // mostly dead (503 0%, 502 20% reachable) and, when alive, gated (402/403) or not + // a relay (200). Drop it now rather than burn more dials on it. + return DrainFailure.DEAD } diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt index 2b3a5f2f9e..cf1c993344 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/DrainFailureTest.kt @@ -41,37 +41,46 @@ class DrainFailureTest { assertNull(classifyDrainFailure("cannot:timeout (SocketTimeoutException)")) } - // Failing to ESTABLISH the connection is a strong dead signal (0/30 reachable - // fresh): HARD, so one strike drops it. "connect timed out" must be caught here - // and NOT fall through to the alive-but-slow read-timeout branch. + // An HTTP 429 rate-limit is alive and will serve us after backoff: never dead. + // Measured 4/4 such relays reachable when re-probed fresh. @Test - fun connectEstablishmentFailuresAreHard() { - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Connect timed out (SocketTimeoutException)")) - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Unexpected response code for CONNECT: (IOException)")) - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Connection refused (ConnectException)")) - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Failed to connect to /1.2.3.4:443")) - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:No route to host (NoRouteToHostException)")) + fun rateLimitStaysRetryable() { + assertNull(classifyDrainFailure("cannot:Server Misconfigured. Response: 429 Too Many Requests (ProtocolException)")) } - // DNS and TLS misconfig can never work: HARD. + // Failing to ESTABLISH the connection is dead (0/30 reachable fresh). "connect + // timed out" must be caught as DEAD and NOT slip into the read-timeout branch. @Test - fun dnsAndTlsAreHard() { - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Unable to resolve host (UnknownHostException)")) - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Received fatal alert: unrecognized_name (SSLHandshakeException)")) - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:PKIX path building failed: certificate (CertificateException)")) + fun connectEstablishmentFailuresAreDead() { + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Connect timed out (SocketTimeoutException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Unexpected response code for CONNECT: (IOException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Connection refused (ConnectException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Failed to connect to /1.2.3.4:443")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:No route to host (NoRouteToHostException)")) } - // A bad HTTP upgrade is HARD unless the status is a retryable 429/5xx. + // DNS and TLS misconfig can never work: DEAD. @Test - fun httpUpgradeSplitsOnStatus() { - assertEquals(DrainFailure.HARD, classifyDrainFailure("cannot:Server Misconfigured. not a websocket")) - assertEquals(DrainFailure.TRANSIENT, classifyDrainFailure("cannot:Server Misconfigured. Response: 503 (ProtocolException)")) + fun dnsAndTlsAreDead() { + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Unable to resolve host (UnknownHostException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Received fatal alert: unrecognized_name (SSLHandshakeException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:PKIX path building failed: certificate (CertificateException)")) } - // A mid-stream reset (connection already established) might clear: TRANSIENT. + // Every other bad HTTP upgrade won't serve us this run (measured 503 0%, 502 20% + // reachable; 402/403 gated; 200 not a relay) — DEAD, dropped on the first strike. @Test - fun midStreamResetIsTransient() { - assertEquals(DrainFailure.TRANSIENT, classifyDrainFailure("cannot:Connection reset (SocketException)")) - assertEquals(DrainFailure.TRANSIENT, classifyDrainFailure("cannot:Broken pipe (SocketException)")) + fun deadOrGatedHttpUpgradesAreDead() { + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Server Misconfigured. not a websocket")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Server Misconfigured. Response: 503 Service Unavailable (ProtocolException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Server Misconfigured. Response: 502 Bad Gateway (ProtocolException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Server Misconfigured. Response: 402 Payment Required (ProtocolException)")) + } + + // A mid-stream reset won't hand us events this run either: DEAD. + @Test + fun midStreamResetIsDead() { + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Connection reset (SocketException)")) + assertEquals(DrainFailure.DEAD, classifyDrainFailure("cannot:Broken pipe (SocketException)")) } } From a824f6e09b8ab2921382c7a36e07f9d62c9f9e1f Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 20:42:09 +0000 Subject: [PATCH 11/30] perf(graperank): drop the per-batch awaitAll barrier in Phase B MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Phase B drained users in 256-user batches: a worker called drainGated for the whole batch and awaitAll'd every relay in it, so one slow relay held the worker (and the batch's already-finished fast relays' contact lists) for the full 10s fast window before anything was ingested. With 24 workers all waiting out their batches' slowest relay at once, progress dropped to 0 lists/sec in waves. Restructure to drain each relay independently and stream its result the instant it resolves — no per-batch join. A per-user counter (relaysLeft) tracks how many of a user's relays are still outstanding; the single-writer consumer finalizes a user (ingest, or count a failed outbox attempt) only when the last of its relays resolves, so correctness is unchanged. Concurrency is now a semaphore over relay-units rather than an implicit batches×fan-out product; drainConcurrency becomes "concurrent relay drains" (default 1024, ~the old 24-batch fan-out). A straggler the outbox model routes nowhere is finalized directly as a miss. Fast relays' lists are now ingested immediately instead of behind a batch's slowest relay, removing the 0/s stalls on slow-relay-heavy rounds. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 2 +- .../graperank/GrapeRankDataCrawler.kt | 151 ++++++++++-------- 2 files changed, 88 insertions(+), 65 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index fd9a64dfc3..638d240410 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -394,7 +394,7 @@ object GrapeRankCommand { parkTimeoutMs = args.longFlag("park-timeout", 40L) * 1000, diagnose = args.bool("diagnose"), insertBatchSize = args.intFlag("insert-batch", 500), - drainConcurrency = args.intFlag("drain-concurrency", 24), + drainConcurrency = args.intFlag("drain-concurrency", 1024), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), // shedDeadDiscovery / shardRotations keep their benchmarked-best // Config defaults. diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 9607d68ba2..08f48eed31 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -50,9 +50,9 @@ import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.coroutineScope import kotlinx.coroutines.delay -import kotlinx.coroutines.joinAll import kotlinx.coroutines.launch import kotlinx.coroutines.selects.select +import kotlinx.coroutines.sync.Semaphore import kotlinx.coroutines.withTimeoutOrNull import kotlin.concurrent.atomics.AtomicLong import kotlin.concurrent.atomics.ExperimentalAtomicApi @@ -138,12 +138,13 @@ class GrapeRankDataCrawler( * [IEventStore.batchInsert]. The outbox model streams the same events from * many relays through a single SQLite writer, so batching amortizes the * per-transaction + writer-mutex cost across the batch (coerced to `>= 1`). - * @param drainConcurrency how many outbox batches drain at once (the worker - * pool size). A GLOBAL bound (memory / open sockets); the per-relay - * concurrent-sub cap is enforced separately by [AdaptiveRelayLimiter]. Keep it - * moderate: a higher global fan-out re-floods busy hubs faster than demotion - * catches up (an A/B at 64 ran ~2x slower with more dead relays), so 24 is the - * validated default and raising it is a probe, not a speedup. + * @param drainConcurrency how many relay-units drain at once — a GLOBAL bound on + * concurrent outbox subscriptions (memory / open sockets); the per-relay + * concurrent-sub cap is enforced separately by [AdaptiveRelayLimiter]. Phase B + * drains each relay independently and streams the result (no per-batch join), + * so this counts relays, not batches. Keep it moderate: a higher global fan-out + * re-floods busy hubs faster than demotion catches up, so raising it is a probe, + * not a speedup. * @param timeoutEvictStrikes evict a relay after this many drains that timed out * (connect timeout or park idle-cut) having delivered NOTHING. Unlike * [classifyDrainFailure] — which never marks a timeout dead, since one slow @@ -162,7 +163,7 @@ class GrapeRankDataCrawler( val parkTimeoutMs: Long = 40_000, val diagnose: Boolean = false, val insertBatchSize: Int = 500, - val drainConcurrency: Int = 24, + val drainConcurrency: Int = 1024, val timeoutEvictStrikes: Int = 3, /** * Also skip proven-dead relays in the kind:10002 discovery sweep @@ -1236,58 +1237,43 @@ class GrapeRankDataCrawler( // on the producer (keeps writeRelayFreq serial) and ingest runs // only on the consumer (keeps done/builder/hopOf serial), now // overlapped with draining instead of blocked behind each batch. - val routed = Channel, Map>>>(config.drainConcurrency * 2) - val drainedOut = Channel(Channel.UNLIMITED) + // Per-user count of relay-units still outstanding. A user is finalized + // (its list ingested, or a failed attempt counted) only when this hits + // zero. The dispatcher sets a user's FULL count before launching any of + // its units, so a fast relay can't finalize the user before its slower + // sibling relays are even scheduled. + val relaysLeft = ConcurrentMap() + val drainedOut = Channel(Channel.UNLIMITED) + // Bound concurrent relay drains. Each holds its slot only for the fast + // window (it parks and releases before parkTimeout), so slots turn over + // quickly. Crucially there is NO per-batch join: every relay is drained + // independently and its result streamed the instant it resolves, so a + // slow relay never holds up other users — fast relays' contact lists are + // ingested immediately instead of waiting out a batch's slowest relay. + val unitGate = Semaphore(config.drainConcurrency) coroutineScope { - // Producer: route each batch by outbox (serial), backpressured - // by the bounded `routed` channel. - val producer = - launch { - for (batch in stragglers.chunked(USER_BATCH)) { - val filters = routeByOutbox(batch.toSet(), backbone) - routed.send(batch to filters) - } - routed.close() - } - // Drain workers: pure network, no shared graph-state writes - // except recordDead (concurrent-safe). Each captures the relays - // that cleanly EOSE'd, so the consumer can tell "answered empty" - // from "timed out" per user. - val workers = - List(config.drainConcurrency) { - launch { - for ((batch, filters) in routed) { - val dead = HashMap() - val answered = HashSet() - val events = drainGated(filters, dead, answered) - recordDead(dead) - drainedOut.send(DrainedBatch(batch, filters, answered, events)) - } - } - } - // Consumer: single-writer ingest, overlapped with draining. + // Consumer: single-writer ingest + per-user completion, overlapped + // with draining. Reads one relay's result at a time. val consumer = launch { for (d in drainedOut) { - relaysContacted += d.filters.keys - // Any relay that gave us an event is proven live + useful. - for ((relay, _) in d.events) liveRelays.add(relay) - - // Per user, record relays that answered (EOSE'd) but did - // not return their kind:3, so they aren't re-queried there. - val returnedByRelay = HashMap>() - for ((relay, ev) in d.events) { - if (ev is ContactListEvent) returnedByRelay.getOrPut(relay) { HashSet() }.add(ev.pubKey) - } - for (relay in d.answered) { - val asked = d.filters[relay]?.flatMapTo(HashSet()) { it.authors.orEmpty() } ?: continue - val returned = returnedByRelay[relay].orEmpty() - for (pk in asked) { - if (pk !in returned) askedEmpty.getOrPut(pk) { ConcurrentSet() }.add(relay) + d.relay?.let { relay -> + relaysContacted += relay + // A relay that gave us an event is proven live + useful. + for ((r, _) in d.events) liveRelays.add(r) + // If it EOSE'd but didn't return a user's kind:3, record + // that so the user isn't re-queried there next round. + if (d.answeredCleanly) { + val returned = HashSet() + for ((_, ev) in d.events) if (ev is ContactListEvent) returned.add(ev.pubKey) + for (pk in d.users) if (pk !in returned) askedEmpty.getOrPut(pk) { ConcurrentSet() }.add(relay) } } - - for (pk in d.batch) { + // One of each covered user's relays just resolved; finalize + // the user once all of them have. + for (pk in d.users) { + val left = relaysLeft.merge(pk, -1) { a, b -> a + b } ?: -1 + if (left > 0) continue if (pk in done) continue val contacts = contactsOf(pk) if (contacts != null) { @@ -1301,8 +1287,44 @@ class GrapeRankDataCrawler( } } } - producer.join() - workers.joinAll() + // Dispatcher: route each batch by outbox (serial on this coroutine, + // keeping writeRelayFreq single-writer), then launch one independent + // drain per relay. The inner scope joins all drains before we close + // the results channel. + coroutineScope { + for (batch in stragglers.chunked(USER_BATCH)) { + val filters = routeByOutbox(batch.toSet(), backbone) + // Set every routed user's full relay count BEFORE any drain runs. + val routedUsers = HashSet() + for ((_, fs) in filters) { + for (f in fs) { + for (a in f.authors.orEmpty()) { + relaysLeft.merge(a, 1) { x, y -> x + y } + routedUsers.add(a) + } + } + } + for ((relay, fs) in filters) { + val users = fs.flatMapTo(HashSet()) { it.authors.orEmpty() } + unitGate.acquire() + launch { + try { + val dead = HashMap() + val answered = HashSet() + val events = drainGated(mapOf(relay to fs), dead, answered) + recordDead(dead) + drainedOut.send(DrainedUnit(relay, users, relay in answered, events)) + } finally { + unitGate.release() + } + } + } + // A straggler the outbox model routed nowhere gets no unit — + // finalize it (a missed attempt) so it isn't stuck pending. + val orphans = batch.filterTo(HashSet()) { it !in routedUsers } + if (orphans.isNotEmpty()) drainedOut.send(DrainedUnit(null, orphans, false, emptyList())) + } + } drainedOut.close() consumer.join() } @@ -1383,15 +1405,16 @@ class GrapeRankDataCrawler( } /** - * One Phase-B batch after draining: the users asked for, the relay->filters map - * they were routed through, the relays that cleanly EOSE'd ([answered]), and the - * fresh events. Carries enough for the consumer to attribute "answered but - * empty" per user without re-deriving the routing. + * One relay's drained result, streamed to the consumer the moment it resolves — + * the [users] it covered, whether it cleanly EOSE'd ([answeredCleanly]), and the + * fresh [events]. A null [relay] is a "routed nowhere" marker: the covered users + * had no relay to query, so they carry no events and are finalized as a missed + * attempt. */ - private class DrainedBatch( - val batch: List, - val filters: Map>, - val answered: Set, + private class DrainedUnit( + val relay: NormalizedRelayUrl?, + val users: Set, + val answeredCleanly: Boolean, val events: List>, ) From 56724454b7e5cc6abf0b46d4f175ab0dbece9d9b Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 21:03:03 +0000 Subject: [PATCH 12/30] perf(graperank): raise drain concurrency to 4096 to match the old fan-out Dropping the per-batch awaitAll (previous commit) made the rounds faster but a hop-3 A/B regressed total wall (727s vs 532s): the Semaphore(1024) throttled concurrent relay drains to ~249 parked at peak vs the old batch model's ~4,438, so slow-relay park windows that the old model absorbed during the rounds spilled into a long serial finishing drain. Raise the default so the parked work drains inside the rounds again. Value under validation. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt | 2 +- .../quartz/experimental/graperank/GrapeRankDataCrawler.kt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 638d240410..a99f422948 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -394,7 +394,7 @@ object GrapeRankCommand { parkTimeoutMs = args.longFlag("park-timeout", 40L) * 1000, diagnose = args.bool("diagnose"), insertBatchSize = args.intFlag("insert-batch", 500), - drainConcurrency = args.intFlag("drain-concurrency", 1024), + drainConcurrency = args.intFlag("drain-concurrency", 4096), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), // shedDeadDiscovery / shardRotations keep their benchmarked-best // Config defaults. diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 08f48eed31..89227628f7 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -163,7 +163,7 @@ class GrapeRankDataCrawler( val parkTimeoutMs: Long = 40_000, val diagnose: Boolean = false, val insertBatchSize: Int = 500, - val drainConcurrency: Int = 1024, + val drainConcurrency: Int = 4096, val timeoutEvictStrikes: Int = 3, /** * Also skip proven-dead relays in the kind:10002 discovery sweep From f19f965435fe5e4b4c91a19d48d39bd613b74cca Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 21:18:19 +0000 Subject: [PATCH 13/30] Revert "perf(graperank): raise drain concurrency to 4096 to match the old fan-out" This reverts commit 56724454b7e5cc6abf0b46d4f175ab0dbece9d9b. --- .../com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt | 2 +- .../quartz/experimental/graperank/GrapeRankDataCrawler.kt | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index a99f422948..638d240410 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -394,7 +394,7 @@ object GrapeRankCommand { parkTimeoutMs = args.longFlag("park-timeout", 40L) * 1000, diagnose = args.bool("diagnose"), insertBatchSize = args.intFlag("insert-batch", 500), - drainConcurrency = args.intFlag("drain-concurrency", 4096), + drainConcurrency = args.intFlag("drain-concurrency", 1024), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), // shedDeadDiscovery / shardRotations keep their benchmarked-best // Config defaults. diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 89227628f7..08f48eed31 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -163,7 +163,7 @@ class GrapeRankDataCrawler( val parkTimeoutMs: Long = 40_000, val diagnose: Boolean = false, val insertBatchSize: Int = 500, - val drainConcurrency: Int = 4096, + val drainConcurrency: Int = 1024, val timeoutEvictStrikes: Int = 3, /** * Also skip proven-dead relays in the kind:10002 discovery sweep From a6a401fcd4a0c0a76182d516adbaa23698035a86 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 21:18:19 +0000 Subject: [PATCH 14/30] Revert "perf(graperank): drop the per-batch awaitAll barrier in Phase B" This reverts commit a824f6e09b8ab2921382c7a36e07f9d62c9f9e1f. --- .../amethyst/cli/commands/GrapeRankCommand.kt | 2 +- .../graperank/GrapeRankDataCrawler.kt | 151 ++++++++---------- 2 files changed, 65 insertions(+), 88 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 638d240410..fd9a64dfc3 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -394,7 +394,7 @@ object GrapeRankCommand { parkTimeoutMs = args.longFlag("park-timeout", 40L) * 1000, diagnose = args.bool("diagnose"), insertBatchSize = args.intFlag("insert-batch", 500), - drainConcurrency = args.intFlag("drain-concurrency", 1024), + drainConcurrency = args.intFlag("drain-concurrency", 24), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), // shedDeadDiscovery / shardRotations keep their benchmarked-best // Config defaults. diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 08f48eed31..9607d68ba2 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -50,9 +50,9 @@ import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.coroutineScope import kotlinx.coroutines.delay +import kotlinx.coroutines.joinAll import kotlinx.coroutines.launch import kotlinx.coroutines.selects.select -import kotlinx.coroutines.sync.Semaphore import kotlinx.coroutines.withTimeoutOrNull import kotlin.concurrent.atomics.AtomicLong import kotlin.concurrent.atomics.ExperimentalAtomicApi @@ -138,13 +138,12 @@ class GrapeRankDataCrawler( * [IEventStore.batchInsert]. The outbox model streams the same events from * many relays through a single SQLite writer, so batching amortizes the * per-transaction + writer-mutex cost across the batch (coerced to `>= 1`). - * @param drainConcurrency how many relay-units drain at once — a GLOBAL bound on - * concurrent outbox subscriptions (memory / open sockets); the per-relay - * concurrent-sub cap is enforced separately by [AdaptiveRelayLimiter]. Phase B - * drains each relay independently and streams the result (no per-batch join), - * so this counts relays, not batches. Keep it moderate: a higher global fan-out - * re-floods busy hubs faster than demotion catches up, so raising it is a probe, - * not a speedup. + * @param drainConcurrency how many outbox batches drain at once (the worker + * pool size). A GLOBAL bound (memory / open sockets); the per-relay + * concurrent-sub cap is enforced separately by [AdaptiveRelayLimiter]. Keep it + * moderate: a higher global fan-out re-floods busy hubs faster than demotion + * catches up (an A/B at 64 ran ~2x slower with more dead relays), so 24 is the + * validated default and raising it is a probe, not a speedup. * @param timeoutEvictStrikes evict a relay after this many drains that timed out * (connect timeout or park idle-cut) having delivered NOTHING. Unlike * [classifyDrainFailure] — which never marks a timeout dead, since one slow @@ -163,7 +162,7 @@ class GrapeRankDataCrawler( val parkTimeoutMs: Long = 40_000, val diagnose: Boolean = false, val insertBatchSize: Int = 500, - val drainConcurrency: Int = 1024, + val drainConcurrency: Int = 24, val timeoutEvictStrikes: Int = 3, /** * Also skip proven-dead relays in the kind:10002 discovery sweep @@ -1237,43 +1236,58 @@ class GrapeRankDataCrawler( // on the producer (keeps writeRelayFreq serial) and ingest runs // only on the consumer (keeps done/builder/hopOf serial), now // overlapped with draining instead of blocked behind each batch. - // Per-user count of relay-units still outstanding. A user is finalized - // (its list ingested, or a failed attempt counted) only when this hits - // zero. The dispatcher sets a user's FULL count before launching any of - // its units, so a fast relay can't finalize the user before its slower - // sibling relays are even scheduled. - val relaysLeft = ConcurrentMap() - val drainedOut = Channel(Channel.UNLIMITED) - // Bound concurrent relay drains. Each holds its slot only for the fast - // window (it parks and releases before parkTimeout), so slots turn over - // quickly. Crucially there is NO per-batch join: every relay is drained - // independently and its result streamed the instant it resolves, so a - // slow relay never holds up other users — fast relays' contact lists are - // ingested immediately instead of waiting out a batch's slowest relay. - val unitGate = Semaphore(config.drainConcurrency) + val routed = Channel, Map>>>(config.drainConcurrency * 2) + val drainedOut = Channel(Channel.UNLIMITED) coroutineScope { - // Consumer: single-writer ingest + per-user completion, overlapped - // with draining. Reads one relay's result at a time. + // Producer: route each batch by outbox (serial), backpressured + // by the bounded `routed` channel. + val producer = + launch { + for (batch in stragglers.chunked(USER_BATCH)) { + val filters = routeByOutbox(batch.toSet(), backbone) + routed.send(batch to filters) + } + routed.close() + } + // Drain workers: pure network, no shared graph-state writes + // except recordDead (concurrent-safe). Each captures the relays + // that cleanly EOSE'd, so the consumer can tell "answered empty" + // from "timed out" per user. + val workers = + List(config.drainConcurrency) { + launch { + for ((batch, filters) in routed) { + val dead = HashMap() + val answered = HashSet() + val events = drainGated(filters, dead, answered) + recordDead(dead) + drainedOut.send(DrainedBatch(batch, filters, answered, events)) + } + } + } + // Consumer: single-writer ingest, overlapped with draining. val consumer = launch { for (d in drainedOut) { - d.relay?.let { relay -> - relaysContacted += relay - // A relay that gave us an event is proven live + useful. - for ((r, _) in d.events) liveRelays.add(r) - // If it EOSE'd but didn't return a user's kind:3, record - // that so the user isn't re-queried there next round. - if (d.answeredCleanly) { - val returned = HashSet() - for ((_, ev) in d.events) if (ev is ContactListEvent) returned.add(ev.pubKey) - for (pk in d.users) if (pk !in returned) askedEmpty.getOrPut(pk) { ConcurrentSet() }.add(relay) + relaysContacted += d.filters.keys + // Any relay that gave us an event is proven live + useful. + for ((relay, _) in d.events) liveRelays.add(relay) + + // Per user, record relays that answered (EOSE'd) but did + // not return their kind:3, so they aren't re-queried there. + val returnedByRelay = HashMap>() + for ((relay, ev) in d.events) { + if (ev is ContactListEvent) returnedByRelay.getOrPut(relay) { HashSet() }.add(ev.pubKey) + } + for (relay in d.answered) { + val asked = d.filters[relay]?.flatMapTo(HashSet()) { it.authors.orEmpty() } ?: continue + val returned = returnedByRelay[relay].orEmpty() + for (pk in asked) { + if (pk !in returned) askedEmpty.getOrPut(pk) { ConcurrentSet() }.add(relay) } } - // One of each covered user's relays just resolved; finalize - // the user once all of them have. - for (pk in d.users) { - val left = relaysLeft.merge(pk, -1) { a, b -> a + b } ?: -1 - if (left > 0) continue + + for (pk in d.batch) { if (pk in done) continue val contacts = contactsOf(pk) if (contacts != null) { @@ -1287,44 +1301,8 @@ class GrapeRankDataCrawler( } } } - // Dispatcher: route each batch by outbox (serial on this coroutine, - // keeping writeRelayFreq single-writer), then launch one independent - // drain per relay. The inner scope joins all drains before we close - // the results channel. - coroutineScope { - for (batch in stragglers.chunked(USER_BATCH)) { - val filters = routeByOutbox(batch.toSet(), backbone) - // Set every routed user's full relay count BEFORE any drain runs. - val routedUsers = HashSet() - for ((_, fs) in filters) { - for (f in fs) { - for (a in f.authors.orEmpty()) { - relaysLeft.merge(a, 1) { x, y -> x + y } - routedUsers.add(a) - } - } - } - for ((relay, fs) in filters) { - val users = fs.flatMapTo(HashSet()) { it.authors.orEmpty() } - unitGate.acquire() - launch { - try { - val dead = HashMap() - val answered = HashSet() - val events = drainGated(mapOf(relay to fs), dead, answered) - recordDead(dead) - drainedOut.send(DrainedUnit(relay, users, relay in answered, events)) - } finally { - unitGate.release() - } - } - } - // A straggler the outbox model routed nowhere gets no unit — - // finalize it (a missed attempt) so it isn't stuck pending. - val orphans = batch.filterTo(HashSet()) { it !in routedUsers } - if (orphans.isNotEmpty()) drainedOut.send(DrainedUnit(null, orphans, false, emptyList())) - } - } + producer.join() + workers.joinAll() drainedOut.close() consumer.join() } @@ -1405,16 +1383,15 @@ class GrapeRankDataCrawler( } /** - * One relay's drained result, streamed to the consumer the moment it resolves — - * the [users] it covered, whether it cleanly EOSE'd ([answeredCleanly]), and the - * fresh [events]. A null [relay] is a "routed nowhere" marker: the covered users - * had no relay to query, so they carry no events and are finalized as a missed - * attempt. + * One Phase-B batch after draining: the users asked for, the relay->filters map + * they were routed through, the relays that cleanly EOSE'd ([answered]), and the + * fresh events. Carries enough for the consumer to attribute "answered but + * empty" per user without re-deriving the routing. */ - private class DrainedUnit( - val relay: NormalizedRelayUrl?, - val users: Set, - val answeredCleanly: Boolean, + private class DrainedBatch( + val batch: List, + val filters: Map>, + val answered: Set, val events: List>, ) From 54ad8375592dda8a2199554edcfde5782bea22d2 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 22:38:46 +0000 Subject: [PATCH 15/30] feat(graperank): TCP reachability pre-probe + .onion skip to cull the dead graveyard MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit At hop-8 the crawl dials into thousands of dead relay hints from old accounts. Most fail slowly: a silently-dropping host has no RST to receive, so the WS connect just hangs to the 7s connectTimeout. First-strike eviction pays that once per host, but with ~3,000 dead hosts that's ~80s of connect-setup serialized through the dispatcher. Add a background reachability culler: a cheap raw TCP connect (one round trip, 2s timeout) over the learned relays COLD-TAIL FIRST, dropping the unreachable ones into deadHosts before the WS path pays its 7s. The key property is that a tight TCP timeout is safe where a tight WS timeout is not — a busy-but-alive relay accepts the SYN instantly at the kernel level and only stalls at the app layer, so the probe separates "unreachable" from "slow" and never false-kills the busy. It only ever marks dead and probes each authority once; a host the WS path already resolved (isDead) is skipped, and a live host passes the probe, so the WS verdict always wins. Injected as an optional Config.reachabilityProbe (JVM: java.net.Socket in the CLI; --no-probe disables); writeRelayFreq becomes concurrent so the culler can read it while routeByOutbox writes. Also: when there's no Tor transport (Config.torEnabled=false), isDead skips every .onion relay on sight — no socket, no wasted connect. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 49 ++++++++ .../graperank/GrapeRankDataCrawler.kt | 117 ++++++++++++++++-- 2 files changed, 153 insertions(+), 13 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index fd9a64dfc3..b1c5941061 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -52,10 +52,15 @@ import com.vitorpamplona.quartz.nip85TrustedAssertions.list.tags.ServiceProvider import com.vitorpamplona.quartz.nip85TrustedAssertions.list.tags.ServiceType import com.vitorpamplona.quartz.nip85TrustedAssertions.users.ContactCardEvent import com.vitorpamplona.quartz.nip85TrustedAssertions.users.tags.RankTag +import kotlinx.coroutines.CancellationException import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.async import kotlinx.coroutines.awaitAll import kotlinx.coroutines.coroutineScope +import kotlinx.coroutines.withContext +import java.net.InetSocketAddress +import java.net.Socket +import java.net.URI import kotlin.math.roundToInt /** @@ -122,6 +127,46 @@ object GrapeRankCommand { "wss://nos.lol", ).mapNotNull { RelayUrlNormalizer.normalizeOrNull(it) }.toSet() + private const val PROBE_TIMEOUT_MS = 2000 + + /** + * Cheap reachability pre-probe: a raw TCP connect (one round trip) with a tight + * timeout. Returns false only when the port won't even accept a socket — a dead + * dropper, refusal, or unroutable/onion/LAN host — which the crawler drops into + * deadHosts before the WS path pays its 7s connectTimeout. A busy-but-alive relay + * accepts the SYN instantly at the kernel level (its slowness is at the app layer), + * so it passes here and is left for the real WS attempt. Unparseable host → true, + * so an odd URL is never culled on a parse quirk — let the WS decide. + */ + private suspend fun tcpReachable(relay: NormalizedRelayUrl): Boolean = + withContext(Dispatchers.IO) { + val hostPort = relayHostPort(relay) ?: return@withContext true + try { + Socket().use { it.connect(InetSocketAddress(hostPort.first, hostPort.second), PROBE_TIMEOUT_MS) } + true + } catch (e: Exception) { + if (e is CancellationException) throw e + false + } + } + + private fun relayHostPort(relay: NormalizedRelayUrl): Pair? = + try { + val uri = URI(relay.url) + val host = uri.host ?: return null + val port = + if (uri.port > 0) { + uri.port + } else if (relay.url.startsWith("wss://", ignoreCase = true)) { + 443 + } else { + 80 + } + host to port + } catch (e: Exception) { + null + } + suspend fun dispatch( dataDir: DataDir, tail: Array, @@ -396,6 +441,10 @@ object GrapeRankCommand { insertBatchSize = args.intFlag("insert-batch", 500), drainConcurrency = args.intFlag("drain-concurrency", 24), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), + // Cheap TCP reachability pre-probe (--no-probe to disable). No Tor + // transport here, so .onion relays are skipped on sight. + reachabilityProbe = if (args.bool("no-probe")) null else ::tcpReachable, + torEnabled = false, // shedDeadDiscovery / shardRotations keep their benchmarked-best // Config defaults. ), diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 9607d68ba2..764d4efd0e 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -49,10 +49,13 @@ import kotlinx.coroutines.awaitAll import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.coroutineScope +import kotlinx.coroutines.currentCoroutineContext import kotlinx.coroutines.delay +import kotlinx.coroutines.isActive import kotlinx.coroutines.joinAll import kotlinx.coroutines.launch import kotlinx.coroutines.selects.select +import kotlinx.coroutines.sync.Semaphore import kotlinx.coroutines.withTimeoutOrNull import kotlin.concurrent.atomics.AtomicLong import kotlin.concurrent.atomics.ExperimentalAtomicApi @@ -177,6 +180,25 @@ class GrapeRankDataCrawler( * 6-rotation wall cost with no completeness loss (1 clears too little). */ val shardRotations: Int = 2, + /** + * Optional cheap reachability pre-probe. Given a relay, returns false if it + * is definitely unreachable from here — a raw TCP connect (one round trip) + * that failed fast. A background culler runs it over the cold tail of learned + * relays and drops the unreachable ones into [deadHosts] BEFORE the expensive + * WS path pays the full 7s connectTimeout on them. It only ever marks dead + * and defers to the WS verdict: a host already proven live/dead is skipped. + * A tight TCP timeout is safe where a tight WS timeout is not — a busy-but- + * alive relay accepts the SYN instantly (kernel-level) and only stalls at the + * app layer, so TCP-reachability separates "unreachable" from "slow". Null + * disables pre-probing. + */ + val reachabilityProbe: (suspend (NormalizedRelayUrl) -> Boolean)? = null, + /** + * Whether this client can reach .onion relays (has a Tor transport). When + * false, every .onion relay is unreachable and [isDead] skips it on sight — + * no socket, no wasted connect attempt. + */ + val torEnabled: Boolean = false, ) /** What the crawl fetched — the counters the caller reports and the graph is built from. */ @@ -219,14 +241,13 @@ class GrapeRankDataCrawler( } /** - * Holds all per-crawl mutable state. Graph state (done/hopOf/builder/ - * writeRelayFreq/liveRelays/relaysContacted) is single-writer by construction - * — Phase A and the Phase-B consumer never run concurrently, and routeByOutbox - * (the only Phase-B producer write, to writeRelayFreq) touches a disjoint field - * — so those stay plain collections. The frontier IS [hopOf]'s key set: a user - * is "discovered" iff it has a hop stamp. Only the state genuinely shared across - * the producer / consumer / drain-worker coroutines is concurrent: relayHints, - * attempts, deadRelays. + * Holds all per-crawl mutable state. Graph state (done/hopOf/builder/liveRelays/ + * relaysContacted) is single-writer by construction — Phase A and the Phase-B + * consumer never run concurrently — so those stay plain collections. The frontier + * IS [hopOf]'s key set: a user is "discovered" iff it has a hop stamp. State + * genuinely shared across the producer / consumer / drain-worker coroutines is + * concurrent: relayHints, attempts, deadRelays. [writeRelayFreq] is also concurrent + * because the background reachability culler reads it while routeByOutbox writes it. */ private inner class CrawlRun( val observer: HexKey, @@ -237,7 +258,7 @@ class GrapeRankDataCrawler( val hopOf = hashMapOf(observer to 0) val done = hashSetOf() val relaysContacted = hashSetOf() - val writeRelayFreq = HashMap() + val writeRelayFreq = ConcurrentMap() val liveRelays = hashSetOf() // Per-relay outcome/latency/yield accounting, written from every drain unit @@ -366,13 +387,20 @@ class GrapeRankDataCrawler( /** * A relay is out of the routing pool if it hard/transient-failed (per-URL - * [deadRelays]) or its whole authority was timeout-evicted ([deadHosts]). + * [deadRelays]) or its whole authority was timeout-evicted ([deadHosts]); a + * .onion relay is dead on sight unless we have a Tor transport, since every + * connect to it would only hang and fail. */ - fun isDead(relay: NormalizedRelayUrl): Boolean = relay in deadRelays || authorityOf(relay.url) in deadHosts + fun isDead(relay: NormalizedRelayUrl): Boolean = + relay in deadRelays || + authorityOf(relay.url) in deadHosts || + (!config.torEnabled && relay.url.contains(".onion")) /** The busiest live relays we've learned, excluding the dead ones. */ fun topLiveRelays(cap: Int): List = - writeRelayFreq.entries + writeRelayFreq + .snapshot() + .entries .asSequence() .filter { it.key in liveRelays && !isDead(it.key) } .sortedByDescending { it.value } @@ -380,6 +408,54 @@ class GrapeRankDataCrawler( .map { it.key } .toList() + /** + * Background reachability culler. Cheaply TCP-probes the relays we've learned — + * COLD TAIL FIRST — and drops the unreachable ones into [deadHosts] so the WS + * path never pays the 7s connectTimeout on a dead host. It only ever marks dead + * and probes each authority once: a host already resolved by the WS path + * ([isDead]) is skipped, and a live host would pass the TCP probe anyway, so the + * WS verdict always wins ("if the websocket gets there first, let it run"). The + * cold-tail ordering keeps it off the hot relays the crawl is actively dialing. + * Runs on [bgScope] until the crawl cancels it. + */ + private suspend fun cullUnreachable(probe: suspend (NormalizedRelayUrl) -> Boolean) { + val probed = HashSet() // authorities; only ever touched by this coroutine's loop + val gate = Semaphore(PROBE_CONCURRENCY) + while (currentCoroutineContext().isActive) { + // Least-written relays are the niche/dead long tail the WS path reaches + // last — probing them first buys the most head start with the least + // contention against the busy relays already being connected. + val batch = + writeRelayFreq + .snapshot() + .entries + .asSequence() + .filter { authorityOf(it.key.url) !in probed && !isDead(it.key) } + .sortedBy { it.value } + .map { it.key } + .toList() + if (batch.isEmpty()) { + delay(PROBE_IDLE_MS) + continue + } + coroutineScope { + for (relay in batch) { + val authority = authorityOf(relay.url) + if (!probed.add(authority)) continue + gate.acquire() + launch { + try { + // Re-check: the WS path may have resolved it while queued. + if (!isDead(relay) && !probe(relay)) deadHosts.add(authority) + } finally { + gate.release() + } + } + } + } + } + } + /** * Feed a user's contact list into the graph, harvest relay hints, stamp * the hop distance of newly-seen follows, and add them to the frontier. @@ -616,7 +692,7 @@ class GrapeRankDataCrawler( for (pk in pubkeys) { val write = relaysOf(pk)?.writeRelaysNorm()?.takeIf { it.isNotEmpty() } - write?.forEach { writeRelayFreq[it] = (writeRelayFreq[it] ?: 0) + 1 } + write?.forEach { writeRelayFreq.merge(it, 1) { a, b -> a + b } } val relays = when { write == null -> relayHints[pk]?.snapshot().orEmpty() + backbone + fallback @@ -1172,6 +1248,12 @@ class GrapeRankDataCrawler( // Runs on [scope], so scope.cancel() at crawl end stops it. scope.launch { progressTicker() } + // Background reachability culler: cheaply TCP-probes the cold tail of + // learned relays and drops the unreachable ones into deadHosts before the + // WS path pays the full connectTimeout on them. Runs on [scope], stopped + // by scope.cancel() at crawl end. + config.reachabilityProbe?.let { probe -> scope.launch { cullUnreachable(probe) } } + while (rounds < config.maxRounds) { // Fold in whatever the parked (slow-but-alive) relays have delivered // since the last round — their late contact lists expand the frontier @@ -1573,6 +1655,15 @@ class GrapeRankDataCrawler( // Authors per REQ filter — keeps individual subscriptions within relay limits. private const val AUTHORS_PER_FILTER = 300 + // Concurrent TCP reachability probes in the background culler. Raw sockets are + // cheap and short-lived; the per-relay WS limiter is unaffected (this never + // opens a REQ), so this only bounds file descriptors during the cull. + private const val PROBE_CONCURRENCY = 256 + + // Re-scan interval for the culler when it has probed everything learned so far + // and is waiting for new relays to be discovered. + private const val PROBE_IDLE_MS = 2000L + // A single REQ can match up to authors×kinds events; a relay that caps its // response below that silently drops the tail (measured: user.kindpag.es // returns at most ~100 events per REQ and ignores our limit). Any page that From b02461f00044b255e6f9e5d98b2024cedc229505 Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 22:54:02 +0000 Subject: [PATCH 16/30] fix(graperank): reachability culler skips already-live relays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The culler filtered candidates by !isDead and not-yet-probed, but not by liveRelays — so it probed relays the WS path had already proven live, wasting a probe and opening a needless TCP connection to the hot relays the crawl depends on. Skip any authority already in liveRelays up front. liveRelays becomes a ConcurrentSet so the background culler can read it while the crawl writes it. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 32 ++++++++++++------- 1 file changed, 20 insertions(+), 12 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 764d4efd0e..12dd36e1a3 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -241,13 +241,14 @@ class GrapeRankDataCrawler( } /** - * Holds all per-crawl mutable state. Graph state (done/hopOf/builder/liveRelays/ + * Holds all per-crawl mutable state. Graph state (done/hopOf/builder/ * relaysContacted) is single-writer by construction — Phase A and the Phase-B * consumer never run concurrently — so those stay plain collections. The frontier * IS [hopOf]'s key set: a user is "discovered" iff it has a hop stamp. State * genuinely shared across the producer / consumer / drain-worker coroutines is - * concurrent: relayHints, attempts, deadRelays. [writeRelayFreq] is also concurrent - * because the background reachability culler reads it while routeByOutbox writes it. + * concurrent: relayHints, attempts, deadRelays. [writeRelayFreq] and [liveRelays] + * are also concurrent because the background reachability culler reads them (to + * find candidates and skip already-live authorities) while the crawl writes them. */ private inner class CrawlRun( val observer: HexKey, @@ -259,7 +260,7 @@ class GrapeRankDataCrawler( val done = hashSetOf() val relaysContacted = hashSetOf() val writeRelayFreq = ConcurrentMap() - val liveRelays = hashSetOf() + val liveRelays = ConcurrentSet() // Per-relay outcome/latency/yield accounting, written from every drain unit // (fast + parked) across every round. Dumped at crawl end; the raw signal a @@ -412,16 +413,21 @@ class GrapeRankDataCrawler( * Background reachability culler. Cheaply TCP-probes the relays we've learned — * COLD TAIL FIRST — and drops the unreachable ones into [deadHosts] so the WS * path never pays the 7s connectTimeout on a dead host. It only ever marks dead - * and probes each authority once: a host already resolved by the WS path - * ([isDead]) is skipped, and a live host would pass the TCP probe anyway, so the - * WS verdict always wins ("if the websocket gets there first, let it run"). The - * cold-tail ordering keeps it off the hot relays the crawl is actively dialing. - * Runs on [bgScope] until the crawl cancels it. + * and probes each authority once. Any host the WS path already resolved is + * skipped: dead ones via [isDead], and hosts already proven LIVE ([liveRelays]) + * are filtered out up front so we never waste a probe — or a needless TCP hit — + * on a working relay we depend on. Combined with the cold-tail ordering, the + * probe stays off the hot relays the crawl is actively dialing, and the WS + * verdict always wins ("if the websocket gets there first, let it run"). Runs on + * [bgScope] until the crawl cancels it. */ private suspend fun cullUnreachable(probe: suspend (NormalizedRelayUrl) -> Boolean) { val probed = HashSet() // authorities; only ever touched by this coroutine's loop val gate = Semaphore(PROBE_CONCURRENCY) while (currentCoroutineContext().isActive) { + // Never probe a host the WS path already proved live — wasted work and + // a needless TCP hit on the hot relays we depend on. + val liveAuthorities = liveRelays.snapshot().mapTo(HashSet()) { authorityOf(it.url) } // Least-written relays are the niche/dead long tail the WS path reaches // last — probing them first buys the most head start with the least // contention against the busy relays already being connected. @@ -430,8 +436,10 @@ class GrapeRankDataCrawler( .snapshot() .entries .asSequence() - .filter { authorityOf(it.key.url) !in probed && !isDead(it.key) } - .sortedBy { it.value } + .filter { + val authority = authorityOf(it.key.url) + authority !in probed && authority !in liveAuthorities && !isDead(it.key) + }.sortedBy { it.value } .map { it.key } .toList() if (batch.isEmpty()) { @@ -1309,7 +1317,7 @@ class GrapeRankDataCrawler( val backbone = topLiveRelays(BACKBONE_SIZE).toSet() // Snapshot of every relay we've seen work, for the wide Tier-2 // sweep (taken now, before the Phase-B workers mutate liveRelays). - val allLive = liveRelays.filterTo(HashSet()) { !isDead(it) } + val allLive = liveRelays.snapshot().filterTo(HashSet()) { !isDead(it) } ensureRelayLists(stragglers.toSet(), allLive, scope) // Continuous worker pool instead of chunked awaitAll barriers, so From d6db83b43dc5b9ce1cc8b2dd09c93e60dc5a26fa Mon Sep 17 00:00:00 2001 From: Claude Date: Wed, 8 Jul 2026 23:11:48 +0000 Subject: [PATCH 17/30] fix(graperank): run reachability probe on an isolated thread pool MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The probe does blocking DNS + TCP connect, and dead-domain DNS lookups hang well past the connect timeout. On the shared Dispatchers.IO those hanging lookups starved the crawl's own IO: an A/B at hop-3 showed probe-on 981s vs probe-off 517s, the entire +464s landing on the finishing drain (rounds were identical). Coverage was unchanged (91.84% vs 91.74%), so the probe classification is correct — it was purely IO contention. Give the probe its own fixed daemon pool (128 threads) so its blocking work can never touch the crawl's IO, and align the culler's concurrency to it. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 13 ++++++++++++- .../experimental/graperank/GrapeRankDataCrawler.kt | 2 +- 2 files changed, 13 insertions(+), 2 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index b1c5941061..99f6294dd6 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -54,6 +54,7 @@ import com.vitorpamplona.quartz.nip85TrustedAssertions.users.ContactCardEvent import com.vitorpamplona.quartz.nip85TrustedAssertions.users.tags.RankTag import kotlinx.coroutines.CancellationException import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.asCoroutineDispatcher import kotlinx.coroutines.async import kotlinx.coroutines.awaitAll import kotlinx.coroutines.coroutineScope @@ -61,6 +62,7 @@ import kotlinx.coroutines.withContext import java.net.InetSocketAddress import java.net.Socket import java.net.URI +import java.util.concurrent.Executors import kotlin.math.roundToInt /** @@ -129,6 +131,15 @@ object GrapeRankCommand { private const val PROBE_TIMEOUT_MS = 2000 + // The probe does BLOCKING DNS + TCP connect, and dead-domain DNS lookups can hang + // far past the connect timeout. On the shared Dispatchers.IO those hanging lookups + // starve the crawl's own IO — measured +462s on the finishing drain at hop-3. Run + // them on a dedicated, isolated daemon pool instead so the crawl's IO is untouched. + private val probeDispatcher = + Executors + .newFixedThreadPool(128) { r -> Thread(r, "relay-probe").apply { isDaemon = true } } + .asCoroutineDispatcher() + /** * Cheap reachability pre-probe: a raw TCP connect (one round trip) with a tight * timeout. Returns false only when the port won't even accept a socket — a dead @@ -139,7 +150,7 @@ object GrapeRankCommand { * so an odd URL is never culled on a parse quirk — let the WS decide. */ private suspend fun tcpReachable(relay: NormalizedRelayUrl): Boolean = - withContext(Dispatchers.IO) { + withContext(probeDispatcher) { val hostPort = relayHostPort(relay) ?: return@withContext true try { Socket().use { it.connect(InetSocketAddress(hostPort.first, hostPort.second), PROBE_TIMEOUT_MS) } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 12dd36e1a3..db1345f0ea 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -1666,7 +1666,7 @@ class GrapeRankDataCrawler( // Concurrent TCP reachability probes in the background culler. Raw sockets are // cheap and short-lived; the per-relay WS limiter is unaffected (this never // opens a REQ), so this only bounds file descriptors during the cull. - private const val PROBE_CONCURRENCY = 256 + private const val PROBE_CONCURRENCY = 128 // Re-scan interval for the culler when it has probed everything learned so far // and is waiting for new relays to be discovered. From 405f5fd70b5581cd22f254daa8ace431290e4f70 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 02:20:28 +0000 Subject: [PATCH 18/30] refactor(cli): rename `graperank sync` to `graperank crawl` MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The network-only WoT data traversal is a crawl, not a sync — and main now ships negentropy sync (`amy sync`, `graperank update`), so the old verb name was ambiguous. Rename the subcommand and its handler to `crawl`, keeping `sync` as a back-compat alias so existing scripts keep working. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 28 +++++++++++-------- 1 file changed, 16 insertions(+), 12 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 3e53f050d7..8accb7e4c8 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -85,13 +85,14 @@ import kotlin.math.roundToInt * * The crawl and the computation are separable, because the crawl persists every * event it fetches to the store and the score is a pure function over it: - * - `amy graperank sync [OBSERVER]` — network only: crawl the reachable graph's - * kind 3/10000/1984/10002 into the local store. Idempotent and cumulative, so - * run it a few times to make sure everything is loaded. Scores nothing. + * - `amy graperank crawl [OBSERVER]` — network only: crawl the reachable graph's + * kind 3/10000/1984/10002 into the local store (aliased as the former `sync`). + * Idempotent and cumulative, so run it a few times to make sure everything is + * loaded. Scores nothing. * - `amy graperank score [OBSERVER]` — local only: build the graph from the store * and score (same as bare `--offline`). Instant and param-tunable; repeat with * different `--rigor`/`--attenuation`/cutoffs without re-crawling. - * - bare `amy graperank [OBSERVER]` — the convenience combo: sync then score. + * - bare `amy graperank [OBSERVER]` — the convenience combo: crawl then score. * * Sub-verbs complete the NIP-85 provider experience — the discovery layer that * lets clients find and consume those assertions: @@ -189,7 +190,9 @@ object GrapeRankCommand { "register" -> register(dataDir, tail.drop(1).toTypedArray()) "providers" -> providers(dataDir, tail.drop(1).toTypedArray()) "operator" -> operator(dataDir, tail.drop(1).toTypedArray()) - "sync" -> sync(dataDir, tail.drop(1).toTypedArray()) + // `sync` is the pre-rename name kept as a back-compat alias; `crawl` is + // canonical (disambiguates from negentropy `amy sync` / `graperank update`). + "crawl", "sync" -> crawl(dataDir, tail.drop(1).toTypedArray()) "update" -> update(dataDir, tail.drop(1).toTypedArray()) "score" -> run(dataDir, tail.drop(1).toTypedArray(), forceOffline = true) else -> run(dataDir, tail) @@ -421,7 +424,7 @@ object GrapeRankCommand { /** * Configure the outbox-model crawler from the crawl flags on [args] plus the - * account's relay policy. Shared by the bare command and `graperank sync`. + * account's relay policy. Shared by the bare command and `graperank crawl`. * Relay policy — where a stranger's kind:10002 is found (index/discovery * aggregators + general defaults) and best-effort general relays that might * hold content when an outbox is unknown — lives in app code, so the quartz @@ -476,12 +479,13 @@ object GrapeRankCommand { } /** - * `amy graperank sync [OBSERVER]` — network-only WoT data sync. Crawls the - * reachable follow/mute/report graph into the local store (kind 3/10000/1984/ - * 10002) and reports what it loaded, WITHOUT scoring. Idempotent + cumulative: - * run it a few times to make sure everything is loaded, then `graperank score`. + * `amy graperank crawl [OBSERVER]` — network-only WoT data crawl (aliased as the + * former `sync`). Crawls the reachable follow/mute/report graph into the local + * store (kind 3/10000/1984/10002) and reports what it loaded, WITHOUT scoring. + * Idempotent + cumulative: run it a few times to make sure everything is loaded, + * then `graperank score`. */ - private suspend fun sync( + private suspend fun crawl( dataDir: DataDir, rest: Array, ): Int { @@ -574,7 +578,7 @@ object GrapeRankCommand { "relay_lists_in_store" to result.relayListsInStore, "authors_with_outbox" to result.authorsWithOutbox, "relays" to 0, - "note" to "no kind:10002 write relays in the local store — run `graperank sync` first", + "note" to "no kind:10002 write relays in the local store — run `graperank crawl` first", ), ) return 0 From f1b5be9fdfa76a10cccbb9d1bb968f6a488ab443 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 03:02:10 +0000 Subject: [PATCH 19/30] perf(graperank): stop re-discovering outboxes; co-fetch kind:3 on indexers only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Round-8 profiling showed the crawl re-querying the same never-had-a-10002 users' outboxes every round they recirculated — ~144k slow kind:10002 drains (p50 17.4s) against a static discovery set, dragging the round to ~18 users/s. 1. ensureRelayLists guards with `relayListDiscoverySwept`: each user's outbox discovery runs once. The discovery relay set is static, so a second sweep of a user still lacking a 10002 cannot find one the first missed. 2. The discovery REQ to the bounded INDEXER set co-fetches [10002, 3]: the outbox lookup already pays the round-trip and an indexer holding a user's 10002 often holds their kind:3, so we harvest the contact list as a cheap byproduct. The wide "every live relay" completeness sweep stays 10002-ONLY — co-fetching kind:3 across thousands of relays downloaded the same big contact lists repeatedly and inflated the fire-and-forget bgScope sweep the finishing drain waits on (measured +260s at hop-3; the indexer-only co-fetch keeps coverage flat at baseline speed). 3. harvestFromStore folds any already-stored kind:3 into the graph at Phase-A time so Phase B never re-drains a list we hold (also speeds re-runs). Verified same-session hop-3: pre-fix 685s / narrowed 690s / wide-co-fetch 945s, coverage 91.74% across all. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankDataCrawler.kt | 52 +++++++++++++++++-- 1 file changed, 48 insertions(+), 4 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index db1345f0ea..7516c9ba85 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -258,6 +258,13 @@ class GrapeRankDataCrawler( // is the discovered frontier — no separate `discovered` set to keep in sync. val hopOf = hashMapOf(observer to 0) val done = hashSetOf() + + // Users we've already run the outbox-discovery sweep for (indexers + the wide + // live-relay pass in [ensureRelayLists]). The discovery relay set is static, so + // re-asking it for the same never-had-a-10002 user each round it recirculates is + // pure waste — the round-8 profile showed this as ~144k slow kind:10002 drains. + // Single-writer: only the round loop's [ensureRelayLists] touches it. + val relayListDiscoverySwept = hashSetOf() val relaysContacted = hashSetOf() val writeRelayFreq = ConcurrentMap() val liveRelays = ConcurrentSet() @@ -512,6 +519,26 @@ class GrapeRankDataCrawler( return got } + /** + * Mark done any still-pending user whose kind:3 is already in the store — a + * previous round's [ensureRelayLists] co-fetch, a late parked delivery, or a + * prior run's data — folding it into the graph so Phase B never spends an outbox + * drain re-pulling a contact list we already hold. Single-writer: called only + * from the round loop at Phase-A time, before the drain workers start. Returns + * the count newly fed. + */ + suspend fun harvestFromStore(authors: Collection): Int { + var got = 0 + for (pk in authors) { + if (pk in done) continue + val contacts = contactsOf(pk) ?: continue + done += pk + ingest(pk, contacts) + got++ + } + return got + } + /** * Sharded backbone sweep (see SHARD_RELAYS). Splits the missing authors * across the top live relays — one shard per relay, so no relay gets the @@ -602,18 +629,23 @@ class GrapeRankDataCrawler( allLiveRelays: Set, bgScope: CoroutineScope, ) { - val missing = pubkeys.filter { relaysOf(it) == null } + // Only sweep users we have never swept: the discovery relay set is static, + // so a second sweep of a user still lacking a 10002 can't find one we didn't + // already miss. Mark them swept up-front so the wide pass below is gated too. + val missing = pubkeys.filter { relaysOf(it) == null && it !in relayListDiscoverySwept } if (missing.isEmpty()) return + relayListDiscoverySwept.addAll(missing) suspend fun query( authors: List, relays: Set, + kinds: List, ) { if (relays.isEmpty() || authors.isEmpty()) return val filters = relays.associateWith { authors.chunked(AUTHORS_PER_FILTER).map { chunk -> - Filter(kinds = listOf(AdvertisedRelayListEvent.KIND), authors = chunk) + Filter(kinds = kinds, authors = chunk) } } drainGated(filters, null) @@ -625,12 +657,20 @@ class GrapeRankDataCrawler( } else { config.relayListDiscoveryRelays } - query(missing, discovery) + // On the bounded indexer set, co-fetch the contact list in the same REQ: the + // outbox lookup already pays the round-trip and an indexer holding a user's + // 10002 often holds their kind:3, so we harvest it as a cheap byproduct. + query(missing, discovery, listOf(AdvertisedRelayListEvent.KIND, ContactListEvent.KIND)) val stillMissing = missing.filter { relaysOf(it) == null } val wide = allLiveRelays - discovery if (stillMissing.isNotEmpty() && wide.isNotEmpty()) { - bgScope.launch { query(stillMissing, wide) } + // The wide net is EVERY live relay (thousands). Ask it for kind:10002 ONLY — + // co-fetching kind:3 here would download the same big contact lists from + // hundreds of relays and inflate this fire-and-forget bgScope sweep, which + // the finishing drain then waits on (measured +300s at hop-3). The outbox + // this finds routes the user's kind:3 to their own relays in the next round. + bgScope.launch { query(stillMissing, wide, listOf(AdvertisedRelayListEvent.KIND)) } } } @@ -1306,6 +1346,10 @@ class GrapeRankDataCrawler( // the majority cheaply (early rounds no-op until a backbone is learned). val fedBeforeA = contactListsFed shardedSweep(pending) + // Fold any kind:3 already sitting in the store — a prior round's + // ensureRelayLists co-fetch, a late parked delivery, or a previous run — + // so Phase B doesn't re-drain contact lists we already hold. + harvestFromStore(pending) val phaseAMs = roundMark.elapsedNow().inWholeMilliseconds val phaseAFed = contactListsFed - fedBeforeA From b5e0ef53e2c4a07602e4e2acc8a11827ac59c4fe Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 16:23:20 +0000 Subject: [PATCH 20/30] =?UTF-8?q?feat(nip66):=20RelayReachabilityStore=20?= =?UTF-8?q?=E2=80=94=20dead-relay=20cache=20backed=20by=20kind:30166?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A durable, shareable relay-reachability cache backed by the EventStore as NIP-66 kind:30166 Relay Discovery events, so the crawler, the WoT updater, and future runs share liveness knowledge instead of each rediscovering dead relays from an in-memory set wiped at process exit. - 30166 is addressable by its d-tag (relay URL) → one replaceable status slot per (monitor, relay), with created_at giving a free TTL. - Reachable → 30166 with rtt-open; dead → 30166 without (NIP-66 has no explicit offline field; liveness is inferred from a fresh successful open). Live wins over dead within the TTL, so third-party monitors' 30166 can be ingested. - snapshot() loads the fresh set once (not a per-request hot-path query); record() flushes a run's findings. A relay is only skipped for the TTL, never permanently — consistent with the outbox rule that every advertised write relay is tried. Reuses the existing RelayDiscoveryEvent. jvmTest covers record/reload, live-overrides-dead, TTL expiry, and .onion→Tor network tagging. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../reachability/RelayReachabilityStore.kt | 162 ++++++++++++++++++ .../RelayReachabilityStoreTest.kt | 114 ++++++++++++ 2 files changed, 276 insertions(+) create mode 100644 quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt create mode 100644 quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt new file mode 100644 index 0000000000..ee1445779e --- /dev/null +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt @@ -0,0 +1,162 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip66RelayMonitor.reachability + +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl +import com.vitorpamplona.quartz.nip01Core.signers.NostrSigner +import com.vitorpamplona.quartz.nip01Core.store.IEventStore +import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.RelayDiscoveryEvent +import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.networkType +import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.rtt +import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.NetworkType +import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.RttType +import com.vitorpamplona.quartz.utils.TimeUtils + +/** + * A durable, shareable relay-reachability cache backed by an [IEventStore] as + * NIP-66 **kind:30166 Relay Discovery** events — so the crawler, the WoT updater, + * and future runs all read and write the *same* liveness knowledge instead of each + * rediscovering dead relays from an in-memory set that is wiped when the process ends. + * + * ## Why NIP-66 / the event store + * A 30166 event is addressable by its `d`-tag (the normalized relay URL), so the + * store keeps exactly **one replaceable record per (monitor, relay)** — a natural + * per-relay status slot with a `created_at` timestamp that gives us a free TTL. The + * event store gives us persistence, cross-procedure sharing, and interop for free: + * 30166 events published by *other* monitors (nostr.watch et al.) can be ingested to + * seed reachability without probing, and our own records can be published back. + * + * ## How "dead" is represented + * NIP-66 has no explicit offline field; liveness is inferred from a fresh record that + * carries an `rtt-open` (a successful connection). This cache follows that convention: + * - **reachable** → a 30166 **with** `rtt-open`, `created_at` = probe time. + * - **dead** → a 30166 **without** `rtt-open` ("we checked, could not open"), + * `created_at` = probe time. + * + * So a fresh rtt-less record distinguishes *checked-and-dead* from *never-checked* + * (no record). When both a dead and a live record exist within the TTL for the same + * relay, **live wins** — any recent successful open overrides an earlier failure, + * whether the two came from us across time or from two different monitors. + * + * ## Not a replacement for the hot path + * [snapshot] is meant to be loaded ONCE at the start of a run into whatever in-memory + * structure the caller already uses for per-request `isDead` checks; [record] flushes + * a run's findings back at the end. It is deliberately not queried per routing decision. + * + * A relay is only ever skipped for the TTL window, never permanently — consistent with + * the outbox rule that every advertised write relay must be tried: a TTL'd record is + * "skip for now", not "ignore this author's home forever". + * + * ## The signer is a dedicated monitor service identity + * [signer] should be a **machine-level monitor key**, NOT a user/observer account: per + * NIP-66 a monitor is its own pubkey (which also publishes a kind:10166 announcement, + * a kind:0 profile and a kind:10002). Publishing these under the observer's key would + * conflate the WoT identity with a relay-monitoring service. [snapshot] still honours + * records from ANY author (so third-party monitors can be ingested); only [record] + * writes under this monitor key. + */ +class RelayReachabilityStore( + private val store: IEventStore, + private val signer: NostrSigner, + private val ttlSeconds: Long = DEFAULT_TTL_SECONDS, +) { + /** + * An in-memory view of the fresh (within-TTL) reachability records. [dead] holds + * relays proven unreachable and not since seen live; [live] holds relays with a + * recent successful open. A relay absent from both is simply unknown — re-probe it. + */ + class Snapshot( + val dead: Set, + val live: Set, + ) { + fun isKnownDead(relay: NormalizedRelayUrl) = relay in dead + + val size: Int get() = dead.size + live.size + } + + /** + * Load every 30166 record fresher than [ttlSeconds] and fold it into a [Snapshot]. + * Records from any monitor are honoured (live-wins), so ingesting third-party + * monitors' 30166 into [store] transparently improves the result. + */ + suspend fun snapshot(now: Long = TimeUtils.now()): Snapshot { + val since = now - ttlSeconds + val events = + store.query( + Filter(kinds = listOf(RelayDiscoveryEvent.KIND), since = since), + ) + val live = HashSet() + val dead = HashSet() + for (ev in events) { + val relay = ev.relay() ?: continue + if (ev.rttOpen() != null) live.add(relay) else dead.add(relay) + } + // A recent successful open (from us later, or from another monitor) overrides + // an earlier dead mark for the same relay. + dead.removeAll(live) + return Snapshot(dead, live) + } + + /** + * Persist a run's reachability findings as 30166 events: each [reachable] relay as + * a record WITH `rtt-open`, each [dead] relay (that is not also reachable) as one + * WITHOUT. Signed by [signer] and inserted into [store]; being addressable, each + * replaces this monitor's prior record for that relay, so the store stays bounded + * at roughly the number of distinct relays. + */ + suspend fun record( + reachable: Set, + dead: Set, + now: Long = TimeUtils.now(), + rttOpenMs: Long = 0, + ) { + for (relay in reachable) writeOne(relay, up = true, now, rttOpenMs) + for (relay in dead) if (relay !in reachable) writeOne(relay, up = false, now, rttOpenMs) + } + + private suspend fun writeOne( + relay: NormalizedRelayUrl, + up: Boolean, + now: Long, + rttOpenMs: Long, + ) { + val template = + RelayDiscoveryEvent.build(relay, createdAt = now) { + networkType(networkTypeOf(relay)) + if (up) rtt(RttType.OPEN, rttOpenMs) + } + store.insert(signer.sign(template)) + } + + companion object { + /** Default freshness window: a relay's status is trusted for a day, then re-probed. */ + const val DEFAULT_TTL_SECONDS = 24L * 60 * 60 + + /** NIP-66 `n` network type inferred from the URL, so a `.onion`/i2p relay is tagged correctly. */ + fun networkTypeOf(relay: NormalizedRelayUrl): NetworkType = + when { + relay.url.contains(".onion") -> NetworkType.TOR + relay.url.contains(".i2p") -> NetworkType.I2P + else -> NetworkType.CLEARNET + } + } +} diff --git a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt new file mode 100644 index 0000000000..63aa6815c7 --- /dev/null +++ b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt @@ -0,0 +1,114 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip66RelayMonitor.reachability + +import com.vitorpamplona.quartz.nip01Core.crypto.KeyPair +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer +import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerInternal +import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.Secp256k1Instance +import kotlinx.coroutines.runBlocking +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertFalse +import kotlin.test.assertTrue + +class RelayReachabilityStoreTest { + private fun store() = + EventStore( + dbName = null, + indexStrategy = DefaultIndexingStrategy(), + ) + + private fun cache(store: EventStore) = + RelayReachabilityStore( + store = store, + signer = NostrSignerInternal(KeyPair()), + ttlSeconds = 3600, + ) + + private val live1 = RelayUrlNormalizer.normalize("wss://alive.example.com") + private val live2 = RelayUrlNormalizer.normalize("wss://also-alive.example.com") + private val dead1 = RelayUrlNormalizer.normalize("wss://dead.example.com") + private val dead2 = RelayUrlNormalizer.normalize("wss://gone.example.com") + private val onion = RelayUrlNormalizer.normalize("wss://abc.onion") + + @Test + fun recordsAndReloadsReachability() = + runBlocking { + Secp256k1Instance + val store = store() + val cache = cache(store) + val now = 1_000_000L + + cache.record(reachable = setOf(live1, live2), dead = setOf(dead1, dead2), now = now) + + val snap = cache.snapshot(now = now) + assertEquals(setOf(live1, live2), snap.live) + assertEquals(setOf(dead1, dead2), snap.dead) + assertTrue(snap.isKnownDead(dead1)) + assertFalse(snap.isKnownDead(live1)) + } + + @Test + fun aFreshSuccessfulOpenOverridesAnEarlierDeadMark() = + runBlocking { + Secp256k1Instance + val store = store() + val cache = cache(store) + + // Marked dead first, then seen alive a second later (addressable replace). + cache.record(reachable = emptySet(), dead = setOf(dead1), now = 1_000L) + cache.record(reachable = setOf(dead1), dead = emptySet(), now = 1_001L) + + val snap = cache.snapshot(now = 1_001L) + assertTrue(dead1 in snap.live) + assertFalse(snap.isKnownDead(dead1)) + } + + @Test + fun recordsOlderThanTheTtlAreIgnored() = + runBlocking { + Secp256k1Instance + val store = store() + val cache = cache(store) // ttl = 3600s + + cache.record(reachable = emptySet(), dead = setOf(dead1), now = 1_000L) + + // "now" is well past the 1h TTL from when dead1 was recorded. + val snap = cache.snapshot(now = 1_000L + 3601L) + assertFalse(snap.isKnownDead(dead1)) + assertEquals(0, snap.size) + } + + @Test + fun onionRelayIsTaggedTorNetwork() { + assertEquals( + com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.NetworkType.TOR, + RelayReachabilityStore.networkTypeOf(onion), + ) + assertEquals( + com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.NetworkType.CLEARNET, + RelayReachabilityStore.networkTypeOf(live1), + ) + } +} From 8ced11b6b0fcc8c3f93c25ad3181cad76d2b3127 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 16:30:45 +0000 Subject: [PATCH 21/30] feat(graperank): share dead-relay knowledge via the NIP-66 reachability cache MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Wire RelayReachabilityStore into the crawler and the WoT updater so liveness is shared across procedures and runs instead of each rediscovering dead relays. - OperatorKeys.monitorKey(): a dedicated machine monitor identity derived from the operator master (domain "relay-monitor:"), independent of any account — the 30166 records are published under this, not the observer key. - Context.reachability: a RelayReachabilityStore over the shared store, signed by the monitor key. - Crawler: Config.knownDeadRelays seeds deadRelays before the run; Stats now returns the final dead/live sets. GrapeRankCommand seeds from snapshot().dead and flushes the crawl's verdicts back via reachability.record(). - Updater: Config.knownDead skips proven-dead relays from the reconcile plan — a dead relay cannot serve its authors, so reconciling it only burns a timeout. Live author-advertised relays are always synced. All behind --no-reachability-cache. TTL'd (24h), so a recovered relay is retried once its record ages out — a "skip for now", never a permanent ignore, keeping the outbox rule that every live advertised relay is tried. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../com/vitorpamplona/amethyst/cli/Context.kt | 16 +++++++++ .../amethyst/cli/OperatorKeys.kt | 19 ++++++++++ .../amethyst/cli/commands/GrapeRankCommand.kt | 36 +++++++++++++++++++ .../graperank/GrapeRankDataCrawler.kt | 23 +++++++++++- .../graperank/GrapeRankUpdater.kt | 9 +++++ 5 files changed, 102 insertions(+), 1 deletion(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt index 4643c406ea..e74998b91d 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt @@ -71,6 +71,7 @@ import com.vitorpamplona.quartz.nip60Cashu.wallet.CashuWalletEvent import com.vitorpamplona.quartz.nip61Nutzaps.info.NutzapInfoEvent import com.vitorpamplona.quartz.nip61Nutzaps.nutzap.NutzapEvent import com.vitorpamplona.quartz.nip65RelayList.AdvertisedRelayListEvent +import com.vitorpamplona.quartz.nip66RelayMonitor.reachability.RelayReachabilityStore import com.vitorpamplona.quartz.nip87Ecash.recommendation.MintRecommendationEvent import com.vitorpamplona.quartz.utils.SeenIds import kotlinx.coroutines.CompletableDeferred @@ -258,6 +259,21 @@ class Context( private val storeDelegate: Lazy = lazy { StoreFactory.open(dataDir) } val store: IEventStore by storeDelegate + /** + * Shared relay-reachability cache (NIP-66 kind:30166 records in [store]), signed by + * the machine's dedicated monitor key — derived from the operator master, NOT the + * account (see [OperatorKeys.monitorKey]). The crawler and the WoT updater read its + * dead set to skip proven-dead relays and write their findings back, so liveness + * knowledge is shared across procedures and runs instead of rediscovered each time. + * Lazy so a run that never touches relays doesn't materialize the operator master. + */ + val reachability: RelayReachabilityStore by lazy { + RelayReachabilityStore( + store = store, + signer = NostrSignerInternal(dataDir.operatorKeys().monitorKey()), + ) + } + /** Fully-wired manager. Call [prepare] once before use to load persisted state. */ val marmot: MarmotManager by lazy { MarmotManager(signer, mlsStore, messageStore, keyPackageStore) } diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/OperatorKeys.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/OperatorKeys.kt index 71a7dd6d68..68c095b234 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/OperatorKeys.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/OperatorKeys.kt @@ -123,6 +123,24 @@ class OperatorKeys( } } + /** + * The machine's dedicated NIP-66 relay-monitor identity, derived once from the + * operator master (independent of any amy account). Unlike [serviceKey] this is + * NOT per-observer — the machine publishes relay-reachability (kind:30166) under a + * single, stable monitor pubkey, so a re-probe *replaces* the prior 30166 for a + * relay instead of orphaning it. Re-derivable from the one master seed alone. + */ + fun monitorKey(): KeyPair { + val master = masterPriv() + var counter = 0 + while (true) { + val material = master + "$MONITOR_LABEL$counter".encodeToByteArray() + val kp = runCatching { KeyPair(privKey = sha256(material)) }.getOrNull() + if (kp?.privKey != null) return kp + counter++ + } + } + private fun recordProvider( observerHex: HexKey, providerPubKey: HexKey, @@ -152,5 +170,6 @@ class OperatorKeys( private const val DIR_NAME = "operator" private const val CONFIG_NAME = "operator.json" private const val DERIVATION_LABEL = "graperank-provider:" + private const val MONITOR_LABEL = "relay-monitor:" } } diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 8accb7e4c8..79f86d0081 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -257,6 +257,7 @@ object GrapeRankCommand { val stats = newCrawler(ctx, args).crawl(observer, builder) crawlStats = stats contactListsFed = stats.contactListsFed + flushReachability(ctx, args, stats) reportRelayFeedback(ctx) } else { // Offline: stream contact lists from the local store into the graph. @@ -440,6 +441,11 @@ object GrapeRankCommand { // Aggregator kind:3 recovery for stragglers is on by default; --no-aggregators // disables it for A/B comparison. val aggregators = if (args.bool("no-aggregators")) emptySet() else CONTENT_AGGREGATOR_RELAYS + // Seed the crawl with relays a prior run/monitor proved dead within the cache's + // TTL, so the WS path never re-pays their connect timeouts (--no-reachability-cache + // to skip). The crawl's own final live/dead set is flushed back by the caller. + val knownDead = + if (args.bool("no-reachability-cache")) emptySet() else ctx.reachability.snapshot().dead return GrapeRankDataCrawler( client = ctx.client, store = ctx.store, @@ -447,6 +453,7 @@ object GrapeRankCommand { config = GrapeRankDataCrawler.Config( relayListDiscoveryRelays = discoveryRelays, + knownDeadRelays = knownDead, contentFallbackRelays = contentFallback, contentAggregatorRelays = aggregators, maxRounds = args.intFlag("max-rounds", Int.MAX_VALUE), @@ -478,6 +485,26 @@ object GrapeRankCommand { } } + /** + * Flush the crawl's final live/dead relay verdicts into the shared reachability + * cache (NIP-66 kind:30166) so the next crawl and the WoT updater start warm and + * skip proven-dead relays. Best-effort and behind `--no-reachability-cache`: a + * cache write must never fail the crawl it is summarizing. + */ + private suspend fun flushReachability( + ctx: Context, + args: Args, + stats: GrapeRankDataCrawler.Stats, + ) { + if (args.bool("no-reachability-cache")) return + runCatching { + ctx.reachability.record(reachable = stats.liveRelays, dead = stats.deadRelays) + System.err.println( + "[graperank] reachability cache: recorded ${stats.liveRelays.size} live, ${stats.deadRelays.size} dead", + ) + }.onFailure { System.err.println("[graperank] reachability cache flush failed: ${it.message}") } + } + /** * `amy graperank crawl [OBSERVER]` — network-only WoT data crawl (aliased as the * former `sync`). Crawls the reachable follow/mute/report graph into the local @@ -497,6 +524,7 @@ object GrapeRankCommand { // Persist-only crawl: no in-memory graph (null builder); every event // still lands in the store for a later `score`. val stats = newCrawler(ctx, args).crawl(observer, null) + flushReachability(ctx, args, stats) reportRelayFeedback(ctx) Output.emit( linkedMapOf( @@ -553,6 +581,13 @@ object GrapeRankCommand { Context.openOrAnonymous(dataDir).use { ctx -> ctx.prepare() + // Skip relays a crawl/monitor proved dead within the cache's TTL — a dead + // relay cannot serve its authors, so reconciling it only burns a timeout. + // Live author-advertised relays are always synced (--no-reachability-cache + // to reconcile every relay regardless). + val knownDead = + if (args.bool("no-reachability-cache")) emptySet() else ctx.reachability.snapshot().dead + val updater = GrapeRankUpdater( client = ctx.client, @@ -566,6 +601,7 @@ object GrapeRankCommand { authorChunk = args.intFlag("author-chunk", 500), minAuthors = args.intFlag("min-authors", 1), idleTimeoutMs = args.longFlag("timeout", 30L) * 1000, + knownDead = knownDead, ), log = { System.err.println(it) }, ) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt index 7516c9ba85..1e3606c684 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt @@ -199,6 +199,14 @@ class GrapeRankDataCrawler( * no socket, no wasted connect attempt. */ val torEnabled: Boolean = false, + /** + * Relays a prior run (or another monitor) proved unreachable within the + * reachability cache's TTL — seeded into [deadRelays] before the crawl starts + * so we don't re-pay their connect timeouts. Per-URL (not per-authority): a + * TTL'd "skip for now", re-probed once the record ages out, so it never + * permanently ignores an author's advertised home. See RelayReachabilityStore. + */ + val knownDeadRelays: Set = emptySet(), ) /** What the crawl fetched — the counters the caller reports and the graph is built from. */ @@ -221,6 +229,13 @@ class GrapeRankDataCrawler( val insertMs: Long, /** Verified events handed to the store (duplicates included — the write path dedups). */ val eventsStored: Long, + /** + * Relays proven unreachable this run (connect-establishment failures / seeded + * known-dead), and relays that served ≥1 event. The caller flushes these into + * the reachability cache (kind:30166) so the next run and the updater start warm. + */ + val deadRelays: Set, + val liveRelays: Set, ) /** @@ -277,7 +292,11 @@ class GrapeRankDataCrawler( // Concurrent: touched by more than one of producer/consumer/drain-workers. val relayHints = ConcurrentMap>() val attempts = ConcurrentMap() - val deadRelays = ConcurrentSet() + + // Seeded from the reachability cache (relays proven dead within its TTL) so the + // WS path never re-pays their connect timeouts; the crawl still adds/removes + // more as it goes and flushes the union back at the end. + val deadRelays = ConcurrentSet().apply { config.knownDeadRelays.forEach { add(it) } } // Unproductive-TIMEOUT strikes, keyed by relay AUTHORITY (host[:port]), not the // full URL. [classifyDrainFailure] treats a READ timeout or a park idle-cut — the @@ -1512,6 +1531,8 @@ class GrapeRankDataCrawler( verifyMs = verifyMs, insertMs = insertMs, eventsStored = stored, + deadRelays = deadRelays.snapshot(), + liveRelays = liveRelays.snapshot(), ) } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt index 5d6d1cc85a..750b2be58e 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt @@ -97,6 +97,14 @@ class GrapeRankUpdater( val minAuthors: Int = 1, val idleTimeoutMs: Long = 30_000L, val publishTimeoutSecs: Long = 15, + /** + * Relays proven unreachable within the reachability cache's TTL (kind:30166). + * Skipped from the reconcile plan so we don't burn a connect timeout per dead + * relay — the crawl already found them dead, and a dead relay cannot serve its + * authors anyway. TTL'd, so a recovered relay is retried once the record ages + * out; this never drops a *live* author-advertised relay. See RelayReachabilityStore. + */ + val knownDead: Set = emptySet(), ) { /** Project the shared engine knobs onto a [NegentropyStoreSync.Config]. */ internal fun toEngineConfig() = @@ -170,6 +178,7 @@ class GrapeRankUpdater( val plan = groups.entries .filter { it.value.size >= config.minAuthors } + .filterNot { it.key in config.knownDead } .sortedByDescending { it.value.size } .map { it.key to it.value } From 99e9708a3b56d194b78e47dab58315ad3b2fc791 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 16:40:54 +0000 Subject: [PATCH 22/30] fix(graperank): flush only observed relays; rename to GrapeRankCrawler MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two changes: 1. The reachability flush re-wrote the SEEDED known-dead relays with a fresh created_at every run, refreshing their TTL without a re-probe — so a relay marked dead once (and thereafter skipped, never re-dialed) would stay blacklisted forever as long as crawls kept running, defeating the TTL's re-probe. Stats.deadRelays now reports only relays actually dialed this run (deadRelays - knownDeadRelays); seeded records keep their original timestamp and age out on schedule so the next run re-probes them. 2. Rename GrapeRankDataCrawler -> GrapeRankCrawler (file + all references). Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 12 ++++++------ ...eRankDataCrawler.kt => GrapeRankCrawler.kt} | 18 ++++++++++++------ .../graperank/GrapeRankPublisher.kt | 2 +- .../experimental/graperank/GrapeRankUpdater.kt | 2 +- .../graperank/GrapeRankAuthorityTest.kt | 4 ++-- 5 files changed, 22 insertions(+), 16 deletions(-) rename quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/{GrapeRankDataCrawler.kt => GrapeRankCrawler.kt} (99%) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 79f86d0081..e9a7177e36 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -27,7 +27,7 @@ import com.vitorpamplona.amethyst.cli.Output import com.vitorpamplona.amethyst.commons.defaults.Constants import com.vitorpamplona.amethyst.commons.defaults.DefaultIndexerRelayList import com.vitorpamplona.quartz.experimental.graperank.GrapeRank -import com.vitorpamplona.quartz.experimental.graperank.GrapeRankDataCrawler +import com.vitorpamplona.quartz.experimental.graperank.GrapeRankCrawler import com.vitorpamplona.quartz.experimental.graperank.GrapeRankParams import com.vitorpamplona.quartz.experimental.graperank.GrapeRankPublisher import com.vitorpamplona.quartz.experimental.graperank.GrapeRankUpdater @@ -251,7 +251,7 @@ object GrapeRankCommand { // Crawl telemetry (online path only): rounds, relays contacted, the // per-hop histogram, and the network-bound download time that dominates a // from-scratch run. Null on the offline path. - var crawlStats: GrapeRankDataCrawler.Stats? = null + var crawlStats: GrapeRankCrawler.Stats? = null if (!offline) { val stats = newCrawler(ctx, args).crawl(observer, builder) @@ -434,7 +434,7 @@ object GrapeRankCommand { private suspend fun newCrawler( ctx: Context, args: Args, - ): GrapeRankDataCrawler { + ): GrapeRankCrawler { val discoveryRelays = ctx.bootstrapRelays() + Constants.eventFinderRelays + DefaultIndexerRelayList + EXTRA_DISCOVERY_RELAYS val contentFallback = ctx.bootstrapRelays() + Constants.eventFinderRelays @@ -446,12 +446,12 @@ object GrapeRankCommand { // to skip). The crawl's own final live/dead set is flushed back by the caller. val knownDead = if (args.bool("no-reachability-cache")) emptySet() else ctx.reachability.snapshot().dead - return GrapeRankDataCrawler( + return GrapeRankCrawler( client = ctx.client, store = ctx.store, limiter = ctx.relayLimiter, config = - GrapeRankDataCrawler.Config( + GrapeRankCrawler.Config( relayListDiscoveryRelays = discoveryRelays, knownDeadRelays = knownDead, contentFallbackRelays = contentFallback, @@ -494,7 +494,7 @@ object GrapeRankCommand { private suspend fun flushReachability( ctx: Context, args: Args, - stats: GrapeRankDataCrawler.Stats, + stats: GrapeRankCrawler.Stats, ) { if (args.bool("no-reachability-cache")) return runCatching { diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt similarity index 99% rename from quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt rename to quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index 1e3606c684..00b24cc142 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankDataCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -90,7 +90,7 @@ import kotlin.time.TimeSource * is emitted through [log]; a headless caller routes it to stderr, a UI ignores it. */ @OptIn(ExperimentalAtomicApi::class) -class GrapeRankDataCrawler( +class GrapeRankCrawler( private val client: NostrClient, private val store: IEventStore, private val limiter: AdaptiveRelayLimiter, @@ -230,9 +230,13 @@ class GrapeRankDataCrawler( /** Verified events handed to the store (duplicates included — the write path dedups). */ val eventsStored: Long, /** - * Relays proven unreachable this run (connect-establishment failures / seeded - * known-dead), and relays that served ≥1 event. The caller flushes these into - * the reachability cache (kind:30166) so the next run and the updater start warm. + * Relays this run actually OBSERVED, for the caller to flush into the + * reachability cache (kind:30166): [deadRelays] = relays newly proven + * unreachable this run (a connect-establishment failure we paid), [liveRelays] + * = relays that served ≥1 event. Seeded known-dead relays (skipped, never + * dialed) are deliberately EXCLUDED from [deadRelays] — re-writing them would + * refresh their TTL without a re-probe and blacklist a recovered relay forever; + * their original record must age out so the next run re-probes them. */ val deadRelays: Set, val liveRelays: Set, @@ -896,7 +900,7 @@ class GrapeRankDataCrawler( val ok = event.verify() verifyNanos.addAndFetch(vMark.elapsedNow().inWholeNanoseconds) if (!ok) { - Log.w("GrapeRankDataCrawler") { "dropped event ${event.id.take(8)} kind=${event.kind} — bad signature" } + Log.w("GrapeRankCrawler") { "dropped event ${event.id.take(8)} kind=${event.kind} — bad signature" } continue } if (!seenIds.add(event.id)) continue // lost the race to a mirror; it stores it @@ -1531,7 +1535,9 @@ class GrapeRankDataCrawler( verifyMs = verifyMs, insertMs = insertMs, eventsStored = stored, - deadRelays = deadRelays.snapshot(), + // Only relays we actually dialed this run — exclude the seeded + // known-dead (skipped, not re-probed) so their original TTL stands. + deadRelays = deadRelays.snapshot() - config.knownDeadRelays, liveRelays = liveRelays.snapshot(), ) } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankPublisher.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankPublisher.kt index cd03bd7d04..1ccc26cea2 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankPublisher.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankPublisher.kt @@ -47,7 +47,7 @@ import kotlinx.coroutines.coroutineScope * (it fell below the caller's cutoff, or dropped out of the graph) with a NIP-09 * kind:5 deletion, batched so the frame stays under the ~64KB event cap. * - * Transport-agnostic like [GrapeRankDataCrawler]: it reads prior cards from an + * Transport-agnostic like [GrapeRankCrawler]: it reads prior cards from an * [IEventStore] and emits through an injected [publish] function (event + relays → * per-relay ack), so the store/relay wiring stays in the application while the * reconcile + card-construction logic is reusable (e.g. by the Android app). diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt index 750b2be58e..e84778f4f5 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankUpdater.kt @@ -38,7 +38,7 @@ import com.vitorpamplona.quartz.nip65RelayList.AdvertisedRelayListEvent * (kind:10002), and reports (kind:1984) of every author already known to the * local [store]. * - * Where [GrapeRankDataCrawler] discovers the graph by walking follows outward from + * Where [GrapeRankCrawler] discovers the graph by walking follows outward from * an observer, this refreshes what is *already* known: it reads every kind:10002 in * the store, inverts them into a `write-relay -> authors` map (the outbox model — an * author's events live on the relays they write to), fans those into one filter per diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt index bd83401d39..e43aaf0e02 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankAuthorityTest.kt @@ -25,13 +25,13 @@ import kotlin.test.assertEquals import kotlin.test.assertNotEquals /** - * [GrapeRankDataCrawler.authorityOf] is the key the crawl's timeout-eviction counts + * [GrapeRankCrawler.authorityOf] is the key the crawl's timeout-eviction counts * on. It must collapse the many per-user path URLs the outbox model mints for one * server into a single host, WITHOUT folding a distinct sibling host (e.g. a * `filter.` subdomain) into its parent. */ class GrapeRankAuthorityTest { - private fun auth(url: String) = GrapeRankDataCrawler.authorityOf(url) + private fun auth(url: String) = GrapeRankCrawler.authorityOf(url) @Test fun bareHostIsItsOwnAuthority() { From fdd0788e7f01d3738e514328545889481a6a3f9a Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 16:51:28 +0000 Subject: [PATCH 23/30] feat(graperank): saturation + latency instrumentation (--diagnose) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Answers "are we resource-bound or waiting on relays" without a profiler: - progress ticker gains "Nw/CAPw" (drain workers busy vs drainConcurrency) and "N rl" (rate-limit responses so far) — a rarely-full pool means the producer or the relays are the limit, not concurrency; a climbing rl count is the external ceiling that made concurrency 60 backfire. - crawl-end "latency breakdown": splits each drain's wall into time-to-first-event vs EOSE-wait-AFTER-the-relay's-last-event, and reports the % of drain wall spent waiting for EOSE after the relay was already done, how many drains blew the fast window and parked, and total rate-limit hits. A high EOSE-wait % is the direct case for a shorter/adaptive fast window over more concurrency. All gated on config.diagnose; zero cost on a normal run. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankCrawler.kt | 85 +++++++++++++++++-- 1 file changed, 78 insertions(+), 7 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index 00b24cc142..1b2d8e706f 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -351,6 +351,24 @@ class GrapeRankCrawler( // it finishes (or its park window elapses). val parkedInFlight = AtomicLong(0) + // ── Saturation / latency instrumentation (diagnose only) ───────────────────── + // Are we resource-bound or waiting-on-relays? These answer it without a profiler. + // activeWorkers: Phase-B drain workers busy right now (vs drainConcurrency) — if + // rarely full, adding workers won't help; the producer/relays are the limit. + // throttled: drains that got a relay rate-limit (429/too-many-*) — the EXTERNAL + // ceiling; if it climbs when we push harder, more concurrency backfires. + // The latency sums split a drain's wall time into time-to-first-event vs the + // EOSE-wait AFTER the relay's last event (pure waiting on a done-but-slow relay) + // — a large eose-wait fraction is the case for a shorter/adaptive fast window. + // burnedFastWindow: drains that blew timeoutMs and had to park. + val activeWorkers = AtomicLong(0) + val throttled = AtomicLong(0) + val firstEventSumMs = AtomicLong(0) + val eoseWaitSumMs = AtomicLong(0) + val drainWallSumMs = AtomicLong(0) + val drainSamples = AtomicLong(0) + val burnedFastWindow = AtomicLong(0) + // Background scope owning the parked subscriptions (and Tier-2 relay-list // sweeps). Set in [run]; cancelled once the crawl converges. var bgScope: CoroutineScope? = null @@ -1002,9 +1020,17 @@ class GrapeRankCrawler( val pct = (100L * roundDone / progTarget).coerceIn(0, 100) val remaining = (progTarget - roundDone).coerceAtLeast(0) val eta = if (rate > 0) etaFmt(remaining / rate) else "…" + // Saturation tail: workers busy / cap, and rate-limit hits so far — is + // the pool full (raise concurrency) or starved (producer/relays bound)? + val sat = + if (config.diagnose) { + " · ${activeWorkers.load()}/${config.drainConcurrency}w · ${throttled.load()} rl" + } else { + "" + } log( "[graperank] round $progRound · ${human(roundDone.toLong())}/${human(progTarget.toLong())} ($pct%)" + - " · $rate/s · ~$eta · ${human(events)} ev · $parked slow · ${deadRelays.size()} dead", + " · $rate/s · ~$eta · ${human(events)} ev · $parked slow · ${deadRelays.size()} dead$sat", ) } } @@ -1104,6 +1130,12 @@ class GrapeRankCrawler( classifyDrainFailure(reason)?.let { kind -> into[relay] = kind } } + // A relay asking us to slow down — the external concurrency ceiling. + fun isRateLimit(reason: String): Boolean { + val m = reason.lowercase() + return "429" in m || "too many" in m || "rate" in m || "throttl" in m + } + fun logSlow( relay: NormalizedRelayUrl, reason: String, @@ -1132,6 +1164,12 @@ class GrapeRankCrawler( // this (conflated, so bursts collapse to one) and resets the park // window, so a relay actively streaming is never cut mid-flight. val activity = Channel(Channel.CONFLATED) + // Latency breakdown: elapsed-since-[mark] of the first and last + // event, so the EOSE-wait AFTER the relay's last event (pure + // waiting on a done-but-slow relay) is separable from fetch time. + val mark = TimeSource.Monotonic.markNow() + val firstEvt = AtomicLong(-1) + val lastEvt = AtomicLong(-1) val listener = object : SubscriptionListener { override fun onEvent( @@ -1142,6 +1180,9 @@ class GrapeRankCrawler( ) { unitEvents.trySend(relay to event) activity.trySend(Unit) + val e = mark.elapsedNow().inWholeMilliseconds + firstEvt.compareAndSet(-1, e) + lastEvt.store(e) } override fun onEose( @@ -1168,13 +1209,22 @@ class GrapeRankCrawler( } } client.subscribe(subId, mapOf(subRelay to groupFilters), listener) - val mark = TimeSource.Monotonic.markNow() val reason = withTimeoutOrNull(config.timeoutMs) { done.await() } if (reason != null) { // Terminal within the fast window — resolve this round. val elapsedMs = mark.elapsedNow().inWholeMilliseconds if (reason != "eose") notAnswered.add(subRelay) classify(reason, subRelay, failures) + if (isRateLimit(reason)) throttled.addAndFetch(1) + // Split the wall time: fetch (to first event) vs EOSE-wait + // (after the last event) — only for drains that got events. + val firstE = firstEvt.load() + if (firstE >= 0) { + firstEventSumMs.addAndFetch(firstE) + eoseWaitSumMs.addAndFetch((elapsedMs - lastEvt.load()).coerceAtLeast(0)) + drainWallSumMs.addAndFetch(elapsedMs) + drainSamples.addAndFetch(1) + } if (elapsedMs > SLOW_DRAIN_LOG_MS) logSlow(subRelay, reason, elapsedMs, groupFilters) unitEvents.close() client.unsubscribe(subId) @@ -1195,6 +1245,7 @@ class GrapeRankCrawler( persisted } else { // Still streaming — hand off and let the round move on. + burnedFastWindow.addAndFetch(1) notAnswered.add(subRelay) timedOut.add(subRelay) val scope = bgScope @@ -1414,11 +1465,16 @@ class GrapeRankCrawler( List(config.drainConcurrency) { launch { for ((batch, filters) in routed) { - val dead = HashMap() - val answered = HashSet() - val events = drainGated(filters, dead, answered) - recordDead(dead) - drainedOut.send(DrainedBatch(batch, filters, answered, events)) + activeWorkers.addAndFetch(1) + try { + val dead = HashMap() + val answered = HashSet() + val events = drainGated(filters, dead, answered) + recordDead(dead) + drainedOut.send(DrainedBatch(batch, filters, answered, events)) + } finally { + activeWorkers.addAndFetch(-1) + } } } } @@ -1525,6 +1581,21 @@ class GrapeRankCrawler( "[graperank] write path: $stored events stored, verify ${verifyMs}ms + insert ${insertMs}ms " + "(summed across all drains, batch=${config.insertBatchSize})", ) + if (config.diagnose) { + val n = drainSamples.load().coerceAtLeast(1) + val wall = drainWallSumMs.load().coerceAtLeast(1) + val meanFirst = firstEventSumMs.load() / n + val meanEose = eoseWaitSumMs.load() / n + val eosePct = 100 * eoseWaitSumMs.load() / wall + log( + "[graperank] latency breakdown (drains with events, n=${drainSamples.load()}): " + + "mean time-to-first-event ${meanFirst}ms, mean EOSE-wait-after-last-event ${meanEose}ms " + + "($eosePct% of drain wall spent waiting for EOSE after the relay's last event); " + + "${burnedFastWindow.load()} drains blew the ${config.timeoutMs}ms fast window and parked; " + + "${throttled.load()} rate-limit responses. " + + "High EOSE-wait % → a shorter/adaptive fast window is the lever, not more concurrency.", + ) + } return Stats( rounds = rounds, contactListsFed = contactListsFed, From 4c20522d781c5f9366c264b4aea703bcf2a00261 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 17:20:55 +0000 Subject: [PATCH 24/30] feat(graperank): frame-dispatch-lag metric to attribute EOSE-wait MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds FrameDispatchStats: the lag between a relay frame arriving on the OkHttp reader thread and our per-connection consumer coroutine (on shared Dispatchers.IO) pulling it off the channel — pure our-side pipeline delay, relay send-timing excluded. BasicOkHttpWebSocket stamps arrival before enqueue and records the lag on dequeue; the crawler resets it at start and dumps it in the --diagnose summary. Answers whether a drain's 5s gap between the relay's last event and its EOSE is the relay being slow to SEND eose (low dispatch-lag) or our IO pipeline backing up so the already-arrived eose frame sits queued (high dispatch-lag). Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankCrawler.kt | 6 ++ .../client/accessories/FrameDispatchStats.kt | 84 +++++++++++++++++++ .../sockets/okhttp/BasicOkHttpWebSocket.kt | 17 +++- 3 files changed, 103 insertions(+), 4 deletions(-) create mode 100644 quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index 1b2d8e706f..62e020a735 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -26,6 +26,7 @@ import com.vitorpamplona.quartz.nip01Core.crypto.verify import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.AdaptiveRelayLimiter import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.DrainFailure +import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.FrameDispatchStats import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.classifyDrainFailure import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener @@ -256,6 +257,7 @@ class GrapeRankCrawler( verifyNanos.store(0) insertNanos.store(0) eventsStored.store(0) + if (config.diagnose) FrameDispatchStats.reset() return CrawlRun(observer, builder).run() } @@ -1595,6 +1597,10 @@ class GrapeRankCrawler( "${throttled.load()} rate-limit responses. " + "High EOSE-wait % → a shorter/adaptive fast window is the lever, not more concurrency.", ) + // Attribution: is the EOSE-wait the relay (slow to SEND eose) or us (the + // eose frame arrived but sat in our IO pipeline)? Low dispatch-lag + high + // EOSE-wait ⇒ relay; high dispatch-lag ⇒ our Dispatchers.IO is backed up. + log("[graperank] frame dispatch (our-side pipeline lag): ${FrameDispatchStats.snapshot()}") } return Stats( rounds = rounds, diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt new file mode 100644 index 0000000000..ffd5eaa614 --- /dev/null +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt @@ -0,0 +1,84 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.client.accessories + +import kotlin.concurrent.atomics.AtomicLong +import kotlin.concurrent.atomics.ExperimentalAtomicApi + +/** + * Diagnostic: the lag between a raw relay frame arriving on the socket (the OkHttp + * reader thread's `onMessage`) and our per-connection consumer coroutine actually + * pulling it off the channel to decode + dispatch. Because that consumer runs on the + * shared `Dispatchers.IO`, this lag is precisely OUR-side pipeline delay — channel + * queue wait + coroutine reschedule + time spent decoding earlier frames — with the + * relay's own send timing excluded (the reader thread enqueues the instant bytes land). + * + * It exists to answer one question the [GrapeRankCrawler]'s EOSE-wait metric cannot on + * its own: when a drain shows a 5-second gap between the relay's last event and its + * EOSE, is the relay slow to send EOSE (low dispatch lag) or is our IO pipeline backed + * up so the already-arrived EOSE frame sits queued (high dispatch lag)? Process-global + * and opt-in: a caller [reset]s before a run and reads [snapshot] after. Off unless + * something records into it, so zero cost on normal paths. + */ +@OptIn(ExperimentalAtomicApi::class) +object FrameDispatchStats { + private val count = AtomicLong(0) + private val sumMs = AtomicLong(0) + private val maxMs = AtomicLong(0) + private val over1s = AtomicLong(0) + + fun record(lagMs: Long) { + count.addAndFetch(1) + sumMs.addAndFetch(lagMs) + if (lagMs >= 1000) over1s.addAndFetch(1) + // Lock-free running max. + while (true) { + val cur = maxMs.load() + if (lagMs <= cur || maxMs.compareAndSet(cur, lagMs)) break + } + } + + fun reset() { + count.store(0) + sumMs.store(0) + maxMs.store(0) + over1s.store(0) + } + + class Snapshot( + val frames: Long, + val meanMs: Long, + val maxMs: Long, + val over1s: Long, + ) { + override fun toString() = "frames=$frames, mean dispatch-lag=${meanMs}ms, max=${maxMs}ms, $over1s frames waited >1s in our pipeline" + } + + fun snapshot(): Snapshot { + val n = count.load() + return Snapshot( + frames = n, + meanMs = if (n > 0) sumMs.load() / n else 0, + maxMs = maxMs.load(), + over1s = over1s.load(), + ) + } +} diff --git a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt index f0a9c61cec..1200a264ab 100644 --- a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt +++ b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt @@ -20,6 +20,7 @@ */ package com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp +import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.FrameDispatchStats import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocket import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocketListener @@ -35,6 +36,8 @@ import kotlinx.coroutines.launch import okhttp3.OkHttpClient import okhttp3.Request import okhttp3.Response +import kotlin.time.TimeSource +import kotlin.time.TimeSource.Monotonic.ValueTimeMark import okhttp3.WebSocket as OkHttpWebSocket import okhttp3.WebSocketListener as OkHttpWebSocketListener @@ -71,10 +74,14 @@ class BasicOkHttpWebSocket( // fast as it can send and own the buffering; consumer speed // is handled downstream (CachingEventDecoder, // ParallelEventVerifier). - val incomingMessages: Channel = Channel(Channel.UNLIMITED) + val incomingMessages: Channel> = Channel(Channel.UNLIMITED) val job = // Launch a coroutine to process messages from the channel. scope.launch { - for (message in incomingMessages) { + for ((arrivedAt, message) in incomingMessages) { + // Lag from raw socket arrival to us pulling it off the channel = + // OUR-side pipeline delay (queue wait + IO reschedule + decoding + // earlier frames), with the relay's send timing excluded. + FrameDispatchStats.record(arrivedAt.elapsedNow().inWholeMilliseconds) out.onMessage(message) } } @@ -92,8 +99,10 @@ class BasicOkHttpWebSocket( text: String, ) { // Never blocks (unlimited channel): the OkHttp reader - // thread must stay free to keep draining the socket. - incomingMessages.trySendBlocking(text) + // thread must stay free to keep draining the socket. Stamp the + // arrival instant here (on the reader thread, before any of our + // queueing) so downstream dispatch lag is measurable. + incomingMessages.trySendBlocking(TimeSource.Monotonic.markNow() to text) } override fun onClosed( From ea1093adaf6e6e673f2b49665d9f1de814cd6ea1 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 17:56:22 +0000 Subject: [PATCH 25/30] perf(relay): isolate WS frame dispatch onto a dedicated pool, off Dispatchers.IO MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Measured on a GrapeRank crawl, frame decode/dispatch (per-connection consumer coroutines) ran on the shared Dispatchers.IO — the same pool that runs the store's blocking SQLite inserts. During event floods, frame coroutines queued behind those inserts: mean 200ms and up to 3.5s of dispatch lag, with 43k frames waiting >1s in our pipeline. That lag also skews the relay-idle/EOSE timing the crawler reads. Give frame processing its own daemon thread pool (sized to a small multiple of cores; decode is light + CPU-bound), shared across all connections. Frame delivery stays prompt regardless of what the IO pool is doing. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../sockets/okhttp/BasicOkHttpWebSocket.kt | 20 +++++++++++++++++-- 1 file changed, 18 insertions(+), 2 deletions(-) diff --git a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt index 1200a264ab..4497909ce6 100644 --- a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt +++ b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt @@ -28,7 +28,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder import com.vitorpamplona.quartz.utils.Log import kotlinx.coroutines.CoroutineExceptionHandler import kotlinx.coroutines.CoroutineScope -import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.asCoroutineDispatcher import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.channels.trySendBlocking @@ -36,6 +36,7 @@ import kotlinx.coroutines.launch import okhttp3.OkHttpClient import okhttp3.Request import okhttp3.Response +import java.util.concurrent.Executors import kotlin.time.TimeSource import kotlin.time.TimeSource.Monotonic.ValueTimeMark import okhttp3.WebSocket as OkHttpWebSocket @@ -52,6 +53,21 @@ class BasicOkHttpWebSocket( CoroutineExceptionHandler { _, throwable -> Log.e("BasicOkHttpWebSocket", "WebsocketListener Caught exception: ${throwable.message}", throwable) } + + // Frame decode + dispatch runs on its OWN pool, isolated from Dispatchers.IO. + // The shared IO pool also runs the store write path (blocking SQLite inserts); + // measured on a GrapeRank crawl, frame-processing coroutines were queueing + // behind those inserts, adding a 200ms MEAN and up to 3.5s TAIL of dispatch lag + // during event floods — which in turn skews the relay-idle/EOSE timing the + // crawler reads. A dedicated daemon pool keeps frame delivery prompt regardless + // of what the IO pool is doing. Shared across all connections; sized to the + // machine (decode is light + CPU-bound, so a small multiple of cores suffices). + private val frameDispatcher = + Executors + .newFixedThreadPool( + (Runtime.getRuntime().availableProcessors() * 2).coerceIn(4, 32), + ) { r -> Thread(r, "ws-frame").apply { isDaemon = true } } + .asCoroutineDispatcher() } private var socket: OkHttpWebSocket? = null @@ -63,7 +79,7 @@ class BasicOkHttpWebSocket( val listener = object : OkHttpWebSocketListener() { - val scope = CoroutineScope(Dispatchers.IO + exceptionHandler) + val scope = CoroutineScope(frameDispatcher + exceptionHandler) // UNLIMITED on purpose — do NOT bound this channel. The app // holds 2000+ relay connections; a bounded buffer under a From a5c2a8b2c18995ce9dd49313494ee89deb5134bf Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 18:19:43 +0000 Subject: [PATCH 26/30] =?UTF-8?q?revert(relay):=20frame=20dispatch=20back?= =?UTF-8?q?=20to=20Dispatchers.IO=20=E2=80=94=20dedicated=20pool=20regress?= =?UTF-8?q?ed?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The dedicated frame-dispatch pool (ea1093ad) made dispatch lag WORSE, not better: mean 200ms→460ms, max 3.5s→5.5s, frames>1s 43k→76k. The pool was sized cores*2 (=8 here) vs Dispatchers.IO's 64 threads, so it cut frame-processing parallelism ~8x. Lesson: the our-side lag is dominated by per-connection serial decode throughput / thread count, NOT cross-contention with the store's IO writes — the experiment ruled that hypothesis out. Reverting to shared IO; keep FrameDispatchStats. EOSE-wait is confirmed dominantly relay-side (200ms our-mean vs ~5s eose-wait). Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../sockets/okhttp/BasicOkHttpWebSocket.kt | 20 ++----------------- 1 file changed, 2 insertions(+), 18 deletions(-) diff --git a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt index 4497909ce6..1200a264ab 100644 --- a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt +++ b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt @@ -28,7 +28,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder import com.vitorpamplona.quartz.utils.Log import kotlinx.coroutines.CoroutineExceptionHandler import kotlinx.coroutines.CoroutineScope -import kotlinx.coroutines.asCoroutineDispatcher +import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.channels.trySendBlocking @@ -36,7 +36,6 @@ import kotlinx.coroutines.launch import okhttp3.OkHttpClient import okhttp3.Request import okhttp3.Response -import java.util.concurrent.Executors import kotlin.time.TimeSource import kotlin.time.TimeSource.Monotonic.ValueTimeMark import okhttp3.WebSocket as OkHttpWebSocket @@ -53,21 +52,6 @@ class BasicOkHttpWebSocket( CoroutineExceptionHandler { _, throwable -> Log.e("BasicOkHttpWebSocket", "WebsocketListener Caught exception: ${throwable.message}", throwable) } - - // Frame decode + dispatch runs on its OWN pool, isolated from Dispatchers.IO. - // The shared IO pool also runs the store write path (blocking SQLite inserts); - // measured on a GrapeRank crawl, frame-processing coroutines were queueing - // behind those inserts, adding a 200ms MEAN and up to 3.5s TAIL of dispatch lag - // during event floods — which in turn skews the relay-idle/EOSE timing the - // crawler reads. A dedicated daemon pool keeps frame delivery prompt regardless - // of what the IO pool is doing. Shared across all connections; sized to the - // machine (decode is light + CPU-bound, so a small multiple of cores suffices). - private val frameDispatcher = - Executors - .newFixedThreadPool( - (Runtime.getRuntime().availableProcessors() * 2).coerceIn(4, 32), - ) { r -> Thread(r, "ws-frame").apply { isDaemon = true } } - .asCoroutineDispatcher() } private var socket: OkHttpWebSocket? = null @@ -79,7 +63,7 @@ class BasicOkHttpWebSocket( val listener = object : OkHttpWebSocketListener() { - val scope = CoroutineScope(frameDispatcher + exceptionHandler) + val scope = CoroutineScope(Dispatchers.IO + exceptionHandler) // UNLIMITED on purpose — do NOT bound this channel. The app // holds 2000+ relay connections; a bounded buffer under a From e02f00384a63b96ed4a530876fb736cccd1e5ddb Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 18:19:43 +0000 Subject: [PATCH 27/30] perf(graperank): adaptive idle-EOSE cutoff (--eose-idle-ms) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Measured: relays deliver their events in ~0.6s then sit ~4.6s (86% of drain wall) before sending EOSE — mostly relay-side (our pipeline adds only ~200ms). So instead of waiting the full 10s fast window then parking, close a drain that has delivered >=1 event and then gone silent for eoseIdleMs, treating it as complete ("eose-idle"). awaitTerminalOrQuiescent: the idle timer arms only AFTER the first event, so a relay merely slow to answer still gets the full timeoutMs and is never cut prematurely; a still-streaming relay keeps resetting the window. eose-idle paginates if the page was capped and clears timeout strikes (it delivered), but joins notAnswered (no clean EOSE, so its missing authors are retried elsewhere). Off by default (eoseIdleMs=0), CLI --eose-idle-ms, so it can be A/B'd against the plain fast window. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../amethyst/cli/commands/GrapeRankCommand.kt | 3 + .../graperank/GrapeRankCrawler.kt | 62 ++++++++++++++++++- 2 files changed, 63 insertions(+), 2 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index e9a7177e36..811556662f 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -464,6 +464,9 @@ object GrapeRankCommand { insertBatchSize = args.intFlag("insert-batch", 500), drainConcurrency = args.intFlag("drain-concurrency", 24), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), + // Adaptive EOSE cutoff: close a drain that delivered then fell silent + // this many ms, instead of waiting the full fast window. 0 = off. + eoseIdleMs = args.longFlag("eose-idle-ms", 0L), // Cheap TCP reachability pre-probe (--no-probe to disable). No Tor // transport here, so .onion relays are skipped on sight. reachabilityProbe = if (args.bool("no-probe")) null else ::tcpReachable, diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index 62e020a735..c4fbddb034 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -208,6 +208,15 @@ class GrapeRankCrawler( * permanently ignores an author's advertised home. See RelayReachabilityStore. */ val knownDeadRelays: Set = emptySet(), + /** + * Adaptive EOSE cutoff, in ms; 0 disables (the plain fast window). Measured: + * relays deliver their events in ~0.6s then sit ~4.6s more before sending EOSE + * (86% of drain wall). When >0, a drain that has delivered ≥1 event and then + * gone silent for this long is treated as complete instead of waiting out the + * full [timeoutMs] and parking. The idle timer only arms AFTER the first event, + * so a relay merely slow to answer still gets the full [timeoutMs]. + */ + val eoseIdleMs: Long = 0, ) /** What the crawl fetched — the counters the caller reports and the graph is built from. */ @@ -960,6 +969,46 @@ class GrapeRankCrawler( } } + /** + * Fast-window wait with an adaptive EOSE cutoff. Measured: relays answer in + * ~0.6s then go quiet, but sit ~4.6s more before sending EOSE (86% of drain + * wall). So once a relay has delivered events AND then gone silent for [idleMs], + * treat it as complete ("eose-idle") instead of waiting out the full [hardCapMs] + * fast window and parking. Returns: + * - a terminal reason (eose/closed/cannot) if one arrives, + * - "eose-idle" if the relay delivered then fell quiet for [idleMs] (done), + * - null if [hardCapMs] elapsed while it was still streaming or never answered + * (→ park, exactly as the plain fast window would). + * The idle timer only ARMS after the first event, so a relay merely slow to send + * its first response gets the full [hardCapMs] and is never cut prematurely. + */ + private suspend fun awaitTerminalOrQuiescent( + done: CompletableDeferred, + activity: Channel, + idleMs: Long, + hardCapMs: Long, + ): String? { + val start = TimeSource.Monotonic.markNow() + var sawEvent = false + while (true) { + val remaining = hardCapMs - start.elapsedNow().inWholeMilliseconds + if (remaining <= 0) return if (sawEvent) EOSE_IDLE else null + val window = if (sawEvent) minOf(idleMs, remaining) else remaining + val r = + withTimeoutOrNull(window) { + select { + done.onAwait { it } + activity.onReceive { ACTIVITY } + } + } + when (r) { + null -> return if (sawEvent) EOSE_IDLE else null // quiet after data → done; else park + ACTIVITY -> sawEvent = true // an event landed — arm/reset the idle window + else -> return r // terminal reason (eose/closed/cannot) + } + } + } + /** * Fold one late-delivered event from a parked relay into the graph. Only the * round loop calls this (directly or via [foldLateHarvest]), so graph state @@ -1211,7 +1260,12 @@ class GrapeRankCrawler( } } client.subscribe(subId, mapOf(subRelay to groupFilters), listener) - val reason = withTimeoutOrNull(config.timeoutMs) { done.await() } + val reason = + if (config.eoseIdleMs > 0) { + awaitTerminalOrQuiescent(done, activity, config.eoseIdleMs, config.timeoutMs) + } else { + withTimeoutOrNull(config.timeoutMs) { done.await() } + } if (reason != null) { // Terminal within the fast window — resolve this round. val elapsedMs = mark.elapsedNow().inWholeMilliseconds @@ -1235,7 +1289,9 @@ class GrapeRankCrawler( val persisted = persist(drained) // A full page from a clean EOSE may be the relay's cap, not the // whole answer — background-paginate the remainder into lateHarvest. - if (reason == "eose") paginateIfCapped(subRelay, groupFilters, drained) + // Paginate on a clean EOSE and on an idle-cutoff: in both + // cases a full page may be the relay's cap, not the whole answer. + if (reason == "eose" || reason == EOSE_IDLE) paginateIfCapped(subRelay, groupFilters, drained) // Alive if it EOSE'd or handed us anything; a connect-timeout that // gave nothing (classifyDrainFailure leaves it retryable forever) // earns a strike toward eviction instead. @@ -1779,6 +1835,7 @@ class GrapeRankCrawler( ): Outcome = when { reason == "eose" -> if (parked) Outcome.SLOW_EOSE else Outcome.FAST_EOSE + reason == EOSE_IDLE -> Outcome.FAST_EOSE // delivered then quiet: a successful fast completion reason == "timeout" -> if (parked) Outcome.PARK_TIMEOUT else Outcome.FAST_TIMEOUT reason.startsWith("cannot") -> Outcome.CANNOT reason.startsWith("closed:") -> { @@ -1894,6 +1951,7 @@ class GrapeRankCrawler( // (resets the window). A control string that can't collide with a relay's // CLOSED/cannot message, which are the only other select results. private const val ACTIVITY = "activity" + private const val EOSE_IDLE = " eose-idle" // Times we re-query an unreachable user's outbox before giving up, so the // crawl still terminates on a finite graph. From 1f01407442f2054987a7f12332b5fb05ed5a03d2 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 18:52:50 +0000 Subject: [PATCH 28/30] Revert "perf(graperank): adaptive idle-EOSE cutoff (--eose-idle-ms)" This reverts commit e02f00384a63b96ed4a530876fb736cccd1e5ddb. --- .../amethyst/cli/commands/GrapeRankCommand.kt | 3 - .../graperank/GrapeRankCrawler.kt | 62 +------------------ 2 files changed, 2 insertions(+), 63 deletions(-) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt index 811556662f..e9a7177e36 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/GrapeRankCommand.kt @@ -464,9 +464,6 @@ object GrapeRankCommand { insertBatchSize = args.intFlag("insert-batch", 500), drainConcurrency = args.intFlag("drain-concurrency", 24), timeoutEvictStrikes = args.intFlag("timeout-evict", 3), - // Adaptive EOSE cutoff: close a drain that delivered then fell silent - // this many ms, instead of waiting the full fast window. 0 = off. - eoseIdleMs = args.longFlag("eose-idle-ms", 0L), // Cheap TCP reachability pre-probe (--no-probe to disable). No Tor // transport here, so .onion relays are skipped on sight. reachabilityProbe = if (args.bool("no-probe")) null else ::tcpReachable, diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index c4fbddb034..62e020a735 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -208,15 +208,6 @@ class GrapeRankCrawler( * permanently ignores an author's advertised home. See RelayReachabilityStore. */ val knownDeadRelays: Set = emptySet(), - /** - * Adaptive EOSE cutoff, in ms; 0 disables (the plain fast window). Measured: - * relays deliver their events in ~0.6s then sit ~4.6s more before sending EOSE - * (86% of drain wall). When >0, a drain that has delivered ≥1 event and then - * gone silent for this long is treated as complete instead of waiting out the - * full [timeoutMs] and parking. The idle timer only arms AFTER the first event, - * so a relay merely slow to answer still gets the full [timeoutMs]. - */ - val eoseIdleMs: Long = 0, ) /** What the crawl fetched — the counters the caller reports and the graph is built from. */ @@ -969,46 +960,6 @@ class GrapeRankCrawler( } } - /** - * Fast-window wait with an adaptive EOSE cutoff. Measured: relays answer in - * ~0.6s then go quiet, but sit ~4.6s more before sending EOSE (86% of drain - * wall). So once a relay has delivered events AND then gone silent for [idleMs], - * treat it as complete ("eose-idle") instead of waiting out the full [hardCapMs] - * fast window and parking. Returns: - * - a terminal reason (eose/closed/cannot) if one arrives, - * - "eose-idle" if the relay delivered then fell quiet for [idleMs] (done), - * - null if [hardCapMs] elapsed while it was still streaming or never answered - * (→ park, exactly as the plain fast window would). - * The idle timer only ARMS after the first event, so a relay merely slow to send - * its first response gets the full [hardCapMs] and is never cut prematurely. - */ - private suspend fun awaitTerminalOrQuiescent( - done: CompletableDeferred, - activity: Channel, - idleMs: Long, - hardCapMs: Long, - ): String? { - val start = TimeSource.Monotonic.markNow() - var sawEvent = false - while (true) { - val remaining = hardCapMs - start.elapsedNow().inWholeMilliseconds - if (remaining <= 0) return if (sawEvent) EOSE_IDLE else null - val window = if (sawEvent) minOf(idleMs, remaining) else remaining - val r = - withTimeoutOrNull(window) { - select { - done.onAwait { it } - activity.onReceive { ACTIVITY } - } - } - when (r) { - null -> return if (sawEvent) EOSE_IDLE else null // quiet after data → done; else park - ACTIVITY -> sawEvent = true // an event landed — arm/reset the idle window - else -> return r // terminal reason (eose/closed/cannot) - } - } - } - /** * Fold one late-delivered event from a parked relay into the graph. Only the * round loop calls this (directly or via [foldLateHarvest]), so graph state @@ -1260,12 +1211,7 @@ class GrapeRankCrawler( } } client.subscribe(subId, mapOf(subRelay to groupFilters), listener) - val reason = - if (config.eoseIdleMs > 0) { - awaitTerminalOrQuiescent(done, activity, config.eoseIdleMs, config.timeoutMs) - } else { - withTimeoutOrNull(config.timeoutMs) { done.await() } - } + val reason = withTimeoutOrNull(config.timeoutMs) { done.await() } if (reason != null) { // Terminal within the fast window — resolve this round. val elapsedMs = mark.elapsedNow().inWholeMilliseconds @@ -1289,9 +1235,7 @@ class GrapeRankCrawler( val persisted = persist(drained) // A full page from a clean EOSE may be the relay's cap, not the // whole answer — background-paginate the remainder into lateHarvest. - // Paginate on a clean EOSE and on an idle-cutoff: in both - // cases a full page may be the relay's cap, not the whole answer. - if (reason == "eose" || reason == EOSE_IDLE) paginateIfCapped(subRelay, groupFilters, drained) + if (reason == "eose") paginateIfCapped(subRelay, groupFilters, drained) // Alive if it EOSE'd or handed us anything; a connect-timeout that // gave nothing (classifyDrainFailure leaves it retryable forever) // earns a strike toward eviction instead. @@ -1835,7 +1779,6 @@ class GrapeRankCrawler( ): Outcome = when { reason == "eose" -> if (parked) Outcome.SLOW_EOSE else Outcome.FAST_EOSE - reason == EOSE_IDLE -> Outcome.FAST_EOSE // delivered then quiet: a successful fast completion reason == "timeout" -> if (parked) Outcome.PARK_TIMEOUT else Outcome.FAST_TIMEOUT reason.startsWith("cannot") -> Outcome.CANNOT reason.startsWith("closed:") -> { @@ -1951,7 +1894,6 @@ class GrapeRankCrawler( // (resets the window). A control string that can't collide with a relay's // CLOSED/cannot message, which are the only other select results. private const val ACTIVITY = "activity" - private const val EOSE_IDLE = " eose-idle" // Times we re-query an unreachable user's outbox before giving up, so the // crawl still terminates on a finite graph. From aa5c8f04913456f4ad7a1e2ba25c6cc0884de591 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 19:52:42 +0000 Subject: [PATCH 29/30] =?UTF-8?q?revert(relay):=20drop=20FrameDispatchStat?= =?UTF-8?q?s=20=E2=80=94=20per-frame=20cost=20on=20the=20shared=20WS=20hot?= =?UTF-8?q?=20path?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit FrameDispatchStats stamped a ValueTimeMark on every relay frame and recorded a contended atomic per frame in BasicOkHttpWebSocket — the WebSocket layer used by the whole app, unconditionally, forever — to answer a one-time question that only graperank --diagnose read. It served its purpose (proved the our-side dispatch lag is ~200ms mean and the EOSE-wait is dominantly relay-side, so the crawler is network-bound), but the ongoing per-frame Pair allocation + atomic contention on every client's relay traffic isn't worth carrying. Revert the channel back to Channel and delete the stats holder. The diagnose-gated saturation ticker and per-drain latency breakdown stay — they're crawler-local, off the hot path. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankCrawler.kt | 6 -- .../client/accessories/FrameDispatchStats.kt | 84 ------------------- .../sockets/okhttp/BasicOkHttpWebSocket.kt | 17 +--- 3 files changed, 4 insertions(+), 103 deletions(-) delete mode 100644 quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index 62e020a735..1b2d8e706f 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -26,7 +26,6 @@ import com.vitorpamplona.quartz.nip01Core.crypto.verify import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.AdaptiveRelayLimiter import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.DrainFailure -import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.FrameDispatchStats import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.classifyDrainFailure import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener @@ -257,7 +256,6 @@ class GrapeRankCrawler( verifyNanos.store(0) insertNanos.store(0) eventsStored.store(0) - if (config.diagnose) FrameDispatchStats.reset() return CrawlRun(observer, builder).run() } @@ -1597,10 +1595,6 @@ class GrapeRankCrawler( "${throttled.load()} rate-limit responses. " + "High EOSE-wait % → a shorter/adaptive fast window is the lever, not more concurrency.", ) - // Attribution: is the EOSE-wait the relay (slow to SEND eose) or us (the - // eose frame arrived but sat in our IO pipeline)? Low dispatch-lag + high - // EOSE-wait ⇒ relay; high dispatch-lag ⇒ our Dispatchers.IO is backed up. - log("[graperank] frame dispatch (our-side pipeline lag): ${FrameDispatchStats.snapshot()}") } return Stats( rounds = rounds, diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt deleted file mode 100644 index ffd5eaa614..0000000000 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/client/accessories/FrameDispatchStats.kt +++ /dev/null @@ -1,84 +0,0 @@ -/* - * Copyright (c) 2025 Vitor Pamplona - * - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to use, - * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the - * Software, and to permit persons to whom the Software is furnished to do so, - * subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS - * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR - * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN - * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION - * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - */ -package com.vitorpamplona.quartz.nip01Core.relay.client.accessories - -import kotlin.concurrent.atomics.AtomicLong -import kotlin.concurrent.atomics.ExperimentalAtomicApi - -/** - * Diagnostic: the lag between a raw relay frame arriving on the socket (the OkHttp - * reader thread's `onMessage`) and our per-connection consumer coroutine actually - * pulling it off the channel to decode + dispatch. Because that consumer runs on the - * shared `Dispatchers.IO`, this lag is precisely OUR-side pipeline delay — channel - * queue wait + coroutine reschedule + time spent decoding earlier frames — with the - * relay's own send timing excluded (the reader thread enqueues the instant bytes land). - * - * It exists to answer one question the [GrapeRankCrawler]'s EOSE-wait metric cannot on - * its own: when a drain shows a 5-second gap between the relay's last event and its - * EOSE, is the relay slow to send EOSE (low dispatch lag) or is our IO pipeline backed - * up so the already-arrived EOSE frame sits queued (high dispatch lag)? Process-global - * and opt-in: a caller [reset]s before a run and reads [snapshot] after. Off unless - * something records into it, so zero cost on normal paths. - */ -@OptIn(ExperimentalAtomicApi::class) -object FrameDispatchStats { - private val count = AtomicLong(0) - private val sumMs = AtomicLong(0) - private val maxMs = AtomicLong(0) - private val over1s = AtomicLong(0) - - fun record(lagMs: Long) { - count.addAndFetch(1) - sumMs.addAndFetch(lagMs) - if (lagMs >= 1000) over1s.addAndFetch(1) - // Lock-free running max. - while (true) { - val cur = maxMs.load() - if (lagMs <= cur || maxMs.compareAndSet(cur, lagMs)) break - } - } - - fun reset() { - count.store(0) - sumMs.store(0) - maxMs.store(0) - over1s.store(0) - } - - class Snapshot( - val frames: Long, - val meanMs: Long, - val maxMs: Long, - val over1s: Long, - ) { - override fun toString() = "frames=$frames, mean dispatch-lag=${meanMs}ms, max=${maxMs}ms, $over1s frames waited >1s in our pipeline" - } - - fun snapshot(): Snapshot { - val n = count.load() - return Snapshot( - frames = n, - meanMs = if (n > 0) sumMs.load() / n else 0, - maxMs = maxMs.load(), - over1s = over1s.load(), - ) - } -} diff --git a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt index 1200a264ab..f0a9c61cec 100644 --- a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt +++ b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/BasicOkHttpWebSocket.kt @@ -20,7 +20,6 @@ */ package com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp -import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.FrameDispatchStats import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocket import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocketListener @@ -36,8 +35,6 @@ import kotlinx.coroutines.launch import okhttp3.OkHttpClient import okhttp3.Request import okhttp3.Response -import kotlin.time.TimeSource -import kotlin.time.TimeSource.Monotonic.ValueTimeMark import okhttp3.WebSocket as OkHttpWebSocket import okhttp3.WebSocketListener as OkHttpWebSocketListener @@ -74,14 +71,10 @@ class BasicOkHttpWebSocket( // fast as it can send and own the buffering; consumer speed // is handled downstream (CachingEventDecoder, // ParallelEventVerifier). - val incomingMessages: Channel> = Channel(Channel.UNLIMITED) + val incomingMessages: Channel = Channel(Channel.UNLIMITED) val job = // Launch a coroutine to process messages from the channel. scope.launch { - for ((arrivedAt, message) in incomingMessages) { - // Lag from raw socket arrival to us pulling it off the channel = - // OUR-side pipeline delay (queue wait + IO reschedule + decoding - // earlier frames), with the relay's send timing excluded. - FrameDispatchStats.record(arrivedAt.elapsedNow().inWholeMilliseconds) + for (message in incomingMessages) { out.onMessage(message) } } @@ -99,10 +92,8 @@ class BasicOkHttpWebSocket( text: String, ) { // Never blocks (unlimited channel): the OkHttp reader - // thread must stay free to keep draining the socket. Stamp the - // arrival instant here (on the reader thread, before any of our - // queueing) so downstream dispatch lag is measurable. - incomingMessages.trySendBlocking(TimeSource.Monotonic.markNow() to text) + // thread must stay free to keep draining the socket. + incomingMessages.trySendBlocking(text) } override fun onClosed( From cc17b29bc0fce376a3a699f96db9263395580597 Mon Sep 17 00:00:00 2001 From: Claude Date: Thu, 9 Jul 2026 20:43:38 +0000 Subject: [PATCH 30/30] fix(graperank): stop dropping events, un-evict live-but-slow hosts, re-sweep new relays MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addresses correctness/perf issues found in the crawler + reachability audit: - deadHosts permanent eviction (#1): an authority that accrued timeoutEvictStrikes before its first EOSE was evicted forever — clearTimeoutStrikes only zeroed the counter and could not un-evict, contradicting the "a host that ever produces is never evicted" invariant. Add a producedHosts set that isDead() consults, so a proven-productive authority is never treated as dead even if a concurrent strike from the 24-worker fan-out raced it into deadHosts. - Parking-disabled event loss (#2): when parking is off (no bgScope, or parkTimeoutMs <= timeoutMs), a relay that streamed events but didn't EOSE in the fast window had its buffer dropped without persist() and reported count 0. Drain, persist, and return those events like the other two branches; strike only when nothing was delivered. - Wide-sweep over-narrowing (#4): relayListDiscoverySwept excluded an already-swept straggler from the wide pass even though the wide net grows each round, so a 10002 hosted only on a later-learned relay was never fetched. Gate the wide pass on the asked-relay set (wideRelaysSwept) instead: new users get the full net, older stragglers get only newly-appeared relays, no (user, relay) pair asked twice. - Onion detection (#10): replace loose relay.url.contains(".onion") with RelayUrlNormalizer.isOnion() in isDead() and networkTypeOf(), fixing the foo.onionfake.com false positive and the store/crawler disagreement. - rtt-open=0 semantics (#9): document that the crawler's reachable records use rtt-open purely as a liveness flag (0 = latency not probed), not a real 0 ms measurement, and must not be published as authoritative latency data. deadHosts is deliberately still NOT persisted to the 24h reachability cache (#8): a timeout eviction means "too slow under our fan-out this run", not "proven unreachable", so persisting it would blacklist slow-but-live hubs across runs. Co-Authored-By: Claude Opus 4.8 Claude-Session: https://claude.ai/code/session_01MSW59hJtP4Yn8fnRUxc7F5 --- .../graperank/GrapeRankCrawler.kt | 133 ++++++++++++------ .../reachability/RelayReachabilityStore.kt | 11 +- .../RelayReachabilityStoreTest.kt | 19 +-- 3 files changed, 112 insertions(+), 51 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt index 1b2d8e706f..63ebda32bf 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/graperank/GrapeRankCrawler.kt @@ -32,6 +32,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer import com.vitorpamplona.quartz.nip01Core.store.IEventStore import com.vitorpamplona.quartz.nip02FollowList.ContactListEvent import com.vitorpamplona.quartz.nip09Deletions.DeletionEvent @@ -278,12 +279,21 @@ class GrapeRankCrawler( val hopOf = hashMapOf(observer to 0) val done = hashSetOf() - // Users we've already run the outbox-discovery sweep for (indexers + the wide - // live-relay pass in [ensureRelayLists]). The discovery relay set is static, so - // re-asking it for the same never-had-a-10002 user each round it recirculates is - // pure waste — the round-8 profile showed this as ~144k slow kind:10002 drains. + // Users we've already run the STATIC-relay outbox-discovery sweep for (the + // indexer set in [ensureRelayLists]). That relay set never changes, so re-asking + // it for the same never-had-a-10002 user each round it recirculates is pure waste + // — the round-8 profile showed this as ~144k slow kind:10002 drains. // Single-writer: only the round loop's [ensureRelayLists] touches it. val relayListDiscoverySwept = hashSetOf() + + // Relays the WIDE (every-live-relay) recovery pass in [ensureRelayLists] has + // already asked. Unlike the discovery set the wide net GROWS as the crawl learns + // relays, so gating that pass on the swept-user set would never re-ask an old + // straggler for a home relay discovered after its first sweep. Tracking the asked + // relay set instead lets a late-appearing relay still surface an old straggler's + // 10002 while never asking the same (user, relay) pair twice. + // Single-writer: only the round loop's [ensureRelayLists] touches it. + val wideRelaysSwept = hashSetOf() val relaysContacted = hashSetOf() val writeRelayFreq = ConcurrentMap() val liveRelays = ConcurrentSet() @@ -315,12 +325,21 @@ class GrapeRankCrawler( // reaches the threshold on any single one — but they are one server, and it // times out on all of them. We strike the authority and, past // [Config.timeoutEvictStrikes], mark it dead in [deadHosts] so every URL under - // it is skipped. A clean EOSE or any delivered event clears the authority (see - // [clearTimeoutStrikes]), so a host that ever produces is never evicted — only - // the connect-but-silent / dead-endpoint class is. + // it is skipped. A clean EOSE or any delivered event records the authority in + // [producedHosts] (see [clearTimeoutStrikes]), and [isDead] treats an authority + // as dead ONLY while it is in [deadHosts] AND NOT in [producedHosts] — so a host + // that ever produces is never evicted, even if concurrent strikes from the + // 24-worker fan-out raced it into [deadHosts] at the same instant it EOSE'd. + // Only the connect-but-silent / dead-endpoint class stays evicted. val deadTimeoutStrikes = ConcurrentMap() val deadHosts = ConcurrentSet() + // Authorities that ever produced (EOSE or a delivered event) this run. Membership + // here overrides [deadHosts] in [isDead], making the "ever produces ⇒ never + // evicted" invariant race-free: the strike path can lose to a clear and still add + // to [deadHosts], but the gate consults this set and lets the proven host run. + val producedHosts = ConcurrentSet() + // Crawl-wide dedup of event ids, shared across all concurrent drains and // every round. The outbox model mirrors the SAME event (especially kind:10002 // relay lists) across many relays, indexers, and rounds; a per-drain set only @@ -416,34 +435,42 @@ class GrapeRankCrawler( val limit = config.timeoutEvictStrikes if (limit <= 0) return val authority = authorityOf(relay.url) + if (authority in producedHosts) return // proven productive — never evict on timeouts if (authority in deadHosts) return if (deadTimeoutStrikes.merge(authority, 1) { a, b -> a + b } >= limit) deadHosts.add(authority) } /** * A relay just proved its host can produce — a clean EOSE or an actual event — - * so wipe any timeout strikes the authority accrued. Prevents an occasionally- - * slow but useful host (a busy backbone hub, or a multi-path relay where some - * paths are slow) from accumulating its way to eviction across a long crawl. + * so record the authority in [producedHosts] (permanently protecting it from + * timeout eviction for the rest of the run) and wipe any timeout strikes it + * accrued. Prevents an occasionally-slow but useful host (a busy backbone hub, or + * a multi-path relay where some paths are slow) from accumulating its way to + * eviction across a long crawl — and, via the [isDead] gate, un-evicts one that a + * concurrent strike already pushed into [deadHosts] at the same instant. */ fun clearTimeoutStrikes(relay: NormalizedRelayUrl) { + val authority = authorityOf(relay.url) + producedHosts.add(authority) // ConcurrentMap exposes no remove; reset the count to 0 atomically (0 is // below any positive eviction threshold, so it reads as "unstruck"). Guard // on a prior entry so we don't insert a 0 for every host that ever answers. - val authority = authorityOf(relay.url) if (deadTimeoutStrikes[authority] != null) deadTimeoutStrikes.merge(authority, 0) { _, _ -> 0 } } /** * A relay is out of the routing pool if it hard/transient-failed (per-URL - * [deadRelays]) or its whole authority was timeout-evicted ([deadHosts]); a - * .onion relay is dead on sight unless we have a Tor transport, since every - * connect to it would only hang and fail. + * [deadRelays]) or its whole authority was timeout-evicted ([deadHosts]) and has + * not since proven productive ([producedHosts] wins, so a slow-but-live host is + * never permanently evicted); a .onion relay is dead on sight unless we have a + * Tor transport, since every connect to it would only hang and fail. */ - fun isDead(relay: NormalizedRelayUrl): Boolean = - relay in deadRelays || - authorityOf(relay.url) in deadHosts || - (!config.torEnabled && relay.url.contains(".onion")) + fun isDead(relay: NormalizedRelayUrl): Boolean { + val authority = authorityOf(relay.url) + return relay in deadRelays || + (authority in deadHosts && authority !in producedHosts) || + (!config.torEnabled && RelayUrlNormalizer.isOnion(relay.url)) + } /** The busiest live relays we've learned, excluding the dead ones. */ fun topLiveRelays(cap: Int): List = @@ -670,13 +697,6 @@ class GrapeRankCrawler( allLiveRelays: Set, bgScope: CoroutineScope, ) { - // Only sweep users we have never swept: the discovery relay set is static, - // so a second sweep of a user still lacking a 10002 can't find one we didn't - // already miss. Mark them swept up-front so the wide pass below is gated too. - val missing = pubkeys.filter { relaysOf(it) == null && it !in relayListDiscoverySwept } - if (missing.isEmpty()) return - relayListDiscoverySwept.addAll(missing) - suspend fun query( authors: List, relays: Set, @@ -698,20 +718,40 @@ class GrapeRankCrawler( } else { config.relayListDiscoveryRelays } - // On the bounded indexer set, co-fetch the contact list in the same REQ: the - // outbox lookup already pays the round-trip and an indexer holding a user's - // 10002 often holds their kind:3, so we harvest it as a cheap byproduct. - query(missing, discovery, listOf(AdvertisedRelayListEvent.KIND, ContactListEvent.KIND)) - val stillMissing = missing.filter { relaysOf(it) == null } + // First-time DISCOVERY sweep on the static indexer set: a user only needs it + // once (re-asking a static set can't find a 10002 we already missed). Co-fetch + // the contact list in the same REQ — an indexer holding a user's 10002 often + // holds their kind:3, a cheap byproduct of a round-trip we already pay. + val freshlyMissing = pubkeys.filter { relaysOf(it) == null && it !in relayListDiscoverySwept } + val freshlyMissingSet = freshlyMissing.toHashSet() + relayListDiscoverySwept.addAll(freshlyMissing) + query(freshlyMissing, discovery, listOf(AdvertisedRelayListEvent.KIND, ContactListEvent.KIND)) + + // WIDE recovery sweep for kind:10002 ONLY (co-fetching kind:3 across thousands + // of relays inflates this fire-and-forget sweep, which the finishing drain then + // waits on — measured +300s at hop-3). The wide net grows every round, so it is + // gated on the asked-RELAY set, not the swept-user set: + // - a user first swept this round is asked the whole current wide net; + // - a user swept earlier and still missing is asked ONLY the relays that + // appeared since — so its home relay, discovered late, still surfaces. + // No (user, relay) pair is asked twice; every straggler eventually sees every + // live relay. The added olderStillMissing×newWide work self-limits: newWide + // shrinks toward zero as the relay universe is exhausted. val wide = allLiveRelays - discovery - if (stillMissing.isNotEmpty() && wide.isNotEmpty()) { - // The wide net is EVERY live relay (thousands). Ask it for kind:10002 ONLY — - // co-fetching kind:3 here would download the same big contact lists from - // hundreds of relays and inflate this fire-and-forget bgScope sweep, which - // the finishing drain then waits on (measured +300s at hop-3). The outbox - // this finds routes the user's kind:3 to their own relays in the next round. - bgScope.launch { query(stillMissing, wide, listOf(AdvertisedRelayListEvent.KIND)) } + val newWide = wide - wideRelaysSwept + wideRelaysSwept.addAll(wide) + + val freshStillMissing = freshlyMissing.filter { relaysOf(it) == null } + val olderStillMissing = pubkeys.filter { it !in freshlyMissingSet && relaysOf(it) == null } + val hasWork = + (freshStillMissing.isNotEmpty() && wide.isNotEmpty()) || + (olderStillMissing.isNotEmpty() && newWide.isNotEmpty()) + if (hasWork) { + bgScope.launch { + query(freshStillMissing, wide, listOf(AdvertisedRelayListEvent.KIND)) + query(olderStillMissing, newWide, listOf(AdvertisedRelayListEvent.KIND)) + } } } @@ -1284,17 +1324,26 @@ class GrapeRankCrawler( parkedInFlight.addAndFetch(-1) } } + // Parked: events are persisted into lateHarvest by the + // coroutine above, so this round contributes nothing here. + emptyList() } else { + // Parking disabled (no bgScope, or parkTimeoutMs <= timeoutMs): + // drain and persist whatever streamed during the fast window + // instead of dropping it, then return it so the round ingests + // it exactly like the fast path — the other two branches persist, + // this one must too or those events are lost and re-queried. val toMs = mark.elapsedNow().inWholeMilliseconds logSlow(subRelay, "timeout", toMs, groupFilters) - telemetry.record(subRelay, RelayTelemetry.Outcome.FAST_TIMEOUT, toMs, authorsIn(groupFilters), 0) unitEvents.close() client.unsubscribe(subId) - // Parking disabled: a fast timeout with nothing delivered is - // the same unproductive-timeout signal, so strike it here too. - if (unitEvents.tryReceive().isFailure) strikeUnproductiveTimeout(subRelay) + val drained = buildList { for (e in unitEvents) add(e) } + telemetry.record(subRelay, RelayTelemetry.Outcome.FAST_TIMEOUT, toMs, authorsIn(groupFilters), drained.size) + // Nothing delivered → same unproductive-timeout signal, strike it; + // anything delivered proves the host productive and clears it. + if (drained.isEmpty()) strikeUnproductiveTimeout(subRelay) else clearTimeoutStrikes(subRelay) + persist(drained) } - emptyList() } } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt index ee1445779e..93f6648639 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStore.kt @@ -22,6 +22,7 @@ package com.vitorpamplona.quartz.nip66RelayMonitor.reachability import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer import com.vitorpamplona.quartz.nip01Core.signers.NostrSigner import com.vitorpamplona.quartz.nip01Core.store.IEventStore import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.RelayDiscoveryEvent @@ -122,6 +123,14 @@ class RelayReachabilityStore( * WITHOUT. Signed by [signer] and inserted into [store]; being addressable, each * replaces this monitor's prior record for that relay, so the store stays bounded * at roughly the number of distinct relays. + * + * [rttOpenMs] is the measured open round-trip in ms. It defaults to 0 as a **liveness + * flag only** — presence of the `rtt-open` tag, not its magnitude, is what [snapshot] + * reads as "reachable", and a caller that merely proved a relay served events (like + * the crawler) has no dedicated probe latency to report. A `0` therefore means + * "reachable, latency not probed by this writer", NOT a real 0 ms measurement. Do NOT + * publish these records to the wider network as authoritative latency data until a + * dedicated monitor probe supplies a real [rttOpenMs]; aggregators rank by it. */ suspend fun record( reachable: Set, @@ -154,7 +163,7 @@ class RelayReachabilityStore( /** NIP-66 `n` network type inferred from the URL, so a `.onion`/i2p relay is tagged correctly. */ fun networkTypeOf(relay: NormalizedRelayUrl): NetworkType = when { - relay.url.contains(".onion") -> NetworkType.TOR + RelayUrlNormalizer.isOnion(relay.url) -> NetworkType.TOR relay.url.contains(".i2p") -> NetworkType.I2P else -> NetworkType.CLEARNET } diff --git a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt index 63aa6815c7..44ce0df4fc 100644 --- a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt +++ b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip66RelayMonitor/reachability/RelayReachabilityStoreTest.kt @@ -25,6 +25,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerInternal import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.NetworkType import com.vitorpamplona.quartz.utils.Secp256k1Instance import kotlinx.coroutines.runBlocking import kotlin.test.Test @@ -51,6 +52,11 @@ class RelayReachabilityStoreTest { private val dead1 = RelayUrlNormalizer.normalize("wss://dead.example.com") private val dead2 = RelayUrlNormalizer.normalize("wss://gone.example.com") private val onion = RelayUrlNormalizer.normalize("wss://abc.onion") + private val onionPath = RelayUrlNormalizer.normalize("wss://abc.onion/npub1x") + + // Contains the literal ".onion" as a substring but is NOT a Tor host — a loose + // `contains(".onion")` would misclassify it; the normalizer's isOnion must not. + private val fakeOnion = RelayUrlNormalizer.normalize("wss://relay.onionfake.com") @Test fun recordsAndReloadsReachability() = @@ -102,13 +108,10 @@ class RelayReachabilityStoreTest { @Test fun onionRelayIsTaggedTorNetwork() { - assertEquals( - com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.NetworkType.TOR, - RelayReachabilityStore.networkTypeOf(onion), - ) - assertEquals( - com.vitorpamplona.quartz.nip66RelayMonitor.discovery.tags.NetworkType.CLEARNET, - RelayReachabilityStore.networkTypeOf(live1), - ) + assertEquals(NetworkType.TOR, RelayReachabilityStore.networkTypeOf(onion)) + assertEquals(NetworkType.TOR, RelayReachabilityStore.networkTypeOf(onionPath)) + assertEquals(NetworkType.CLEARNET, RelayReachabilityStore.networkTypeOf(live1)) + // A host that merely contains ".onion" as a substring is clearnet, not Tor. + assertEquals(NetworkType.CLEARNET, RelayReachabilityStore.networkTypeOf(fakeOnion)) } }