mirror of
https://github.com/vitorpamplona/amethyst.git
synced 2026-10-05 19:28:25 +00:00
Merge pull request #3458 from vitorpamplona/claude/nostrclient-receiver-perf-d8u27o
Add production benchmarks and negentropy sync optimizations
This commit is contained in:
@@ -100,6 +100,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.RelayLogger
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.RelayOfflineTracker
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.stats.RelayReqStats
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.stats.RelayStats
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder
|
||||
import com.vitorpamplona.quartz.nip03Timestamp.VerificationStateCache
|
||||
import com.vitorpamplona.quartz.nip03Timestamp.okhttp.OkHttpBitcoinExplorer
|
||||
import com.vitorpamplona.quartz.nip03Timestamp.ots.OtsBlockHeightCache
|
||||
@@ -499,8 +500,10 @@ class AppModules(
|
||||
OkHttpLnurlEndpointResolver(roleBasedHttpClientBuilder::okHttpClientForMoney)
|
||||
}
|
||||
|
||||
// Provides a relay pool
|
||||
val client: INostrClient = NostrClient(websocketBuilder, applicationIOScope)
|
||||
// Provides a relay pool. The caching decoder skips re-parsing EVENT frames
|
||||
// that arrive again via another subscription or relay (14-57% of frames in
|
||||
// production measurements).
|
||||
val client: INostrClient = NostrClient(websocketBuilder, applicationIOScope, CachingEventDecoder())
|
||||
|
||||
// Self-heals the "Tor Active but every circuit dead" state the lifecycle watchdogs can't
|
||||
// see (they only arm while Connecting). Watches Tor-routed relay outcomes and, when enough
|
||||
|
||||
@@ -76,6 +76,13 @@ class OkHttpWebSocket(
|
||||
val out: WebSocketListener,
|
||||
) : okhttp3.WebSocketListener() {
|
||||
val scope = CoroutineScope(Dispatchers.IO + exceptionHandler)
|
||||
|
||||
// UNLIMITED on purpose — do NOT bound this channel. The app holds
|
||||
// 2000+ relay connections; a bounded buffer under a slow consumer
|
||||
// would block OkHttp reader threads and park the backlog on the
|
||||
// relay's outbound buffers — infrastructure that isn't ours. Drain
|
||||
// the remote as fast as it can send; consumer speed is handled
|
||||
// downstream.
|
||||
val incomingMessages: Channel<String> = Channel(Channel.UNLIMITED)
|
||||
val job = // Launch a coroutine to process messages from the channel.
|
||||
scope.launch {
|
||||
|
||||
+4
-2
@@ -112,6 +112,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.RelayOfflineTracker
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.auth.EmptyIAuthStatus
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.auth.RelayAuthenticator
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerInternal
|
||||
import com.vitorpamplona.quartz.nip01Core.signers.SignerExceptions
|
||||
@@ -310,8 +311,9 @@ class AccountViewModel(
|
||||
// but uses a SupervisorJob so child failures are independent.
|
||||
val customScope = CoroutineScope(viewModelScope.coroutineContext + SupervisorJob())
|
||||
|
||||
// Provides a relay pool
|
||||
val newClient = NostrClient(Amethyst.instance.websocketBuilder, customScope)
|
||||
// Provides a relay pool. Crawls hit many relays with overlapping
|
||||
// filters, so the duplicate-frame decoder pays off most here.
|
||||
val newClient = NostrClient(Amethyst.instance.websocketBuilder, customScope, CachingEventDecoder())
|
||||
|
||||
// Authenticates with relays (registers itself with the client).
|
||||
RelayAuthenticator(
|
||||
|
||||
@@ -45,6 +45,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.publishAndConfirmDetailed
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer
|
||||
@@ -115,6 +116,9 @@ class Context(
|
||||
val client: NostrClient =
|
||||
NostrClient(
|
||||
websocketBuilder = BasicOkHttpWebSocket.Builder { okhttp },
|
||||
// Skips re-parsing EVENT frames that arrive again via another
|
||||
// subscription or relay.
|
||||
decoder = CachingEventDecoder(),
|
||||
)
|
||||
|
||||
/**
|
||||
|
||||
@@ -26,28 +26,27 @@ import com.vitorpamplona.amethyst.cli.DataDir
|
||||
import com.vitorpamplona.amethyst.cli.Output
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.core.HexKey
|
||||
import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.NegentropySyncException
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.negentropyReconcile
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.toHttp
|
||||
import com.vitorpamplona.quartz.nip77Negentropy.NegErrMessage
|
||||
import com.vitorpamplona.quartz.nip77Negentropy.NegMsgMessage
|
||||
import com.vitorpamplona.quartz.nip77Negentropy.NegentropySession
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import kotlinx.coroutines.channels.Channel.Factory.UNLIMITED
|
||||
import kotlinx.coroutines.withTimeoutOrNull
|
||||
import okhttp3.OkHttpClient
|
||||
import okhttp3.Request
|
||||
import okhttp3.Response
|
||||
import okhttp3.WebSocket
|
||||
import okhttp3.WebSocketListener
|
||||
import kotlinx.coroutines.coroutineScope
|
||||
import kotlinx.coroutines.joinAll
|
||||
import kotlinx.coroutines.launch
|
||||
import java.util.concurrent.atomic.AtomicInteger
|
||||
|
||||
/**
|
||||
* `amy sync --relay URL [filter flags] [--down] [--up] [--timeout SECS]`
|
||||
*
|
||||
* NIP-77 Negentropy set-reconciliation between the local event store and a
|
||||
* relay (nak's `sync`, adapted to amy's local-store model). First it
|
||||
* negotiates the symmetric difference under [filter], then closes the loop:
|
||||
* relay (nak's `sync`, adapted to amy's local-store model). The protocol
|
||||
* itself is quartz's [negentropyReconcile]: it pins the relay with a
|
||||
* keep-alive subscription, splits the filter by `created_at` window whenever
|
||||
* the relay caps the set (strfry `max_sync_events`), and streams the two
|
||||
* directions of the diff as each round completes. This command closes the
|
||||
* loop on that stream:
|
||||
*
|
||||
* --down (default) download events the relay has and we lack (REQ by id)
|
||||
* --up upload events we have and the relay lacks (EVENT)
|
||||
@@ -55,13 +54,30 @@ import okhttp3.WebSocketListener
|
||||
* Pass both for a full bidirectional sync. The filter flags are the same as
|
||||
* `fetch`/`subscribe`; an empty filter reconciles the whole store.
|
||||
*
|
||||
* Thin assembly only: the negentropy protocol lives in quartz
|
||||
* (`NegentropySession`); this file drives the WebSocket round-trips and
|
||||
* reuses `Context.drain` / `Context.publish` for the NIP-01 follow-up.
|
||||
* Both directions are pipelined with the reconcile: need-id batches feed
|
||||
* [DOWNLOAD_WORKERS] concurrent by-id REQ drains and have-ids feed a single
|
||||
* uploader, so downloads and uploads overlap the remaining reconcile rounds
|
||||
* instead of waiting for the full diff. Every downloaded event funnels
|
||||
* through `Context.drain`'s verify-and-store path, unchanged.
|
||||
*
|
||||
* Thin assembly only: the windowing, streaming, and back-pressure live in
|
||||
* quartz (`negentropyReconcile`); this file only routes ids to
|
||||
* `Context.drain` / `Context.publish`.
|
||||
*/
|
||||
object SyncCommand {
|
||||
private const val ID_CHUNK = 500
|
||||
|
||||
/**
|
||||
* Concurrent by-id download REQs. With [RECONCILE_CONCURRENCY] NEG
|
||||
* sessions and the keep-alive, peak concurrent subscriptions on the
|
||||
* relay are DOWNLOAD_WORKERS + RECONCILE_CONCURRENCY + 1 = 7 — well
|
||||
* under the common NIP-11 `max_subscriptions` floor of 20.
|
||||
*/
|
||||
private const val DOWNLOAD_WORKERS = 4
|
||||
|
||||
/** Overlapped `created_at`-window reconciles after an over-cap split. */
|
||||
private const val RECONCILE_CONCURRENCY = 2
|
||||
|
||||
suspend fun run(
|
||||
dataDir: DataDir,
|
||||
rest: Array<String>,
|
||||
@@ -83,127 +99,79 @@ object SyncCommand {
|
||||
ctx.prepare()
|
||||
val localEvents = ctx.store.query<Event>(filter)
|
||||
val localById = localEvents.associateBy { it.id }
|
||||
val localEntries = localEvents.map { IdAndTime(it.createdAt, it.id) }
|
||||
|
||||
val diff =
|
||||
negotiate(relay.toHttp(), filter, localEvents, timeoutMs)
|
||||
?: return Output.error("timeout", "no negentropy response from ${relay.url} within ${timeoutMs}ms")
|
||||
if (diff.error != null) {
|
||||
return Output.error("sync_error", diff.error)
|
||||
}
|
||||
val downloaded = AtomicInteger(0)
|
||||
val uploaded = AtomicInteger(0)
|
||||
|
||||
// needIds = relay has, we lack; haveIds = we have, relay lacks.
|
||||
var downloaded = 0
|
||||
if (down && diff.needIds.isNotEmpty()) {
|
||||
diff.needIds.chunked(ID_CHUNK).forEach { chunk ->
|
||||
val got = ctx.drain(mapOf(relay to listOf(Filter(ids = chunk))), timeoutMs)
|
||||
downloaded += got.size
|
||||
val result =
|
||||
try {
|
||||
coroutineScope {
|
||||
// needIds = relay has, we lack; haveIds = we have, relay lacks.
|
||||
// Bounded so a slow download back-pressures the reconcile
|
||||
// rounds instead of piling ids up in memory.
|
||||
val needBatches = Channel<List<HexKey>>(DOWNLOAD_WORKERS * 2)
|
||||
// Unbounded is fine here: have-ids reference events we already
|
||||
// hold locally, so memory is bounded by the local set.
|
||||
val haveBatches = Channel<List<HexKey>>(Channel.UNLIMITED)
|
||||
|
||||
val downloaders =
|
||||
List(DOWNLOAD_WORKERS) {
|
||||
launch {
|
||||
for (batch in needBatches) {
|
||||
val got = ctx.drain(mapOf(relay to listOf(Filter(ids = batch))), timeoutMs)
|
||||
downloaded.addAndGet(got.size)
|
||||
}
|
||||
}
|
||||
}
|
||||
val uploader =
|
||||
launch {
|
||||
for (batch in haveBatches) {
|
||||
for (id in batch) {
|
||||
val ev = localById[id] ?: continue
|
||||
val ack = ctx.publish(ev, setOf(relay))
|
||||
if (ack.values.any { it }) uploaded.incrementAndGet()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val reconcile =
|
||||
try {
|
||||
ctx.client.negentropyReconcile(
|
||||
relay = relay,
|
||||
filter = filter,
|
||||
localEntries = localEntries,
|
||||
batchSize = ID_CHUNK,
|
||||
idleTimeoutMs = timeoutMs,
|
||||
reconcileConcurrency = RECONCILE_CONCURRENCY,
|
||||
onHaveIds = if (up) { batch -> haveBatches.send(batch) } else null,
|
||||
onNeedIds = { batch -> if (down) needBatches.send(batch) },
|
||||
)
|
||||
} finally {
|
||||
needBatches.close()
|
||||
haveBatches.close()
|
||||
}
|
||||
|
||||
downloaders.joinAll()
|
||||
uploader.join()
|
||||
reconcile
|
||||
}
|
||||
} catch (e: NegentropySyncException) {
|
||||
return Output.error("sync_error", e.message ?: "negentropy sync failed")
|
||||
}
|
||||
}
|
||||
|
||||
var uploaded = 0
|
||||
if (up && diff.haveIds.isNotEmpty()) {
|
||||
diff.haveIds.forEach { id ->
|
||||
val ev = localById[id] ?: return@forEach
|
||||
val ack = ctx.publish(ev, setOf(relay))
|
||||
if (ack.values.any { it }) uploaded++
|
||||
}
|
||||
}
|
||||
|
||||
Output.emit(
|
||||
mapOf(
|
||||
"relay" to relay.url,
|
||||
"local_events" to localEvents.size,
|
||||
"rounds" to diff.rounds,
|
||||
"need" to diff.needIds.size,
|
||||
"have" to diff.haveIds.size,
|
||||
"downloaded" to downloaded,
|
||||
"uploaded" to uploaded,
|
||||
"windows" to result.windows,
|
||||
"need" to result.needCount,
|
||||
"have" to result.haveCount,
|
||||
"downloaded" to downloaded.get(),
|
||||
"uploaded" to uploaded.get(),
|
||||
),
|
||||
)
|
||||
return 0
|
||||
}
|
||||
}
|
||||
|
||||
private data class Diff(
|
||||
val haveIds: List<HexKey>,
|
||||
val needIds: List<HexKey>,
|
||||
val rounds: Int,
|
||||
val error: String?,
|
||||
)
|
||||
|
||||
/**
|
||||
* Drive one NIP-77 reconciliation over a raw WebSocket and return the
|
||||
* symmetric difference. Mirrors geode's interop sync driver: the protocol
|
||||
* state machine is [NegentropySession]; this only shuttles frames.
|
||||
*/
|
||||
private suspend fun negotiate(
|
||||
httpUrl: String,
|
||||
filter: Filter,
|
||||
localEvents: List<Event>,
|
||||
timeoutMs: Long,
|
||||
subId: String = "amy-sync",
|
||||
maxRounds: Int = 64,
|
||||
): Diff? {
|
||||
val incoming = Channel<String>(UNLIMITED)
|
||||
val client = OkHttpClient.Builder().build()
|
||||
val ws =
|
||||
client.newWebSocket(
|
||||
Request.Builder().url(httpUrl).build(),
|
||||
object : WebSocketListener() {
|
||||
override fun onMessage(
|
||||
webSocket: WebSocket,
|
||||
text: String,
|
||||
) {
|
||||
incoming.trySend(text)
|
||||
}
|
||||
|
||||
override fun onClosing(
|
||||
webSocket: WebSocket,
|
||||
code: Int,
|
||||
reason: String,
|
||||
) {
|
||||
incoming.close()
|
||||
}
|
||||
|
||||
override fun onFailure(
|
||||
webSocket: WebSocket,
|
||||
t: Throwable,
|
||||
response: Response?,
|
||||
) {
|
||||
incoming.close(t)
|
||||
}
|
||||
},
|
||||
)
|
||||
|
||||
return try {
|
||||
withTimeoutOrNull(timeoutMs) {
|
||||
val session = NegentropySession(subId, filter, localEvents, frameSizeLimit = 0)
|
||||
ws.send(OptimizedJsonMapper.toJson(session.open()))
|
||||
|
||||
val have = mutableSetOf<HexKey>()
|
||||
val need = mutableSetOf<HexKey>()
|
||||
var rounds = 0
|
||||
while (rounds < maxRounds) {
|
||||
val raw = incoming.receive()
|
||||
rounds++
|
||||
when (val msg = OptimizedJsonMapper.fromJsonToMessage(raw)) {
|
||||
is NegErrMessage -> return@withTimeoutOrNull Diff(have.toList(), need.toList(), rounds, "${msg.subId}: ${msg.reason}")
|
||||
is NegMsgMessage -> {
|
||||
val r = session.processMessage(msg.message)
|
||||
have += r.haveIds
|
||||
need += r.needIds
|
||||
if (r.isComplete()) {
|
||||
return@withTimeoutOrNull Diff(have.toList(), need.toList(), rounds, null)
|
||||
}
|
||||
ws.send(OptimizedJsonMapper.toJson(r.nextCmd!!))
|
||||
}
|
||||
else -> {} // ignore NOTICE/etc. and keep waiting
|
||||
}
|
||||
}
|
||||
Diff(have.toList(), need.toList(), rounds, "did not converge in $maxRounds rounds")
|
||||
}
|
||||
} finally {
|
||||
ws.close(1000, "amy-sync-done")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+4
-1
@@ -25,6 +25,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.listeners.RelayConnectionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.IRelayClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.EventMessage
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.Message
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.Command
|
||||
@@ -56,7 +57,9 @@ data class RelayMetrics(
|
||||
open class RelayConnectionManager(
|
||||
websocketBuilder: WebsocketBuilder,
|
||||
) : RelayConnectionListener {
|
||||
private val _client = NostrClient(websocketBuilder)
|
||||
// The caching decoder skips re-parsing EVENT frames that arrive again via
|
||||
// another subscription or relay.
|
||||
private val _client = NostrClient(websocketBuilder, decoder = CachingEventDecoder())
|
||||
|
||||
/** Exposes the underlying INostrClient for subscription coordinators */
|
||||
val client: com.vitorpamplona.quartz.nip01Core.relay.client.INostrClient get() = _client
|
||||
|
||||
@@ -25,6 +25,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.server.RelaySession
|
||||
import io.ktor.server.websocket.DefaultWebSocketServerSession
|
||||
import io.ktor.websocket.Frame
|
||||
import io.ktor.websocket.readText
|
||||
import kotlinx.coroutines.cancel
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import kotlinx.coroutines.channels.ClosedSendChannelException
|
||||
import kotlinx.coroutines.channels.consumeEach
|
||||
@@ -97,17 +98,34 @@ internal class WebSocketSessionPump(
|
||||
}
|
||||
val session =
|
||||
server.connect { json ->
|
||||
// The channel itself is UNLIMITED, so trySend can't
|
||||
// report "full". Enforce the cap explicitly: increment
|
||||
// first, refuse if we'd cross the bound, otherwise
|
||||
// enqueue.
|
||||
val depth = outstanding.incrementAndGet()
|
||||
if (depth > MAX_OUTGOING_BUFFER) {
|
||||
outstanding.decrementAndGet()
|
||||
droppedForBackpressure = true
|
||||
outQueue.close()
|
||||
return@connect
|
||||
// Backpressure, not instant drop. A full backlog usually
|
||||
// means a REQ replay is producing faster than the socket
|
||||
// writer ships — normal for bulk downloads to a HEALTHY
|
||||
// client — so pace the producer (this blocks the replay
|
||||
// coroutine's thread; the live-fanout path already
|
||||
// documents that a slow sub blocks the ingest writer).
|
||||
// Only a client that stays behind for the whole deadline
|
||||
// is genuinely slow; then close FOR REAL. The previous
|
||||
// code dropped at the cap immediately — killing healthy
|
||||
// bulk clients the moment the O(n²)-replay throttle was
|
||||
// fixed — and only closed outQueue, leaving the socket
|
||||
// half-dead (no EOSE, no close frame; the client hung).
|
||||
var waitedMs = 0L
|
||||
while (outstanding.get() >= MAX_OUTGOING_BUFFER && !droppedForBackpressure) {
|
||||
if (waitedMs >= SLOW_CLIENT_DEADLINE_MS) {
|
||||
droppedForBackpressure = true
|
||||
outQueue.close()
|
||||
// Actually terminate the connection so the client
|
||||
// sees the failure instead of waiting forever.
|
||||
ws.cancel()
|
||||
return@connect
|
||||
}
|
||||
Thread.sleep(BACKPRESSURE_POLL_MS)
|
||||
waitedMs += BACKPRESSURE_POLL_MS
|
||||
}
|
||||
if (droppedForBackpressure) return@connect
|
||||
|
||||
outstanding.incrementAndGet()
|
||||
val res = outQueue.trySend(json)
|
||||
if (!res.isSuccess) {
|
||||
// Channel was closed concurrently (e.g. teardown).
|
||||
@@ -147,5 +165,16 @@ internal class WebSocketSessionPump(
|
||||
* channel (≈ a few hundred bytes), not the full cap.
|
||||
*/
|
||||
const val MAX_OUTGOING_BUFFER: Int = 8192
|
||||
|
||||
/**
|
||||
* How long a producer will wait for the writer to drain a full
|
||||
* backlog before the client is declared slow and disconnected.
|
||||
* A healthy client on any sane link drains 8192 frames orders of
|
||||
* magnitude faster than this.
|
||||
*/
|
||||
const val SLOW_CLIENT_DEADLINE_MS: Long = 30_000
|
||||
|
||||
/** Poll interval while pacing a producer against a full backlog. */
|
||||
const val BACKPRESSURE_POLL_MS: Long = 5
|
||||
}
|
||||
}
|
||||
|
||||
@@ -158,7 +158,7 @@ class Nip77NegentropyTest {
|
||||
val client = WireClient(hub, relayUrl)
|
||||
try {
|
||||
val session =
|
||||
NegentropySession(
|
||||
NegentropySession.fromEvents(
|
||||
subId = "neg-1",
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
localEvents = clientEvents,
|
||||
@@ -205,7 +205,7 @@ class Nip77NegentropyTest {
|
||||
val client = WireClient(hub, relayUrl)
|
||||
try {
|
||||
// First session.
|
||||
val s1 = NegentropySession("neg", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
val s1 = NegentropySession.fromEvents("neg", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
client.send(OptimizedJsonMapper.toJson(s1.open()))
|
||||
val response = client.nextMessage() as NegMsgMessage
|
||||
val r1 = s1.processMessage(response.message)
|
||||
@@ -219,7 +219,7 @@ class Nip77NegentropyTest {
|
||||
// server must build a new session and respond. If the
|
||||
// close didn't free state, this would either error or
|
||||
// continue the previous reconciliation.
|
||||
val s2 = NegentropySession("neg", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
val s2 = NegentropySession.fromEvents("neg", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
client.send(OptimizedJsonMapper.toJson(s2.open()))
|
||||
val resp2 = client.nextMessage() as NegMsgMessage
|
||||
val r2 = s2.processMessage(resp2.message)
|
||||
@@ -262,7 +262,7 @@ class Nip77NegentropyTest {
|
||||
val client = WireClient(capped, capUrl)
|
||||
try {
|
||||
val session =
|
||||
NegentropySession(
|
||||
NegentropySession.fromEvents(
|
||||
subId = "neg-overflow",
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
localEvents = emptyList(),
|
||||
@@ -295,7 +295,7 @@ class Nip77NegentropyTest {
|
||||
val client = WireClient(capped, capUrl)
|
||||
try {
|
||||
repeat(2) { i ->
|
||||
val s = NegentropySession("ok-$i", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
val s = NegentropySession.fromEvents("ok-$i", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
client.send(OptimizedJsonMapper.toJson(s.open()))
|
||||
// Drain the NEG-MSG response so the next OPEN goes
|
||||
// through cleanly.
|
||||
@@ -303,7 +303,7 @@ class Nip77NegentropyTest {
|
||||
}
|
||||
|
||||
// Third OPEN — should be rejected with a NOTICE.
|
||||
val third = NegentropySession("third", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
val third = NegentropySession.fromEvents("third", Filter(kinds = listOf(1)), localEvents = emptyList())
|
||||
client.send(OptimizedJsonMapper.toJson(third.open()))
|
||||
val response = client.nextMessage()
|
||||
assertTrue(response is NoticeMessage, "expected NOTICE, got ${response::class.simpleName}")
|
||||
@@ -327,12 +327,12 @@ class Nip77NegentropyTest {
|
||||
try {
|
||||
// First open with localEvents = a; next we'll re-open
|
||||
// and confirm the new session sees a fresh state.
|
||||
val first = NegentropySession("dup", Filter(kinds = listOf(1)), localEvents = a)
|
||||
val first = NegentropySession.fromEvents("dup", Filter(kinds = listOf(1)), localEvents = a)
|
||||
client.send(OptimizedJsonMapper.toJson(first.open()))
|
||||
client.nextMessage() as NegMsgMessage // discard
|
||||
|
||||
// Re-OPEN with same subId, different localEvents.
|
||||
val second = NegentropySession("dup", Filter(kinds = listOf(1)), localEvents = a + b)
|
||||
val second = NegentropySession.fromEvents("dup", Filter(kinds = listOf(1)), localEvents = a + b)
|
||||
client.send(OptimizedJsonMapper.toJson(second.open()))
|
||||
val resp = client.nextMessage() as NegMsgMessage
|
||||
val r = second.processMessage(resp.message)
|
||||
|
||||
@@ -116,7 +116,7 @@ class InteropSyncDriver(
|
||||
)
|
||||
|
||||
return try {
|
||||
val session = NegentropySession(subId, filter, localEvents, frameSizeLimit)
|
||||
val session = NegentropySession.fromEvents(subId, filter, localEvents, frameSizeLimit)
|
||||
|
||||
// Step 1: NEG-OPEN.
|
||||
check(ws.send(OptimizedJsonMapper.toJson(session.open()))) { "send NEG-OPEN failed" }
|
||||
|
||||
@@ -85,6 +85,12 @@ kotlin {
|
||||
tasks.withType<Test>().configureEach {
|
||||
maxHeapSize = "4g"
|
||||
environment("TEST_RESOURCES_ROOT", rootDir)
|
||||
// Opt-in gate for ProductionReceiverBenchmark (opens sockets to live
|
||||
// relays). `-PprodRelayBench=1` reaches the daemon even when an env
|
||||
// var would not survive daemon reuse.
|
||||
(project.findProperty("prodRelayBench") as? String)?.let {
|
||||
environment("PROD_RELAY_BENCH", it)
|
||||
}
|
||||
}
|
||||
|
||||
tasks.withType<KotlinNativeTest>().configureEach {
|
||||
|
||||
@@ -0,0 +1,681 @@
|
||||
# NostrClient receiver performance — review + production measurements
|
||||
|
||||
**Date:** 2026-07-02
|
||||
**Status:** findings + measurement harness; no production code changed yet
|
||||
**Harness:** `quartz/src/jvmTest/.../relay/prodbench/ProductionReceiverBenchmark.kt`
|
||||
(run with `./gradlew :quartz:jvmTest --tests "*.ProductionReceiverBenchmark" -PprodRelayBench=1`)
|
||||
|
||||
## Question
|
||||
|
||||
Is the NostrClient's single-threaded receiver making us significantly slower?
|
||||
And how do our filters actually behave against production relays?
|
||||
|
||||
## What the receiver actually is (review)
|
||||
|
||||
The receive path is single-threaded **per relay connection**, not globally:
|
||||
|
||||
```
|
||||
OkHttp reader thread (per socket)
|
||||
└─ trySendBlocking → Channel(UNLIMITED) [BasicOkHttpWebSocket / app OkHttpWebSocket]
|
||||
└─ ONE consumer coroutine per connection (Dispatchers.IO)
|
||||
├─ OptimizedJsonMapper.fromJsonToMessage [JSON parse] ~32µs/msg (JVM)
|
||||
├─ PoolRequests.onIncomingMessage [spin-lock + state]
|
||||
│ └─ SubscriptionListener.onEvent [assemblers, inline]
|
||||
└─ NostrClient.listeners.forEach [EventCollector, loggers…]
|
||||
└─ LocalCache.justConsume(…, wasVerified = false)
|
||||
└─ dedup by id → justVerify() [Schnorr verify] ~75µs/event (JVM)
|
||||
└─ note creation, index updates, flow emissions
|
||||
```
|
||||
|
||||
Everything downstream of the channel — parse, subscription state machine, every
|
||||
listener, signature verification, and LocalCache consumption — runs serially on
|
||||
that one consumer coroutine. Different relays run on different coroutines
|
||||
(cross-relay parallelism exists), but they contend on shared state: the
|
||||
`PoolRequests` spin lock, `LocalCache`'s maps, and (on Android) the same
|
||||
`Dispatchers.IO` pool as everything else.
|
||||
|
||||
Notably, the **relay-server side already solved this** on the write path:
|
||||
`IngestQueue` batches incoming EVENTs and runs Schnorr verification for a batch
|
||||
in parallel on `Dispatchers.Default` (`parallelVerify`), precisely because "an
|
||||
event's `verify()` is CPU-bound and parallelisable". The client receiver has no
|
||||
equivalent.
|
||||
|
||||
## Production measurements (2026-07-02)
|
||||
|
||||
Environment: JVM 21, 4-core x86 cloud container (server CPU — phones are
|
||||
5–10× slower per core), relays `relay.damus.io`, `nos.lol`, `relay.primal.net`,
|
||||
`nostr.wine`. All scenarios ran after a discarded warmup pass (JIT warm).
|
||||
`INLINE` = verify on the receiver coroutine (what the app does today via
|
||||
`CacheClientConnector`); `PARALLEL` = verify offloaded to a
|
||||
`Dispatchers.Default` pool.
|
||||
|
||||
### Filter behavior
|
||||
|
||||
| Scenario | received | unique | duplicates | notes |
|
||||
|---|---|---|---|---|
|
||||
| kind-1 firehose, limit 500 ×4 relays | 2013 | 1738 | 14% | all relays honored limit; EOSE 0.3–1.0s after REQ |
|
||||
| notifications (p=busy pubkey, kinds 1,6,7,9735) | 1552 | 986 | 36% | primal returned only 52 (thin index for this query) |
|
||||
| metadata burst (kinds 0,10002, 300 authors) | 1146 | 491 | 57% | each profile fetched ~2.3× across 4 relays |
|
||||
|
||||
- Time from `subscribe()` to REQ-on-the-wire is dominated by connect/TLS
|
||||
(~0.7–1.5s cold); local dispatch is negligible.
|
||||
- Event sizes vary wildly by relay: damus kind-1 averaged **8.7 KB**/event
|
||||
(4.4 MB for 500 events) vs ~0.8 KB on nos.lol — parse cost tracks size
|
||||
(p90 proc 1.5ms on damus vs 0.18ms on nos.lol).
|
||||
- Duplicate factor grows with fan-out: the metadata pattern wastes >half the
|
||||
received bytes on duplicates. Dedup happens *before* verify (LocalCache
|
||||
checks the id first), so duplicates cost parse + dedup lookup, not a verify.
|
||||
|
||||
### Receiver queue behavior (the actual question)
|
||||
|
||||
Queue delay = how long a frame sat in the per-connection channel before the
|
||||
consumer picked it up. This is the direct symptom of a saturated receiver.
|
||||
|
||||
kind-1 firehose (large events), per relay:
|
||||
|
||||
| | INLINE p50 / p90 / max | PARALLEL p50 / p90 / max |
|
||||
|---|---|---|
|
||||
| relay.damus.io | 6.8ms / 26.6ms / 77.9ms | 5.4ms / 38.1ms / 41.5ms |
|
||||
| nos.lol | 4.3ms / 11.2ms / 15.9ms | 1.3ms / 3.8ms / 4.5ms |
|
||||
| relay.primal.net | 5.4ms / 24.5ms / 28.1ms | 1.2ms / 1.8ms / 7.7ms |
|
||||
| nostr.wine | 9.4ms / 30.3ms / 34.8ms | 2.4ms / 8.4ms / 20.7ms |
|
||||
|
||||
metadata burst (small events, fastest arrival — damus delivered at 3300–8400 msg/s):
|
||||
|
||||
| | INLINE p50 / max | PARALLEL p50 / max |
|
||||
|---|---|---|
|
||||
| relay.damus.io | 3.0ms / 10.7ms | 0.14ms / 0.48ms |
|
||||
|
||||
Consumer busy fraction peaked at **53%** (nostr.wine, INLINE, 3000 msg/s) on
|
||||
this fast CPU — i.e. a single relay bursting at production speed already
|
||||
consumes half of one core with inline verification, on hardware much faster
|
||||
than a phone.
|
||||
|
||||
### Single-thread ceilings (offline, captured production frames)
|
||||
|
||||
- JSON parse: **31,000 msg/s** (32µs/msg, mixed sizes incl. damus 8.7KB events)
|
||||
- Schnorr verify: **13,200 events/s** sequential (75µs/event);
|
||||
**34,400 events/s** across 4 cores (2.6–3.0× speedup)
|
||||
- Combined parse+verify ceiling for one receiver coroutine: **~9,000 events/s**
|
||||
on this CPU. On a phone core (5–10× slower): **~1,000–2,000 events/s**, and
|
||||
that is *before* LocalCache's note creation, index updates and flow emissions,
|
||||
which the harness does not model and which run on the same coroutine.
|
||||
|
||||
## Answer
|
||||
|
||||
1. **The receiver is not globally single-threaded** — it is one consumer per
|
||||
relay. Cross-relay parallelism already exists. The per-relay serialization
|
||||
is what caps throughput.
|
||||
|
||||
2. **For steady-state browsing the receiver is not the bottleneck.** Queue
|
||||
delays with today's inline design are single-digit-to-tens of ms; connect
|
||||
latency and relay EOSE times (300ms–1.5s) dominate what the user perceives.
|
||||
|
||||
3. **For burst ingest (app start, feed switch, big profile sync) it is a real
|
||||
cost, and it scales linearly with burst size.** A 5,000-event burst from one
|
||||
fast relay serializes ~375ms of parse+verify on this benchmark machine —
|
||||
plausibly **2–4s per relay on a mid-range phone**, during which that relay's
|
||||
EOSE, OKs and live events all sit behind the backlog. Verification is
|
||||
60–70% of that inline cost and is embarrassingly parallel (measured 2.6–3×
|
||||
on 4 cores; phones have 8).
|
||||
|
||||
4. **Offloading verification cuts queue latency 2–5× and raises the per-relay
|
||||
ceiling ~3×** with no protocol or API change — the exact pattern
|
||||
`IngestQueue.parallelVerify` already uses on the server side.
|
||||
|
||||
## Dispatch stage (post-parse, pre-verify) microbenchmark
|
||||
|
||||
**Harness:** `quartz/src/jvmTest/.../relay/prodbench/DispatchStageBenchmark.kt`
|
||||
(offline, ungated: `./gradlew :quartz:jvmTest --tests "*.DispatchStageBenchmark"`)
|
||||
|
||||
Measures everything between the JSON parse and where verification would start:
|
||||
`NostrClient.onIncomingMessage` → `PoolRequests` (spin lock + sub state) →
|
||||
listeners → id-dedup → handoff to a verify stage. 30k unique synthetic events,
|
||||
2 overlapping subs per relay, delivered by 1 feeder thread (one busy relay) and
|
||||
by 4 (four relays bursting into the shared client). JVM 21, 4 cores.
|
||||
|
||||
| variant | 1 feeder (deliveries/s) | 4 feeders (aggregate deliveries/s) |
|
||||
|---|---|---|
|
||||
| PoolRequests-only | 11.1M | 3.6M |
|
||||
| full dispatch, no-op sink | 10.2M | 4.3M |
|
||||
| + dedup (CHM keySet) | 6.8M | 3.5M |
|
||||
| + dedup + per-event channel handoff | 3.6M | 3.1M |
|
||||
| + dedup + batched handoff (64) | 6.8M | 2.8M |
|
||||
| **early dedup + batched handoff** | 7.2M | **11.4M** |
|
||||
|
||||
Findings:
|
||||
|
||||
1. **Uncontended, the dispatch stage is ~100ns/message** — 300× cheaper than
|
||||
parse (32µs) and 750× cheaper than verify (75µs). One relay can never
|
||||
saturate it; nothing to fix for the single-relay case.
|
||||
2. **It scales negatively under concurrency.** Four feeder threads deliver
|
||||
*less* aggregate throughput (3.6–4.3M/s) than one thread alone (10–11M/s):
|
||||
the `PoolRequests` busy-wait spin lock serializes every EVENT frame from
|
||||
every relay and burns the other cores spinning. Isolated `PoolRequests`
|
||||
shows the same collapse, so the lock (not listeners or dedup) is the cause.
|
||||
At production message rates (~3k msg/s/relay) this is not yet the
|
||||
bottleneck, but it wastes cores the verify pool would want, and it is the
|
||||
structural ceiling once verify moves off the receiver.
|
||||
3. **Handoff granularity matters:** a per-event `Channel.send` costs ~180ns
|
||||
extra per event (halves single-feeder throughput); a 64-event batch makes
|
||||
the handoff essentially free — same conclusion the server's `IngestQueue`
|
||||
already embodies.
|
||||
4. **Early dedup is the biggest lever under multi-relay load:** checking the
|
||||
seen-ids set right after parse — before entering the locked dispatch path —
|
||||
let duplicate frames (75% of deliveries in the 4-relay setup; production
|
||||
showed 14–57% dups) skip the contended section entirely: 2.7–4× the
|
||||
aggregate throughput of every other 4-feeder variant. Semantic caveat: a
|
||||
short-circuited duplicate no longer bumps that sub's per-relay
|
||||
`onNewEvent`/stats counters, so paging/EOSE bookkeeping would need the
|
||||
cheap counters kept ahead of the skip.
|
||||
|
||||
## Bulk download: N-million events from one relay (download + parse only)
|
||||
|
||||
**Harness:** `quartz/src/jvmTest/.../relay/prodbench/BulkDownloadBenchmark.kt`
|
||||
(gated: `./gradlew :quartz:jvmTest --tests "*.BulkDownloadBenchmark" -PprodRelayBench=1`)
|
||||
|
||||
### Local ceilings (geode on localhost TCP, 100k seeded events, 4 cores)
|
||||
|
||||
| strategy | events/s | wall for 100k |
|
||||
|---|---|---|
|
||||
| quartz, 1 conn, one giant REQ | 476 | 210s (and 2 events silently missing) |
|
||||
| quartz, 4 conns time-sharded, giant REQs | 2,001 | 50s |
|
||||
| quartz, 1 conn, paged 1000/page | 12,480 | 8.0s |
|
||||
| **quartz, 4 conns time-sharded + paged** | **24,210** | **4.1s** |
|
||||
| raw socket (no parse), one giant REQ | ~676 | timed out at 120s (81k/100k) |
|
||||
|
||||
- **Giant single REQs are a trap.** The raw (no-parse) variant proves it's
|
||||
server-side: geode streams a 100k-event response at ~700 events/s
|
||||
(per-frame cost in the session pump / query streaming path — needs its own
|
||||
investigation, `WebSocketSessionPump`), and the quartz run came back 2
|
||||
events short. Public relays are worse: they clamp limits and may drop
|
||||
frames or the connection under output backpressure. Paged cursors through
|
||||
the very same server ran 26× faster.
|
||||
- **Sharding the `created_at` range across connections stacks with paging:**
|
||||
4 sharded + paged connections ≈ 2× one paged connection locally (24k/s),
|
||||
and each connection gets its own receiver coroutine, so parse parallelizes
|
||||
for free.
|
||||
|
||||
### Per-frame strategies, offline (50k frames, ~580B each)
|
||||
|
||||
| strategy | events/s | µs/frame |
|
||||
|---|---|---|
|
||||
| full `fromJsonToMessage`, 1 thread | 276k | 3.6 |
|
||||
| full parse, 4 threads | 657k | 1.5 (aggregate) |
|
||||
| id-scan only (archive raw, parse lazily) | 2.96M | 0.34 |
|
||||
|
||||
Parse of small events is ~3.6µs; the earlier production capture (mixed sizes,
|
||||
damus 8.7KB events) measured 32µs. Either way **parse is not the bottleneck
|
||||
for bulk download**: one core parses 10M small events in ~36s, and a 4-core
|
||||
fan-out does it in ~15s. The wire and the relay's page cadence dominate.
|
||||
|
||||
### Production test case: kinds=[30382] on nip85.nosfabrica.com (cap 20k)
|
||||
|
||||
| strategy | wall | events/s | notes |
|
||||
|---|---|---|---|
|
||||
| paged download (until-cursor) | 5.4s | 3,711 | 41 pages ≈ 500/page (relay clamp) |
|
||||
| NIP-77 negentropy sync | 12.7s | 1,575 | full set = 31,241 ids, 9 reconcile windows |
|
||||
|
||||
Negentropy correctly enumerated the whole 31,241-id set (splitting 9 windows
|
||||
around the relay's `max_sync_events` cap) and streamed id-batch REQs — but
|
||||
for a **cold** download it was 2.4× slower than plain paging: the reconcile
|
||||
rounds and by-id lookups cost more than sequential pages when you need
|
||||
*everything anyway*. Negentropy's win is **incremental re-sync**: once the
|
||||
10M events are local, the next sync transfers only fingerprints + the diff
|
||||
instead of re-paging the world. (kind 30382 events carry empty `content` —
|
||||
all data in tags — so the MB/s column reads 0.)
|
||||
|
||||
### Per-connection wall: relay pacing, not client parse (2026-07-03)
|
||||
|
||||
A user report proposed "parallel parse + strictly-ordered dispatch" on the
|
||||
theory that the per-connection serial parse caps a single connection at
|
||||
~4k events/s (fetch-by-id, 12 idle cores, no gain past ~4 concurrent REQs,
|
||||
throughput scaling only with connections). The discriminating test
|
||||
(`BulkDownloadBenchmark.perConnectionWall`) pages the same production query
|
||||
on one connection twice — once through a raw socket that does NO JSON parse
|
||||
(only an `indexOf` scan for `created_at`), once through the full quartz
|
||||
stack:
|
||||
|
||||
| | events/s (nip85.nosfabrica.com, kinds 30382, 20k cap) |
|
||||
|---|---|
|
||||
| raw socket, no parse | 3,382 |
|
||||
| full quartz stack, parse + dispatch | 3,754 |
|
||||
|
||||
Identical within noise. **The ~3.5–4k/s single-connection ceiling is the
|
||||
relay's per-connection page/response cadence, not client CPU** — at 465B
|
||||
events the parser is ~1% busy at this rate (single-thread parse ceiling:
|
||||
276k/s small events, 31k/s for the large-frame production mix). The same
|
||||
relay served our negentropy fetch-by-id at only 1.6k/s with 8 concurrent
|
||||
REQs and an idle client — also server-side. This matches every symptom in
|
||||
the report: per-connection pacing explains "more REQs on one connection
|
||||
don't help" and "throughput scales with connections" just as well as a
|
||||
client-side serial stage would, and "CPU idle" is what a network wall looks
|
||||
like.
|
||||
|
||||
Implications for the proposed parallel-parse pipeline:
|
||||
|
||||
- It will NOT lift the reported 4k/s ceiling on relays with this pacing; the
|
||||
proven lever is **more connections** (created_at- or id-range sharding).
|
||||
- It IS still worth having for the cases where the consumer genuinely
|
||||
saturates: large-frame relays (32µs/frame mix) on phones (5–10× slower
|
||||
cores → ~3–6k/s parse ceiling) being streamed by fast relays, and any
|
||||
setup where verify/consume stays inline on the consumer.
|
||||
- The proposal's ordering constraints are correct and non-negotiable:
|
||||
NEG-MSG/NEG-ERR handshake order, EOSE-only-after-its-EVENTs,
|
||||
per-subscription EVENT order, OK/CLOSED/NOTICE/AUTH stream order. A
|
||||
sequence-numbered reorder buffer after a parallel parse pool satisfies all
|
||||
four (dispatch order = receive order).
|
||||
- The bounded-pipeline half of the proposal is worth doing regardless — the
|
||||
UNLIMITED per-connection channel is the unbounded-memory risk under any
|
||||
slow consumer (see below), and bounding it converts overload into TCP
|
||||
backpressure. One caveat: blocking OkHttp's reader thread also delays its
|
||||
PING/PONG handling, so sustained backpressure can trip relay-side ping
|
||||
timeouts — the bound should be sized generously.
|
||||
|
||||
### Parallel fetch-by-id matrix — full corpus, no cap (2026-07-03)
|
||||
|
||||
**Harness:** `quartz/src/jvmTest/.../relay/prodbench/ByIdFetchBenchmark.kt`.
|
||||
Enumerates every kind-30382 id on nip85.nosfabrica.com (33,000 that day), then
|
||||
each cell re-downloads the FULL corpus by 250-id REQ batches from a shared
|
||||
queue. Connections are pre-established before timing; every cell completed
|
||||
with zero missing ids and zero timed-out batches.
|
||||
|
||||
| cell | events/s |
|
||||
|---|---|
|
||||
| 1 conn × 1 REQ | 1,970 |
|
||||
| 1 conn × 2 REQs | 3,423 |
|
||||
| 1 conn × 4 REQs | 8,026 |
|
||||
| 1 conn × 8 REQs | 17,214 |
|
||||
| 1 conn × 16 REQs | **30,735** |
|
||||
| 2 conns × 4 REQs | 15,557 |
|
||||
| 4 conns × 4 REQs | 27,251 |
|
||||
| 8 conns × 4 REQs | **40,402** |
|
||||
|
||||
This corrects both the user report AND this doc's own earlier "relay
|
||||
per-connection pacing" phrasing:
|
||||
|
||||
1. **No wall at ~4 concurrent REQs** — one connection scales near-linearly to
|
||||
16 in-flight REQs (1,970 → 30,735 events/s). The relay is happy to serve
|
||||
30k+/s down a single socket when the client keeps requests in flight.
|
||||
2. **Connections and per-connection REQs are interchangeable**: 1×16 ≈ 4×4.
|
||||
The real variable is **total in-flight REQs** (Little's law: throughput ≈
|
||||
in-flight ÷ per-REQ latency; a 250-id batch serves in ~130ms here).
|
||||
Diminishing returns start around 32 in-flight (~40k/s) — likely relay
|
||||
query capacity.
|
||||
3. **Every slow configuration measured so far is a serialization problem,
|
||||
not a bandwidth/CPU one**: until-cursor paging is 1-in-flight by
|
||||
construction (~2–3.7k/s ≈ the 1×1 cell); `negentropySync` measured
|
||||
1.6k/s with `maxConcurrentReqs = 8` because its download workers starve —
|
||||
the reconcile rounds pace id production (bounded `idBatches` buffer of
|
||||
`workerCount` batches) and overflow windows recurse **sequentially**
|
||||
(`syncWindow` lo-half then hi-half).
|
||||
4. The reported "stops helping past ~4 REQs" is consistent with
|
||||
`negentropySync`'s internal throttling (or a relay-side subscription cap,
|
||||
or a slow per-event sink) — not with independent by-id REQs on this relay.
|
||||
5. Client parse at 30.7k/s through ONE connection's consumer coroutine was
|
||||
~11% of a core (3.6µs × 30.7k) on this machine — still not the wall. On a
|
||||
phone (5–10× slower) 15–30k/s IS where the single consumer saturates, so
|
||||
the parallel-parse proposal becomes relevant on mobile at exactly the
|
||||
rates this fan-out unlocks.
|
||||
|
||||
**Concrete quartz improvements this implies** (in `negentropySync`):
|
||||
raise `maxConcurrentReqs` (16 measured safe here), deepen the `idBatches`
|
||||
buffer so downloads don't starve on reconcile cadence, and reconcile
|
||||
overflow-split windows concurrently instead of recursing sequentially.
|
||||
Together these should move it from 1.6k/s toward the ~30k/s the same
|
||||
connection demonstrably sustains. At 40k/s, 10M events by id is ~4 minutes
|
||||
(plus id enumeration).
|
||||
|
||||
### Extended matrix on a 2.6M corpus — saturation found (2026-07-03, later)
|
||||
|
||||
Between the two runs the relay was **backfilled from 33k to 2,628,328
|
||||
kind-30382 events** (it's an actively-loading NIP-85 dataset — cross-run
|
||||
comparisons must account for corpus size). Rerun with 125-id batches, cells
|
||||
capped at a 120s budget (all cells partial by design at this corpus size):
|
||||
|
||||
| cell | events/s | effective per-REQ latency |
|
||||
|---|---|---|
|
||||
| 1 conn × 1 REQ | 1,088 | ~115ms |
|
||||
| 1 conn × 4 REQs | 3,953 | ~126ms |
|
||||
| 1 conn × 8 REQs | 5,420 | ~184ms |
|
||||
| 1 conn × 16 REQs | 5,480 | ~365ms |
|
||||
| 1 conn × 20 REQs (relay cap) | 5,464 | ~458ms |
|
||||
| 2 conns × 10 REQs | 10,508 | ~285ms |
|
||||
| 4 conns × 10 REQs | **14,757** | ~542ms |
|
||||
| 8 conns × 10 REQs | 14,617 | ~1.1s |
|
||||
| 16 conns × 10 REQs | 14,921 | ~2.2s |
|
||||
|
||||
Also learned the hard way (first extended run wedged): the relay's NIP-11
|
||||
advertises `limitation.max_subscriptions = 20` — REQs 21+ on one connection
|
||||
aren't just rejected, they wedged the connection, and with the client's
|
||||
5-minute reconnect backoff every subsequent batch burned its full timeout.
|
||||
**A bulk downloader must read NIP-11 and stay under the caps.**
|
||||
|
||||
What changed vs the 33k-corpus run:
|
||||
|
||||
1. **On the big corpus there IS a per-connection wall — at ~8 in-flight
|
||||
REQs / ~5.5k events/s**, where the small corpus scaled linearly to 16 /
|
||||
30.7k/s. Past 8, added in-flight only inflates per-REQ latency (queueing,
|
||||
textbook Little's law). The user report's "stops helping past ~4" is
|
||||
therefore *scale-dependent*: wrong on a hot 33k dataset, roughly right
|
||||
(off by 2×) on a cold 2.6M one — by-id lookups on the larger index cost
|
||||
more and strfry's per-connection service saturates.
|
||||
2. **The relay-wide ceiling is ~15k events/s at ~40 total in-flight** (vs
|
||||
~40k/s on the small corpus). 4 conns × 10 REQs already reaches it; 16
|
||||
conns adds nothing but latency. More connections cannot beat the server's
|
||||
aggregate query capacity.
|
||||
3. Client-side cost remains negligible at these rates — every wall in this
|
||||
table is server-side.
|
||||
|
||||
Practical shape for a bulk-by-id downloader, from the data: read NIP-11 →
|
||||
open ~4 connections × ~10 REQs — and make concurrency **adaptive** (grow
|
||||
in-flight until events/s stops improving / per-REQ latency inflates, then
|
||||
back off), since the same relay's sweet spot moved 8× in one day as its
|
||||
dataset grew. At today's ~15k/s ceiling: 2.6M events ≈ 3 min; 10M ≈ 11 min.
|
||||
|
||||
### negentropySync pipelining (implemented, 2026-07-03)
|
||||
|
||||
`NostrClientNegentropySyncExt` was restructured around the findings — no
|
||||
NIP-11 auto-detection, the budget knobs are the caller's:
|
||||
|
||||
- **Global download-worker pool** spanning the whole sync (was per-window
|
||||
with a join between windows, so the connection idled while the next
|
||||
window's NEG rounds ran).
|
||||
- **`reconcileConcurrency` parameter** — overflow-split windows are
|
||||
reconciled by N concurrent NEG sessions from a shared work queue (was a
|
||||
strictly sequential lo-half/hi-half recursion). Default 1.
|
||||
- **`idBufferBatches` parameter** — depth of the bounded buffer between
|
||||
reconciliation and download workers (was hardcoded to `workerCount`).
|
||||
- KDoc contract: peak subscriptions = `maxConcurrentReqs +
|
||||
reconcileConcurrency + 1`; sizing it under the relay's
|
||||
`limitation.max_subscriptions` is the caller's responsibility.
|
||||
|
||||
All 44 negentropy tests (unit + end-to-end against the in-process relay +
|
||||
geode interop) pass unchanged. Production shootout on the 2.6M corpus,
|
||||
100k-event cap, same day, single connection
|
||||
(`BulkDownloadBenchmark.negentropyPipelineShootout`):
|
||||
|
||||
| variant | events/s | note |
|
||||
|---|---|---|
|
||||
| old implementation (previous day, 31k corpus) | 1,575 | not same-day comparable |
|
||||
| new pipeline, old-equivalent params (8 reqs, seq windows) | 3,601 | global pool already overlaps windows |
|
||||
| new pipeline, tuned (12 reqs, 4 reconcilers, 96-batch buffer) | **4,497** | 17 subs ≤ strfry's 20 cap |
|
||||
|
||||
Tuned reaches **~82% of the measured ~5.5k/s single-connection by-id
|
||||
ceiling**; the 4 reconcilers also produced ids 2.3× faster (need=171,812
|
||||
enumerated vs 109,336 in less wall time). The remaining gap is reconcile
|
||||
burstiness plus by-id query cost. Going past ~5.5k/s on this relay requires
|
||||
multiple connections (caller-level: disjoint `created_at` windows per
|
||||
client), which the matrix caps at ~15k/s relay-wide.
|
||||
|
||||
### negentropyReconcile: the composable half (implemented, 2026-07-03)
|
||||
|
||||
To let callers own the loading strategy (fan the download across several
|
||||
connections/clients, prioritize, or push local events up), the reconcile is
|
||||
now available standalone in `NostrClientNegentropySyncExt`:
|
||||
|
||||
- **`negentropyReconcile(relay, filter, localEntries, …)`** — pure NIP-77
|
||||
diff, no downloads/uploads. Streams both directions in `batchSize` chunks
|
||||
as rounds arrive: `onNeedIds` (relay has, local set lacks → download) and
|
||||
`onHaveIds` (local has, relay lacks → publish). Back-pressured: the
|
||||
callbacks suspend the reconcile round that produced them. Local state goes
|
||||
in as `List<IdAndTime>` (the same 40 B/entry projection
|
||||
`IEventStore.snapshotIdsForNegentropy` returns), and window splits slice
|
||||
it by `createdAt` via binary search, so both sides always reconcile the
|
||||
same timeline slice. Overflow windows split exactly like `negentropySync`,
|
||||
with the same `reconcileConcurrency` knob.
|
||||
- **`negentropyReconcileIds(…): NegentropyIdDiff`** — convenience that
|
||||
materializes `needIds` + `haveIds` lists (KDoc warns about heap on
|
||||
multi-million diffs; the streaming form is the scalable one).
|
||||
- `NegentropySession`'s primary constructor now takes `List<IdAndTime>`
|
||||
(JVM erasure forbids overloading on `List<Event>`); the old event-list
|
||||
form moved to `NegentropySession.fromEvents(…)`, mirroring
|
||||
`NegentropyServerSession` — call sites in `cli`, `geode` tests and
|
||||
`NegentropyManager` migrated. **API note for external quartz users:**
|
||||
`NegentropySession(subId, filter, events)` must become
|
||||
`NegentropySession.fromEvents(subId, filter, events)`.
|
||||
|
||||
`negentropySync` itself now delegates to the same window engine. Covered by
|
||||
`NostrClientNegentropyReconcileTest` (empty-local, partial overlap both
|
||||
ways, identical sets, batch streaming, since/until window slicing) — 49
|
||||
negentropy tests green.
|
||||
|
||||
**`amy sync` migrated onto it (2026-07-03):** the CLI's hand-rolled raw
|
||||
WebSocket negotiate loop (single un-windowed session; a strfry
|
||||
`max_sync_events` overflow was a hard error) is gone. `SyncCommand` now
|
||||
calls `negentropyReconcile` on amy's own `NostrClient` and pipelines the
|
||||
loop-closing: need-id batches feed 4 concurrent by-id `drain`s and have-ids
|
||||
feed an uploader, both overlapping the remaining reconcile rounds (peak 7
|
||||
subs on the relay, under the common cap of 20). Downloads still funnel
|
||||
through amy's verify-and-store path. Output field `rounds` (protocol
|
||||
round-trips) became `windows` (created_at splits). Verified end-to-end
|
||||
against two embedded geode relays: down-only 25/25, up-only 5/5, and
|
||||
bidirectional re-runs converge to a zero diff.
|
||||
|
||||
### Answer for "10M events from one relay, fastest"
|
||||
|
||||
At nosfabrica's measured page cadence, one connection ≈ 3.7k events/s → 10M
|
||||
in ~45 min. To improve, in order:
|
||||
|
||||
1. **Page with until-cursors — never one giant REQ** (silent drops, server
|
||||
slow paths, relay clamps).
|
||||
2. **Shard the `created_at` range across K connections** to the same relay
|
||||
(each shard pages independently; no cursor dependency between shards).
|
||||
Local: 2× at K=4; WAN, where RTT dominates page turnaround, closer to
|
||||
linear until the relay rate-limits per-IP.
|
||||
3. **Don't optimize parse first** — it's 2–5% of the budget. If the goal is
|
||||
archival, id-scan + store raw frames (0.34µs/frame) and parse lazily
|
||||
in parallel later.
|
||||
4. **Use negentropy for the second sync onward**, not the first.
|
||||
5. **Memory:** the per-connection channel is UNLIMITED — at 10M events a
|
||||
sink slower than the socket accumulates heap without bound. A bulk
|
||||
downloader should bound the channel (blocking the OkHttp reader thread is
|
||||
fine — that's TCP backpressure doing its job) and stream events to disk,
|
||||
never hold the set.
|
||||
|
||||
## Implemented optimizations (2026-07-03)
|
||||
|
||||
### PoolRequests lock sharding (done)
|
||||
|
||||
The global spin lock moved into `RequestSubscriptionState` — one lock per
|
||||
subscription — since every compound mutation is scoped to one subId and
|
||||
different subs share no wire state. `decideCommandLocked` now takes the state
|
||||
instance so it never re-enters the (non-reentrant) lock, and the all-subs
|
||||
iterations (connect/disconnect/cannot-connect) lock one sub at a time.
|
||||
`withLock` is inline to keep the per-EVENT hot path allocation-free.
|
||||
|
||||
Validation (DispatchStageBenchmark, PoolRequests-only, 4 feeder threads vs
|
||||
1): scaling flipped from **negative** (11.1M → 3.6M deliveries/s, 0.33×) to
|
||||
**positive** (3.4–7.3× across runs; absolute numbers on the shared CI box are
|
||||
noisy, the scaling direction is consistent across every run). All
|
||||
PoolRequests concurrency + NostrClient + negentropy suites pass.
|
||||
|
||||
### CachingEventDecoder: duplicates skip the re-parse (done)
|
||||
|
||||
`BasicRelayClient`'s per-frame decode step is now a pluggable
|
||||
`MessageDecoder` (default: full parse, unchanged). The opt-in
|
||||
`CachingEventDecoder` scans an EVENT frame for its id (~0.3µs; the scan is
|
||||
JSON-escape-safe, so reposts embedding event JSON in `content` can't confuse
|
||||
it, and ANY irregularity falls back to a full parse) and, on a hit in its
|
||||
generational id→Event cache, synthesizes the `EventMessage` from the
|
||||
already-parsed `Event` with the frame's own subId. Dispatch semantics are
|
||||
IDENTICAL — every subscription still gets its delivery, per-relay
|
||||
bookkeeping still happens — only the redundant parse is skipped. Wire it via
|
||||
`NostrClient(builder, decoder = CachingEventDecoder())`.
|
||||
|
||||
Validation (`DedupDecodeBenchmark`, 60k frames, 67% duplicates — the
|
||||
production-measured dup share is 14–57%): full parse 10.0µs/frame vs caching
|
||||
decoder 1.5µs/frame — **6.5×**, with duplicates costing ~0.3µs instead of
|
||||
10µs. 7 scan-safety unit tests (`CachingEventDecoderTest`) cover repost
|
||||
embedding, escaped subIds, malformed ids, cache rotation, non-EVENT frames.
|
||||
|
||||
Follow-up (2026-07-03): the decoder's spin lock was replaced with lock-free
|
||||
concurrent maps — a new `ConcurrentHashCache` expect/actual
|
||||
(`ConcurrentHashMap` on JVM/Android, `CacheMap` on Apple, copy-on-write on
|
||||
Linux, mirroring `LargeCache`'s per-platform choices) plus atomic counters.
|
||||
Rotation races are deliberately tolerated (documented: every failure mode is
|
||||
a redundant re-parse, never a wrong message).
|
||||
`CachingEventDecoderConcurrencyTest` hammers one shared decoder from 8
|
||||
threads (80k duplicate-heavy frames, capacity 256 so rotations fire
|
||||
constantly) and asserts zero wrong messages with exact counter accounting;
|
||||
the benchmark assertion still holds post-refactor.
|
||||
|
||||
Same-JVM A/B against the resurrected spin-lock implementation (3 rounds,
|
||||
interleaved): **1 thread — parity** (0.79–1.02×, noise around 1.0; the
|
||||
uncontended lock was already ~free), **8 threads sharing one decoder —
|
||||
lock-free wins 3.0–4.5×** (e.g. 127ms → 28ms for 128k frames). The lock was
|
||||
serializing the concurrent hit path exactly the way the old global
|
||||
PoolRequests lock did; ConcurrentHashMap removes it.
|
||||
|
||||
**Adopted in all four front-end clients (2026-07-03):** the Android app's
|
||||
pool (`AppModules.kt`), the Android crawl client
|
||||
(`AccountViewModel.buildCrawlClient` — Event Sync / Cashu discovery, the
|
||||
duplicate-heaviest path), the desktop `RelayConnectionManager`, and amy's
|
||||
`Context`. Each passes `CachingEventDecoder()` at `NostrClient`
|
||||
construction; no other behavior change.
|
||||
|
||||
### Bounded per-connection receive buffer (REVERTED — deliberate design choice)
|
||||
|
||||
A 4096-frame bound on the reader→consumer channel was tried (backpressure
|
||||
via TCP flow control instead of unbounded heap; proven lossless by a
|
||||
slow-consumer test) and then **reverted by explicit maintainer decision**,
|
||||
for two reasons that outweigh the heap tail-risk:
|
||||
|
||||
1. **The remote infrastructure isn't ours.** Backpressure parks the backlog
|
||||
in the RELAY's outbound buffers/TCP window — a client should release the
|
||||
relay from its duties as fast as the relay can send, and own the
|
||||
buffering itself.
|
||||
2. **The app holds 2000+ simultaneous relay connections.** A bounded buffer
|
||||
under a slow consumer blocks OkHttp reader THREADS; at that connection
|
||||
count, blocked readers are a thread-starvation hazard far worse than the
|
||||
heap growth they prevent.
|
||||
|
||||
The `Channel.UNLIMITED` receive queues now carry an explicit "do not bound"
|
||||
comment with this rationale. The slow-consumer risk is addressed from the
|
||||
other side instead: make the consumer fast enough that backlogs don't form
|
||||
(CachingEventDecoder, ParallelEventVerifier, PoolRequests sharding).
|
||||
|
||||
### ParallelEventVerifier: batched verify off the receiver coroutine (done)
|
||||
|
||||
The client-side mirror of `IngestQueue.parallelVerify`:
|
||||
`accessories/ParallelEventVerifier`. `submit(event, context)` is a cheap
|
||||
bounded-channel send from the receiver coroutine; a drain loop batches
|
||||
greedily (up to 256) and fans each batch across `Dispatchers.Default` in
|
||||
core-sized CHUNKS — a per-event `async` measured ~40µs/event of scheduling
|
||||
overhead that swallowed the entire parallel gain, and the per-batch
|
||||
fork/join barrier at batch 64 still cost half of it (64 → 1.2×, 256 → 2.2×,
|
||||
1024 → 3.2× on 4 cores). Callbacks dispatch in submission order, a
|
||||
`preVerified` hook short-circuits already-trusted ids, and the bounded
|
||||
submit channel backpressures the socket.
|
||||
|
||||
Also learned while measuring: `Event` caches derived state after its first
|
||||
verification, so any verify benchmark must use disjoint event sets per pass
|
||||
(the first version accidentally measured warm objects and reported 0.94–1.2×).
|
||||
|
||||
Validation: `ParallelVerifyBenchmark` (4k fresh signed events, 8 signers) —
|
||||
sequential 50.6µs/event vs pipeline 28.1µs/event, **1.8×** on the noisy
|
||||
4-core CI box with an in-benchmark ≥1.5× assertion (crypto-only parallel
|
||||
ceiling measured 2.89× there). 4 correctness tests: valid/tampered routing,
|
||||
submission-order preservation, preVerified short-circuit, callback-crash
|
||||
resilience.
|
||||
|
||||
**App-side scope correction (2026-07-03): do NOT wire this into
|
||||
`CacheClientConnector`.** Code review confirmed the mobile app's
|
||||
verification is ALREADY parallel across relays: each connection's
|
||||
`OkHttpWebSocket` owns its own consumer coroutine on `Dispatchers.IO`, and
|
||||
nothing downstream funnels — `NostrClient.onIncomingMessage` →
|
||||
`EventCollector` → `LocalCache.justConsume` → `justVerify` all run inline
|
||||
on the delivering connection's coroutine, `LargeCache` is a
|
||||
`ConcurrentSkipListMap`, and the one global serializer (`PoolRequests`'
|
||||
lock) is now per-subscription. With 2000+ connections a multi-relay burst
|
||||
already saturates the cores; the verifier would only reshuffle the work.
|
||||
Its real scope is where a SINGLE connection's stream serializes verifies:
|
||||
bulk flows (`fetchAllPages`-style downloads, each client's leg inside
|
||||
`negentropySyncFanOut`, amy/CLI syncs). One second-order note for a future
|
||||
on-device profile: today's verifies run on `Dispatchers.IO` (64 threads on
|
||||
~8 phone cores), so a large burst can oversubscribe cores with CPU-bound
|
||||
crypto and starve the IO pool — if profiling ever shows that, moving verify
|
||||
to a core-bounded dispatcher (which this accessory does) is the fix.
|
||||
|
||||
### negentropySyncFanOut: multi-connection sync (done)
|
||||
|
||||
`accessories/NostrClientNegentropyFanOutExt`: one reconcile feeding by-id
|
||||
download batches to N clients (one socket each) × `reqsPerClient` workers,
|
||||
with reconcile windows ALSO round-robined across the connections. Events
|
||||
funnel through a bounded channel to a single consumer (exact `maxEvents`,
|
||||
single-threaded `onEvent`); everything backpressures; `localEntries` diffing
|
||||
and have-reporting work as in `negentropyReconcile`. 4 in-process
|
||||
multi-client tests (full download, cap, local-diff, single-client
|
||||
degradation) — 18 negentropy tests green.
|
||||
|
||||
Production shootout (100k cap, kind 30382, same-run pairs): fan-out 4×8 vs
|
||||
tuned single client measured **+18% to +64%** (7.3k vs 6.1k/s; 7.5k vs
|
||||
4.5k/s) with heavy relay-side variance — the relay is being actively
|
||||
backfilled and its NEG snapshot caching favors whichever variant runs
|
||||
second. The honest bound: **this relay produces reconcile ids at only
|
||||
~9–10k/s regardless of connection count** (server-side snapshot build), so
|
||||
end-to-end fan-out gains cap there even though the download stage alone
|
||||
scales 2.7× with connections (the by-id matrix). On download-bound syncs
|
||||
(bigger events, relays with faster NEG production, phone-CPU clients) the
|
||||
gain approaches the matrix ratio.
|
||||
|
||||
**Descoped with evidence — adaptive in-flight control (#6):** an AIMD window
|
||||
on download REQs would idle-down against today's reconcile-bound relay and
|
||||
the benchmarks could not demonstrate a win (the criterion for shipping);
|
||||
revisit if a relay shows download-bound behavior with volatile capacity.
|
||||
|
||||
### permessage-deflate: verified active (no change needed)
|
||||
|
||||
The per-connection-wall test now prints the negotiated
|
||||
`Sec-WebSocket-Extensions`; against nip85.nosfabrica.com (strfry) OkHttp
|
||||
negotiates `permessage-deflate; client_no_context_takeover` out of the box.
|
||||
Compression was on the suspect list as a 3–5× bandwidth lever for mobile —
|
||||
it's already engaged; nothing to code.
|
||||
|
||||
### Geode giant-REQ crawl + wedge: fixed (two masked bugs)
|
||||
|
||||
The "giant single REQ streams at ~700 events/s and came back 2 events
|
||||
short" finding decomposed into two server bugs hiding each other:
|
||||
|
||||
1. **O(n²) replay dedup** (`LiveEventStore.query`): the historical-replay
|
||||
dedup set was an immutable Kotlin `Set` under an `AtomicReference` with
|
||||
copy-on-add — `set + id` copies the whole set per streamed event (100k
|
||||
events ≈ 5 billion hash inserts). Fixed with a spin-lock-guarded mutable
|
||||
`HashSet` (same read/write threads as before; lock sections are a single
|
||||
contains/add).
|
||||
2. **Session-pump slow-client policy misfire** (`WebSocketSessionPump`):
|
||||
with the throttle gone, a fast replay instantly overflowed the
|
||||
8192-frame backlog cap — which conflated "client is slow" with "replay
|
||||
outruns the socket writer" (normal for bulk) — and the "drop" only
|
||||
closed the internal queue, leaving the socket half-dead: the client hung
|
||||
with no EOSE/close and silently missed the response tail (the likely
|
||||
story behind the original 99,998/100,000). Now the producer is PACED
|
||||
against a full backlog (bounded blocking wait for the writer, consistent
|
||||
with the documented ingest-fanout behavior) and only a client still
|
||||
behind after 30s is dropped — by actually cancelling the socket.
|
||||
|
||||
Validation: `GiantReqStreamTest` (committed regression guard) — 20k-event
|
||||
single REQ over a real socket: pre-fix 8.4s (~2.4k events/s; 100k measured
|
||||
476–700/s), post-fix **0.7s (~27k events/s)**, all 20k delivered with EOSE.
|
||||
96 geode tests + quartz relay/server suites green.
|
||||
|
||||
## Recommendations (in order of value/risk)
|
||||
|
||||
1. **Move Schnorr verification off the receiver coroutine** in the app's
|
||||
consume path (`CacheClientConnector` → `LocalCache.justConsume`): a bounded
|
||||
verify stage on `Dispatchers.Default` (batch + `async` per batch, or a small
|
||||
worker pool), mirroring `IngestQueue`. Ordering constraint: dedup must stay
|
||||
before verify (it already is), and consumers of `justConsume`'s return value
|
||||
need an async-tolerant path.
|
||||
2. **Keep parse on the receiver coroutine** (32µs/msg is cheap and keeps
|
||||
message ordering per subscription), but consider skipping re-serialization
|
||||
work for duplicates (57% of metadata-burst traffic never needs more than an
|
||||
id lookup).
|
||||
3. **Fix the `PoolRequests` spin lock's negative scaling** — the dispatch
|
||||
microbenchmark shows 4 concurrent relay consumers deliver *less* than one
|
||||
thread through it. Either replace the busy-wait with `synchronized`/a
|
||||
parking lock, or shard the state by subId, and consider the early-dedup
|
||||
short-circuit so duplicate frames never enter the locked path at all.
|
||||
4. Re-run this harness on-device (the same class compiles for Android
|
||||
instrumentation with minor changes) before/after any fix; the offline
|
||||
ceilings section gives the numbers to compare.
|
||||
Vendored
+40
@@ -0,0 +1,40 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils.cache
|
||||
|
||||
import io.github.charlietap.cachemap.cacheMapOf
|
||||
|
||||
actual class ConcurrentHashCache<K : Any, V : Any> {
|
||||
private val map = cacheMapOf<K, V>()
|
||||
|
||||
actual fun get(key: K): V? = map[key]
|
||||
|
||||
actual fun put(
|
||||
key: K,
|
||||
value: V,
|
||||
) {
|
||||
map.put(key, value)
|
||||
}
|
||||
|
||||
actual fun size(): Int = map.size
|
||||
|
||||
actual fun clear() = map.clear()
|
||||
}
|
||||
+10
-1
@@ -30,6 +30,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.pool.RelayPool
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.IRelayClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.Message
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.MessageDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.AuthCmd
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.CloseCmd
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.Command
|
||||
@@ -80,10 +81,18 @@ import kotlinx.coroutines.launch
|
||||
class NostrClient(
|
||||
private val websocketBuilder: WebsocketBuilder,
|
||||
private val parentScope: CoroutineScope = CoroutineScope(Dispatchers.IO + SupervisorJob()),
|
||||
/**
|
||||
* Per-frame decode strategy shared by every relay connection. Pass a
|
||||
* [com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder]
|
||||
* to skip re-parsing duplicate EVENT frames (the same event arriving via
|
||||
* another subscription or another relay reuses the already-parsed Event;
|
||||
* dispatch semantics are unchanged).
|
||||
*/
|
||||
decoder: MessageDecoder = MessageDecoder.Default,
|
||||
) : INostrClient,
|
||||
RelayConnectionListener,
|
||||
AutoCloseable {
|
||||
private val relayPool: RelayPool = RelayPool(websocketBuilder, this)
|
||||
private val relayPool: RelayPool = RelayPool(websocketBuilder, this, decoder)
|
||||
|
||||
/** Scope for all subscriptions. */
|
||||
private val scope = CoroutineScope(parentScope.coroutineContext + SupervisorJob())
|
||||
|
||||
+39
@@ -0,0 +1,39 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.client.accessories
|
||||
|
||||
/**
|
||||
* Outcome of a [negentropySyncFanOut] run.
|
||||
*
|
||||
* @property needCount ids the relay reported missing locally (full set size when uncapped).
|
||||
* @property haveCount ids the local set has that the relay lacks (only counted when
|
||||
* [negentropySyncFanOut]'s `localEntries` is non-empty).
|
||||
* @property downloaded distinct events delivered through `onEvent`.
|
||||
* @property windows `created_at` windows the reconcile split into.
|
||||
* @property connections how many clients actually served download batches.
|
||||
*/
|
||||
class NegentropyFanOutResult(
|
||||
val needCount: Int,
|
||||
val haveCount: Int,
|
||||
val downloaded: Int,
|
||||
val windows: Int,
|
||||
val connections: Int,
|
||||
)
|
||||
+198
@@ -0,0 +1,198 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.client.accessories
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.core.HexKey
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.INostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import kotlinx.coroutines.coroutineScope
|
||||
import kotlinx.coroutines.ensureActive
|
||||
import kotlinx.coroutines.joinAll
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlin.concurrent.atomics.AtomicInt
|
||||
import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
import kotlin.concurrent.atomics.incrementAndFetch
|
||||
import kotlin.coroutines.coroutineContext
|
||||
|
||||
/**
|
||||
* [negentropySync] scaled past the single-connection ceiling: ONE reconcile
|
||||
* (on `clients[0]`) feeding by-id download batches to EVERY client in
|
||||
* [clients] — each client being its own socket to the same [relay].
|
||||
*
|
||||
* Why this shape: a relay serves a single connection only so fast — the
|
||||
* production by-id matrix measured ~5.5k events/s per connection (saturating
|
||||
* at ~8 in-flight REQs) but ~15k events/s across 4 connections; see
|
||||
* `quartz/plans/2026-07-02-nostrclient-receiver-perf.md`. Reconciliation is
|
||||
* cheap and stays on one connection; the downloads are what need the fan-out.
|
||||
*
|
||||
* Mechanics:
|
||||
* - `clients[0]` runs [negentropyReconcile] (windows, overflow splits,
|
||||
* optional [localEntries] diffing — all identical semantics), streaming
|
||||
* need-id batches into a bounded queue;
|
||||
* - every client runs [reqsPerClient] download workers pulling batches off
|
||||
* that shared queue ([fetchByIds]: one REQ per batch, collected to EOSE);
|
||||
* - events funnel through a bounded channel to a single consumer, so
|
||||
* [onEvent] runs single-threaded and [maxEvents] is exact; the bounded
|
||||
* stages back-pressure all the way to the reconcile.
|
||||
*
|
||||
* Reconcile windows round-robin across the clients too (a single connection
|
||||
* produced need-ids at only ~9k/s on a 2.6M corpus and starved the download
|
||||
* workers). Subscription budget per connection is the caller's job (relays
|
||||
* cap concurrent subs; strfry defaults to 20): each client holds at peak
|
||||
* `reqsPerClient + ceil(reconcileConcurrency / clients.size) + 1`.
|
||||
*
|
||||
* Passing a single-element [clients] degrades gracefully to (pipelined)
|
||||
* [negentropySync] behavior. Duplicate clients in the list would just share
|
||||
* a connection — use distinct instances.
|
||||
*
|
||||
* @throws NegentropySyncException when a window cannot be reconciled (same
|
||||
* contract as [negentropySync]); the whole fan-out is cancelled.
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
suspend fun negentropySyncFanOut(
|
||||
clients: List<INostrClient>,
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime> = emptyList(),
|
||||
maxEvents: Int = 0,
|
||||
reqsPerClient: Int = 10,
|
||||
fetchBatch: Int = 250,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 2,
|
||||
onProgress: ((needSoFar: Int, downloaded: Int) -> Unit)? = null,
|
||||
onEvent: (Event) -> Unit,
|
||||
): NegentropyFanOutResult {
|
||||
require(clients.isNotEmpty()) { "at least one client is required" }
|
||||
|
||||
val need = AtomicInt(0)
|
||||
val have = AtomicInt(0)
|
||||
val windows = AtomicInt(0)
|
||||
val used = AtomicInt(0)
|
||||
var downloaded = 0
|
||||
|
||||
// Pin every client's connection open for the whole run: between two
|
||||
// batches (or two reconcile windows) a client briefly has no live
|
||||
// subscription and its pool would otherwise drop the relay.
|
||||
val keepAlives =
|
||||
clients.map { client ->
|
||||
val subId = newSubId()
|
||||
client.subscribe(subId, mapOf(relay to listOf(Filter(ids = listOf(KEEP_ALIVE_ID)))), null)
|
||||
client to subId
|
||||
}
|
||||
|
||||
try {
|
||||
coroutineScope {
|
||||
// Reconcile → download handoff. Bounded so download saturation
|
||||
// back-pressures the reconcile instead of piling up ids.
|
||||
val idBatches = Channel<List<HexKey>>(clients.size * reqsPerClient * 2)
|
||||
|
||||
// Download → consumer funnel; single consumer keeps onEvent
|
||||
// single-threaded and the maxEvents cap exact.
|
||||
val events = Channel<Event>(DELIVERY_BUFFER)
|
||||
|
||||
val workers =
|
||||
clients.flatMap { client ->
|
||||
List(reqsPerClient.coerceAtLeast(1)) {
|
||||
launch {
|
||||
var servedAny = false
|
||||
for (batch in idBatches) {
|
||||
coroutineContext.ensureActive()
|
||||
if (!servedAny) {
|
||||
servedAny = true
|
||||
used.incrementAndFetch()
|
||||
}
|
||||
for (event in client.fetchByIds(relay, batch, idleTimeoutMs)) {
|
||||
events.send(event)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val producer =
|
||||
launch {
|
||||
try {
|
||||
val sorted =
|
||||
if (localEntries.size > 1) localEntries.sortedBy { it.createdAt } else localEntries
|
||||
// Windows reconcile round-robin ACROSS the clients so
|
||||
// server-side snapshot builds parallelize per connection
|
||||
// (a single connection produced ids at only ~9k/s and
|
||||
// starved the download workers).
|
||||
reconcileWindows(
|
||||
clients = clients,
|
||||
relay = relay,
|
||||
filter = filter,
|
||||
localEntries = sorted,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
batchSize = fetchBatch,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
onWindow = { windows.incrementAndFetch() },
|
||||
onNeed = {
|
||||
need.addAndFetch(it)
|
||||
},
|
||||
onHave = { have.addAndFetch(it) },
|
||||
sendNeedBatch = { batch -> idBatches.send(batch) },
|
||||
sendHaveBatch = if (localEntries.isEmpty()) null else { _ -> },
|
||||
)
|
||||
} finally {
|
||||
idBatches.close()
|
||||
}
|
||||
}
|
||||
|
||||
val closer =
|
||||
launch {
|
||||
producer.join()
|
||||
workers.joinAll()
|
||||
events.close()
|
||||
}
|
||||
|
||||
for (event in events) {
|
||||
downloaded++
|
||||
onEvent(event)
|
||||
onProgress?.invoke(need.load(), downloaded)
|
||||
if (maxEvents in 1..downloaded) break
|
||||
}
|
||||
|
||||
// Cap reached (or producer done): stop everything still running.
|
||||
producer.cancel()
|
||||
workers.forEach { it.cancel() }
|
||||
closer.cancel()
|
||||
}
|
||||
} finally {
|
||||
keepAlives.forEach { (client, subId) -> client.unsubscribe(subId) }
|
||||
}
|
||||
|
||||
return NegentropyFanOutResult(
|
||||
needCount = need.load(),
|
||||
haveCount = have.load(),
|
||||
downloaded = downloaded,
|
||||
windows = windows.load(),
|
||||
connections = used.load(),
|
||||
)
|
||||
}
|
||||
|
||||
/** Bounded buffer between download workers and the single delivery consumer. */
|
||||
private const val DELIVERY_BUFFER = 256
|
||||
+445
-94
@@ -31,6 +31,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.Message
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import com.vitorpamplona.quartz.nip77Negentropy.NegErrMessage
|
||||
import com.vitorpamplona.quartz.nip77Negentropy.NegMsgMessage
|
||||
import com.vitorpamplona.quartz.nip77Negentropy.NegentropySession
|
||||
@@ -41,8 +42,14 @@ import kotlinx.coroutines.ensureActive
|
||||
import kotlinx.coroutines.flow.first
|
||||
import kotlinx.coroutines.joinAll
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.sync.Mutex
|
||||
import kotlinx.coroutines.sync.withLock
|
||||
import kotlinx.coroutines.withTimeoutOrNull
|
||||
import kotlin.concurrent.Volatile
|
||||
import kotlin.concurrent.atomics.AtomicInt
|
||||
import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
import kotlin.concurrent.atomics.decrementAndFetch
|
||||
import kotlin.concurrent.atomics.incrementAndFetch
|
||||
import kotlin.coroutines.coroutineContext
|
||||
import kotlin.math.min
|
||||
import kotlin.time.TimeSource
|
||||
@@ -121,9 +128,26 @@ class NegentropySyncResult(
|
||||
* it and the disconnect is turned into a clean abort. Pass `0` to disable the
|
||||
* watchdog entirely and run until the socket drops (download batches keep a finite
|
||||
* internal idle bound regardless, so a single stuck batch can't hang the pipeline).
|
||||
* @param reconcileConcurrency how many `created_at` windows are reconciled at once
|
||||
* after an over-cap split (each holds one NEG session on the connection). `1`
|
||||
* reproduces the old strictly-sequential window walk. Raising it overlaps the
|
||||
* reconcile round-trips of one window with another's — the reconcile cadence is
|
||||
* what starves the download workers on large sets.
|
||||
* @param idBufferBatches depth (in batches of [fetchBatch] ids) of the buffer
|
||||
* between reconciliation and the download workers. Deeper keeps the workers fed
|
||||
* across a window's round-trip gaps; memory is bounded by
|
||||
* `idBufferBatches * fetchBatch` ids.
|
||||
*
|
||||
* **Subscription budget is the caller's job:** at peak this method holds
|
||||
* `maxConcurrentReqs + reconcileConcurrency + 1` (keep-alive) subscriptions on
|
||||
* the connection. Relays cap concurrent subscriptions per connection (NIP-11
|
||||
* `limitation.max_subscriptions`; e.g. strfry defaults to 20) and exceeding the
|
||||
* cap can wedge the connection, not just fail the extra REQ — size the two knobs
|
||||
* to fit the target relay.
|
||||
* @param onProgress optional `(needSoFar, downloaded)` ticks as work proceeds.
|
||||
* @param onEvent called once per distinct event, on the relay reader thread.
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
suspend fun INostrClient.negentropySync(
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
@@ -131,11 +155,13 @@ suspend fun INostrClient.negentropySync(
|
||||
maxConcurrentReqs: Int = 8,
|
||||
fetchBatch: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
idBufferBatches: Int = maxConcurrentReqs * 4,
|
||||
onProgress: ((needSoFar: Int, downloaded: Int) -> Unit)? = null,
|
||||
onEvent: (Event) -> Unit,
|
||||
): NegentropySyncResult {
|
||||
var need = 0
|
||||
var windows = 0
|
||||
val need = AtomicInt(0)
|
||||
val windows = AtomicInt(0)
|
||||
var downloaded = 0
|
||||
|
||||
// Pin the relay in the pool's "desired" set for the whole sync. A NEG-OPEN is not
|
||||
@@ -156,17 +182,19 @@ suspend fun INostrClient.negentropySync(
|
||||
val producer =
|
||||
launch {
|
||||
try {
|
||||
syncWindow(
|
||||
syncPipeline(
|
||||
relay = relay,
|
||||
filter = filter,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
fetchBatch = fetchBatch,
|
||||
maxConcurrentReqs = maxConcurrentReqs,
|
||||
onWindow = { windows++ },
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
idBufferBatches = idBufferBatches,
|
||||
onWindow = { windows.incrementAndFetch() },
|
||||
// Only accumulate here; progress is reported from the
|
||||
// single consumer loop below so the user callback is never
|
||||
// invoked from two coroutines at once.
|
||||
onNeed = { need += it },
|
||||
onNeed = { need.addAndFetch(it) },
|
||||
deliver = { events.send(it) },
|
||||
)
|
||||
} finally {
|
||||
@@ -177,7 +205,7 @@ suspend fun INostrClient.negentropySync(
|
||||
for (event in events) {
|
||||
downloaded++
|
||||
onEvent(event)
|
||||
onProgress?.invoke(need, downloaded)
|
||||
onProgress?.invoke(need.load(), downloaded)
|
||||
if (maxEvents in 1..downloaded) break
|
||||
}
|
||||
|
||||
@@ -190,10 +218,10 @@ suspend fun INostrClient.negentropySync(
|
||||
}
|
||||
|
||||
return NegentropySyncResult(
|
||||
needCount = need,
|
||||
needCount = need.load(),
|
||||
haveCount = 0,
|
||||
downloaded = downloaded,
|
||||
windows = windows,
|
||||
windows = windows.load(),
|
||||
)
|
||||
}
|
||||
|
||||
@@ -204,6 +232,8 @@ suspend fun INostrClient.negentropySync(
|
||||
maxConcurrentReqs: Int = 8,
|
||||
fetchBatch: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
idBufferBatches: Int = maxConcurrentReqs * 4,
|
||||
onProgress: ((needSoFar: Int, downloaded: Int) -> Unit)? = null,
|
||||
onEvent: (Event) -> Unit,
|
||||
): NegentropySyncResult =
|
||||
@@ -214,6 +244,8 @@ suspend fun INostrClient.negentropySync(
|
||||
maxConcurrentReqs = maxConcurrentReqs,
|
||||
fetchBatch = fetchBatch,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
idBufferBatches = idBufferBatches,
|
||||
onProgress = onProgress,
|
||||
onEvent = onEvent,
|
||||
)
|
||||
@@ -257,6 +289,8 @@ suspend fun INostrClient.negentropySyncOrFetch(
|
||||
maxConcurrentReqs: Int = 8,
|
||||
fetchBatch: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
idBufferBatches: Int = maxConcurrentReqs * 4,
|
||||
onProgress: ((needSoFar: Int, downloaded: Int) -> Unit)? = null,
|
||||
onEvent: (Event) -> Unit,
|
||||
): NegentropyOrFetchResult {
|
||||
@@ -283,6 +317,8 @@ suspend fun INostrClient.negentropySyncOrFetch(
|
||||
maxConcurrentReqs = maxConcurrentReqs,
|
||||
fetchBatch = fetchBatch,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
idBufferBatches = idBufferBatches,
|
||||
onProgress = onProgress,
|
||||
) { accept(it) }
|
||||
NegentropyOrFetchResult(delivered, pagedFallback = false, negentropy = result, fallbackCause = null)
|
||||
@@ -306,6 +342,8 @@ suspend fun INostrClient.negentropySyncOrFetch(
|
||||
maxConcurrentReqs: Int = 8,
|
||||
fetchBatch: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
idBufferBatches: Int = maxConcurrentReqs * 4,
|
||||
onProgress: ((needSoFar: Int, downloaded: Int) -> Unit)? = null,
|
||||
onEvent: (Event) -> Unit,
|
||||
): NegentropyOrFetchResult =
|
||||
@@ -316,60 +354,214 @@ suspend fun INostrClient.negentropySyncOrFetch(
|
||||
maxConcurrentReqs = maxConcurrentReqs,
|
||||
fetchBatch = fetchBatch,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
idBufferBatches = idBufferBatches,
|
||||
onProgress = onProgress,
|
||||
onEvent = onEvent,
|
||||
)
|
||||
|
||||
/**
|
||||
* Recursively reconciles [filter] for [relay], splitting by `created_at` windows
|
||||
* whenever the relay rejects the set as too large, and downloading the ids of each
|
||||
* window as it resolves. Runs on a single coroutine; [deliver] funnels events out.
|
||||
* The whole-sync pipeline: a single pool of [maxConcurrentReqs] download workers
|
||||
* fed by up to [reconcileConcurrency] concurrent window reconciliations through a
|
||||
* bounded buffer of [idBufferBatches] id batches.
|
||||
*
|
||||
* Why this shape (measured on a 2.6M-event production set — see
|
||||
* `quartz/plans/2026-07-02-nostrclient-receiver-perf.md`):
|
||||
*
|
||||
* - The workers are GLOBAL, not per-window: with the old per-window pool the
|
||||
* next window's reconcile round-trips only started after the previous
|
||||
* window's last download joined, so the connection idled between windows.
|
||||
* - Overflow-split windows go into a shared work queue processed by
|
||||
* [reconcileConcurrency] reconcilers, overlapping their NEG round-trips.
|
||||
* Each active reconcile holds one NEG session on the connection.
|
||||
* - The id buffer decouples the bursty reconcile id stream from the steady
|
||||
* download stream; it stays bounded so a slow consumer still back-pressures
|
||||
* the relay (memory is O(idBufferBatches × fetchBatch) ids).
|
||||
*
|
||||
* Throws [NegentropySyncException] for any window negentropy cannot reconcile (a
|
||||
* minimal window still over the cap, or an unavailable/erroring relay).
|
||||
* minimal window still over the cap, or an unavailable/erroring relay); the
|
||||
* failure cancels the whole pipeline.
|
||||
*/
|
||||
private suspend fun INostrClient.syncWindow(
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
private suspend fun INostrClient.syncPipeline(
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
idleTimeoutMs: Long,
|
||||
fetchBatch: Int,
|
||||
maxConcurrentReqs: Int,
|
||||
reconcileConcurrency: Int,
|
||||
idBufferBatches: Int,
|
||||
onWindow: () -> Unit,
|
||||
onNeed: (Int) -> Unit,
|
||||
deliver: suspend (Event) -> Unit,
|
||||
) {
|
||||
coroutineContext.ensureActive()
|
||||
) = coroutineScope {
|
||||
val idBatches = Channel<List<HexKey>>(idBufferBatches.coerceAtLeast(1))
|
||||
|
||||
when (val outcome = downloadWindow(relay, filter, idleTimeoutMs, fetchBatch, maxConcurrentReqs, onNeed, deliver)) {
|
||||
is ReconcileOutcome.Complete -> onWindow()
|
||||
|
||||
is ReconcileOutcome.Overflow -> {
|
||||
val lo = filter.since ?: 0L
|
||||
val hi = filter.until ?: TimeUtils.now()
|
||||
if (hi - lo <= MIN_WINDOW_SECONDS) {
|
||||
// A minimal window that still overflows: negentropy genuinely can't
|
||||
// enumerate this slice. Surface it — paging is the caller's call.
|
||||
throw NegentropySyncException(
|
||||
relay = relay,
|
||||
window = filter,
|
||||
reason = NegentropySyncException.Reason.OVER_MAX_SYNC_EVENTS,
|
||||
detail = "created_at window [$lo, $hi] still exceeds the relay's max_sync_events",
|
||||
)
|
||||
} else {
|
||||
val mid = lo + (hi - lo) / 2
|
||||
syncWindow(relay, filter.copy(since = lo, until = mid), idleTimeoutMs, fetchBatch, maxConcurrentReqs, onWindow, onNeed, deliver)
|
||||
syncWindow(relay, filter.copy(since = mid + 1, until = hi), idleTimeoutMs, fetchBatch, maxConcurrentReqs, onWindow, onNeed, deliver)
|
||||
val workers =
|
||||
List(maxConcurrentReqs.coerceAtLeast(1)) {
|
||||
launch {
|
||||
for (batch in idBatches) {
|
||||
coroutineContext.ensureActive()
|
||||
for (event in fetchByIds(relay, batch, idleTimeoutMs)) {
|
||||
deliver(event)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
is ReconcileOutcome.Failed ->
|
||||
throw NegentropySyncException(
|
||||
relay = relay,
|
||||
window = filter,
|
||||
reason = NegentropySyncException.Reason.UNAVAILABLE,
|
||||
detail = outcome.detail,
|
||||
)
|
||||
reconcileWindows(
|
||||
clients = listOf(this@syncPipeline),
|
||||
relay = relay,
|
||||
filter = filter,
|
||||
localEntries = emptyList(),
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
batchSize = fetchBatch,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
onWindow = onWindow,
|
||||
onNeed = onNeed,
|
||||
onHave = {},
|
||||
sendNeedBatch = { batch -> idBatches.send(batch) },
|
||||
sendHaveBatch = null,
|
||||
)
|
||||
|
||||
idBatches.close()
|
||||
workers.joinAll()
|
||||
}
|
||||
|
||||
/**
|
||||
* The shared window engine behind [negentropySync] and [negentropyReconcile]:
|
||||
* reconciles [filter] against [localEntries], splitting into `created_at`
|
||||
* windows whenever the relay rejects the set as too large, with up to
|
||||
* [reconcileConcurrency] windows reconciling at once from a shared work
|
||||
* queue. Each window's local subset is sliced out of [localEntries] (which
|
||||
* MUST be sorted by `createdAt`) so both sides always reconcile the same
|
||||
* slice of the timeline.
|
||||
*
|
||||
* Throws [NegentropySyncException] for any window negentropy cannot reconcile
|
||||
* (a minimal window still over the cap, or an unavailable/erroring relay); the
|
||||
* failure cancels the whole scope.
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
internal suspend fun reconcileWindows(
|
||||
// one or more connections to the SAME relay; reconciler i runs its NEG
|
||||
// sessions on clients[i % size], so concurrent windows spread across
|
||||
// connections (server-side snapshot builds are paced per connection)
|
||||
clients: List<INostrClient>,
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime>,
|
||||
idleTimeoutMs: Long,
|
||||
batchSize: Int,
|
||||
reconcileConcurrency: Int,
|
||||
onWindow: () -> Unit,
|
||||
onNeed: (Int) -> Unit,
|
||||
onHave: (Int) -> Unit,
|
||||
sendNeedBatch: suspend (List<HexKey>) -> Unit,
|
||||
sendHaveBatch: (suspend (List<HexKey>) -> Unit)?,
|
||||
) = coroutineScope {
|
||||
// Windows waiting for (or under) reconciliation. UNLIMITED so a reconciler
|
||||
// re-queueing an overflow split never suspends while holding queue capacity
|
||||
// (the split fan-out is tiny: two Filters per overflow).
|
||||
val pending = Channel<Filter>(Channel.UNLIMITED)
|
||||
|
||||
// Queued-or-running windows. An overflow replaces one window with two
|
||||
// (net +1); a completion is -1; the queue closes when it hits zero.
|
||||
val remaining = AtomicInt(1)
|
||||
pending.send(filter)
|
||||
|
||||
val reconcilers =
|
||||
List(reconcileConcurrency.coerceAtLeast(1)) { reconcilerIndex ->
|
||||
launch {
|
||||
val client = clients[reconcilerIndex % clients.size]
|
||||
for (window in pending) {
|
||||
coroutineContext.ensureActive()
|
||||
|
||||
val outcome =
|
||||
client.reconcileStreaming(
|
||||
relay = relay,
|
||||
filter = window,
|
||||
localEntries = entriesForWindow(localEntries, window.since, window.until),
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
fetchBatch = batchSize,
|
||||
onNeed = onNeed,
|
||||
onHave = onHave,
|
||||
sendNeedBatch = sendNeedBatch,
|
||||
sendHaveBatch = sendHaveBatch,
|
||||
)
|
||||
|
||||
when (outcome) {
|
||||
is ReconcileOutcome.Complete -> {
|
||||
onWindow()
|
||||
if (remaining.decrementAndFetch() == 0) pending.close()
|
||||
}
|
||||
|
||||
is ReconcileOutcome.Overflow -> {
|
||||
val lo = window.since ?: 0L
|
||||
val hi = window.until ?: TimeUtils.now()
|
||||
if (hi - lo <= MIN_WINDOW_SECONDS) {
|
||||
// A minimal window that still overflows: negentropy
|
||||
// genuinely can't enumerate this slice. Surface it —
|
||||
// paging is the caller's call.
|
||||
throw NegentropySyncException(
|
||||
relay = relay,
|
||||
window = window,
|
||||
reason = NegentropySyncException.Reason.OVER_MAX_SYNC_EVENTS,
|
||||
detail = "created_at window [$lo, $hi] still exceeds the relay's max_sync_events",
|
||||
)
|
||||
}
|
||||
val mid = lo + (hi - lo) / 2
|
||||
remaining.incrementAndFetch()
|
||||
pending.send(window.copy(since = lo, until = mid))
|
||||
pending.send(window.copy(since = mid + 1, until = hi))
|
||||
}
|
||||
|
||||
is ReconcileOutcome.Failed ->
|
||||
throw NegentropySyncException(
|
||||
relay = relay,
|
||||
window = window,
|
||||
reason = NegentropySyncException.Reason.UNAVAILABLE,
|
||||
detail = outcome.detail,
|
||||
)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
reconcilers.joinAll()
|
||||
}
|
||||
|
||||
/**
|
||||
* The `createdAt`-range slice of [sorted] (ascending by `createdAt`) that
|
||||
* belongs to the window `[since, until]` (both inclusive, NIP-01 semantics).
|
||||
* Binary-searched so window splits stay O(log n) over multi-million local sets.
|
||||
*/
|
||||
private fun entriesForWindow(
|
||||
sorted: List<IdAndTime>,
|
||||
since: Long?,
|
||||
until: Long?,
|
||||
): List<IdAndTime> {
|
||||
if (sorted.isEmpty() || (since == null && until == null)) return sorted
|
||||
|
||||
val lo = since ?: 0L
|
||||
val hi = until ?: Long.MAX_VALUE
|
||||
|
||||
// first index with createdAt >= lo
|
||||
var start = 0
|
||||
var e = sorted.size
|
||||
while (start < e) {
|
||||
val mid = (start + e) ushr 1
|
||||
if (sorted[mid].createdAt < lo) start = mid + 1 else e = mid
|
||||
}
|
||||
|
||||
// first index with createdAt > hi
|
||||
var end = start
|
||||
e = sorted.size
|
||||
while (end < e) {
|
||||
val mid = (end + e) ushr 1
|
||||
if (sorted[mid].createdAt <= hi) end = mid + 1 else e = mid
|
||||
}
|
||||
|
||||
return if (start >= end) emptyList() else sorted.subList(start, end)
|
||||
}
|
||||
|
||||
private sealed interface ReconcileOutcome {
|
||||
@@ -386,11 +578,203 @@ private sealed interface ReconcileOutcome {
|
||||
}
|
||||
|
||||
/**
|
||||
* Drives one NIP-77 reconciliation of [filter] against an EMPTY local set, sending
|
||||
* Outcome of a [negentropyReconcile] run.
|
||||
*
|
||||
* @property needCount ids the relay has that the local set lacks (streamed to `onNeedIds`).
|
||||
* @property haveCount ids the local set has that the relay lacks (streamed to `onHaveIds`).
|
||||
* @property windows number of `created_at` windows the reconcile split into.
|
||||
*/
|
||||
class NegentropyReconcileResult(
|
||||
val needCount: Int,
|
||||
val haveCount: Int,
|
||||
val windows: Int,
|
||||
)
|
||||
|
||||
/**
|
||||
* Pure NIP-77 reconciliation — no downloads, no uploads. Diffs the relay's
|
||||
* matched set for [filter] against [localEntries] and streams the two
|
||||
* directions of the diff to the caller, who decides what to do with them:
|
||||
*
|
||||
* - **need ids** (`onNeedIds`): the relay has them, the local set doesn't —
|
||||
* fetch them however fits (own REQ fan-out across several connections or
|
||||
* clients, batching, prioritization, …). The by-id fetch matrix in
|
||||
* `quartz/plans/2026-07-02-nostrclient-receiver-perf.md` is the map for
|
||||
* that fan-out.
|
||||
* - **have ids** (`onHaveIds`): the local set has them, the relay doesn't —
|
||||
* publish the corresponding events to push the relay up to date.
|
||||
*
|
||||
* Compared to [negentropySync], which couples the reconcile to a built-in
|
||||
* by-id downloader on the same connection, this is the composable half:
|
||||
* `negentropyReconcile` + caller-side loading is how to sync faster than one
|
||||
* connection allows.
|
||||
*
|
||||
* Ids are streamed in chunks of [batchSize] as reconcile rounds arrive — the
|
||||
* full id set is never materialized here (accumulate them yourself or use
|
||||
* [negentropyReconcileIds] when the set is known to be small). Both callbacks
|
||||
* suspend the reconcile round that produced them, so a slow consumer
|
||||
* back-pressures the relay. With [reconcileConcurrency] > 1 the callbacks are
|
||||
* invoked from that many coroutines concurrently.
|
||||
*
|
||||
* Relay-side overflow (strfry `max_sync_events`) is handled by `created_at`
|
||||
* window splitting, like [negentropySync]. [localEntries] may be in any order;
|
||||
* each window reconciles against the matching `createdAt` slice. The caller
|
||||
* is responsible for [localEntries] actually being the local events matching
|
||||
* [filter] — entries outside the filter would show up as false "have" ids.
|
||||
*
|
||||
* Holds `reconcileConcurrency` NEG sessions plus one keep-alive subscription
|
||||
* on the connection; budget that against the relay's
|
||||
* `limitation.max_subscriptions`.
|
||||
*
|
||||
* @throws NegentropySyncException when a window cannot be reconciled via NIP-77.
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
suspend fun INostrClient.negentropyReconcile(
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime> = emptyList(),
|
||||
batchSize: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
onHaveIds: (suspend (List<HexKey>) -> Unit)? = null,
|
||||
onNeedIds: suspend (List<HexKey>) -> Unit,
|
||||
): NegentropyReconcileResult {
|
||||
val need = AtomicInt(0)
|
||||
val have = AtomicInt(0)
|
||||
val windows = AtomicInt(0)
|
||||
|
||||
// Same connection-pinning trick as negentropySync: a NEG-OPEN is not a REQ,
|
||||
// so without a live subscription the pool would consider the relay unwanted
|
||||
// and disconnect it mid-reconcile.
|
||||
val keepAliveSubId = newSubId()
|
||||
subscribe(keepAliveSubId, mapOf(relay to listOf(Filter(ids = listOf(KEEP_ALIVE_ID)))), null)
|
||||
try {
|
||||
val sorted =
|
||||
if (localEntries.size > 1) {
|
||||
localEntries.sortedBy { it.createdAt }
|
||||
} else {
|
||||
localEntries
|
||||
}
|
||||
|
||||
reconcileWindows(
|
||||
clients = listOf(this),
|
||||
relay = relay,
|
||||
filter = filter,
|
||||
localEntries = sorted,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
batchSize = batchSize,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
onWindow = { windows.incrementAndFetch() },
|
||||
onNeed = { need.addAndFetch(it) },
|
||||
onHave = { have.addAndFetch(it) },
|
||||
sendNeedBatch = onNeedIds,
|
||||
sendHaveBatch = onHaveIds,
|
||||
)
|
||||
} finally {
|
||||
unsubscribe(keepAliveSubId)
|
||||
}
|
||||
|
||||
return NegentropyReconcileResult(
|
||||
needCount = need.load(),
|
||||
haveCount = have.load(),
|
||||
windows = windows.load(),
|
||||
)
|
||||
}
|
||||
|
||||
suspend fun INostrClient.negentropyReconcile(
|
||||
relay: String,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime> = emptyList(),
|
||||
batchSize: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
onHaveIds: (suspend (List<HexKey>) -> Unit)? = null,
|
||||
onNeedIds: suspend (List<HexKey>) -> Unit,
|
||||
): NegentropyReconcileResult =
|
||||
negentropyReconcile(
|
||||
relay = RelayUrlNormalizer.normalize(relay),
|
||||
filter = filter,
|
||||
localEntries = localEntries,
|
||||
batchSize = batchSize,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
onHaveIds = onHaveIds,
|
||||
onNeedIds = onNeedIds,
|
||||
)
|
||||
|
||||
/**
|
||||
* The full id diff from a [negentropyReconcile] run, materialized.
|
||||
*
|
||||
* @property needIds relay has them, the local set doesn't — download these.
|
||||
* @property haveIds local set has them, the relay doesn't — publish these.
|
||||
* @property windows number of `created_at` windows the reconcile split into.
|
||||
*/
|
||||
class NegentropyIdDiff(
|
||||
val needIds: List<HexKey>,
|
||||
val haveIds: List<HexKey>,
|
||||
val windows: Int,
|
||||
)
|
||||
|
||||
/**
|
||||
* Convenience over [negentropyReconcile] that accumulates both directions of
|
||||
* the diff and returns them as lists.
|
||||
*
|
||||
* Materializes the FULL diff in memory: at ~100 B per id string a
|
||||
* million-id diff is ~100 MB of heap. Fine for bounded sets (per-author
|
||||
* sync, recent windows); for open-ended bulk syncs prefer the streaming
|
||||
* [negentropyReconcile] and consume batches as they arrive.
|
||||
*/
|
||||
suspend fun INostrClient.negentropyReconcileIds(
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime> = emptyList(),
|
||||
batchSize: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
): NegentropyIdDiff {
|
||||
val lock = Mutex()
|
||||
val needIds = ArrayList<HexKey>()
|
||||
val haveIds = ArrayList<HexKey>()
|
||||
|
||||
val result =
|
||||
negentropyReconcile(
|
||||
relay = relay,
|
||||
filter = filter,
|
||||
localEntries = localEntries,
|
||||
batchSize = batchSize,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
onHaveIds = { batch -> lock.withLock { haveIds.addAll(batch) } },
|
||||
onNeedIds = { batch -> lock.withLock { needIds.addAll(batch) } },
|
||||
)
|
||||
|
||||
return NegentropyIdDiff(needIds, haveIds, result.windows)
|
||||
}
|
||||
|
||||
suspend fun INostrClient.negentropyReconcileIds(
|
||||
relay: String,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime> = emptyList(),
|
||||
batchSize: Int = 500,
|
||||
idleTimeoutMs: Long = 120_000L,
|
||||
reconcileConcurrency: Int = 1,
|
||||
): NegentropyIdDiff =
|
||||
negentropyReconcileIds(
|
||||
relay = RelayUrlNormalizer.normalize(relay),
|
||||
filter = filter,
|
||||
localEntries = localEntries,
|
||||
batchSize = batchSize,
|
||||
idleTimeoutMs = idleTimeoutMs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
)
|
||||
|
||||
/**
|
||||
* Drives one NIP-77 reconciliation of [filter] against [localEntries], sending
|
||||
* `NEG-OPEN` and walking the rounds itself (rather than via [NegentropyManager]) so
|
||||
* it can apply back-pressure: each round's `needIds` are handed to [sendBatch] —
|
||||
* it can apply back-pressure: each round's `needIds` are handed to [sendNeedBatch] —
|
||||
* which suspends while the download queue is full — *before* the next round is
|
||||
* acked, so the relay's id stream is paced to the downloader and never piles up.
|
||||
* When [sendHaveBatch] is non-null the ids the relay LACKS (we have them locally)
|
||||
* are streamed through it the same way.
|
||||
*
|
||||
* The ids are streamed, not returned; the result is only the terminal outcome.
|
||||
* Always sends `NEG-CLOSE` and removes the listener on the way out.
|
||||
@@ -398,15 +782,18 @@ private sealed interface ReconcileOutcome {
|
||||
private suspend fun INostrClient.reconcileStreaming(
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
localEntries: List<IdAndTime>,
|
||||
idleTimeoutMs: Long,
|
||||
fetchBatch: Int,
|
||||
onNeed: (Int) -> Unit,
|
||||
sendBatch: suspend (List<HexKey>) -> Unit,
|
||||
onHave: (Int) -> Unit,
|
||||
sendNeedBatch: suspend (List<HexKey>) -> Unit,
|
||||
sendHaveBatch: (suspend (List<HexKey>) -> Unit)?,
|
||||
): ReconcileOutcome {
|
||||
val targetUrl = relay
|
||||
val relayClient = getOrCreateRelay(relay)
|
||||
val subId = newSubId()
|
||||
val session = NegentropySession(subId, filter, localEvents = emptyList())
|
||||
val session = NegentropySession(subId, filter, localEntries = localEntries)
|
||||
|
||||
// Reader-thread → driver hand-off. Holds at most one frame: the relay only sends
|
||||
// the next one once we ack, and we ack only after this round's ids are queued.
|
||||
@@ -492,7 +879,17 @@ private suspend fun INostrClient.reconcileStreaming(
|
||||
val end = min(i + fetchBatch, needIds.size)
|
||||
// Copy each batch so the frame's full id list can be freed
|
||||
// as soon as it is chunked; suspends under back-pressure.
|
||||
sendBatch(ArrayList(needIds.subList(i, end)))
|
||||
sendNeedBatch(ArrayList(needIds.subList(i, end)))
|
||||
i = end
|
||||
}
|
||||
}
|
||||
val haveIds = result.haveIds
|
||||
if (haveIds.isNotEmpty() && sendHaveBatch != null) {
|
||||
onHave(haveIds.size)
|
||||
var i = 0
|
||||
while (i < haveIds.size) {
|
||||
val end = min(i + fetchBatch, haveIds.size)
|
||||
sendHaveBatch(ArrayList(haveIds.subList(i, end)))
|
||||
i = end
|
||||
}
|
||||
}
|
||||
@@ -533,52 +930,6 @@ private fun isOverflow(reason: String): Boolean =
|
||||
reason.contains("too many", ignoreCase = true) ||
|
||||
reason.startsWith("blocked", ignoreCase = true)
|
||||
|
||||
/**
|
||||
* Reconciles [filter] and streams its ids straight into a bounded download pool, so
|
||||
* reconciliation and download overlap and peak memory stays independent of the
|
||||
* window's size. At most [maxConcurrentReqs] `REQ`s of [fetchBatch] ids are open at
|
||||
* once; the id queue is bounded so a slow download back-pressures reconciliation.
|
||||
* Returns the terminal [ReconcileOutcome]; events go out through [deliver].
|
||||
*/
|
||||
private suspend fun INostrClient.downloadWindow(
|
||||
relay: NormalizedRelayUrl,
|
||||
filter: Filter,
|
||||
idleTimeoutMs: Long,
|
||||
fetchBatch: Int,
|
||||
maxConcurrentReqs: Int,
|
||||
onNeed: (Int) -> Unit,
|
||||
deliver: suspend (Event) -> Unit,
|
||||
): ReconcileOutcome =
|
||||
coroutineScope {
|
||||
val workerCount = maxConcurrentReqs.coerceAtLeast(1)
|
||||
|
||||
// Bounded: when full, reconcileStreaming suspends instead of letting the
|
||||
// relay's id stream accumulate. This is what keeps memory O(pipeline), not
|
||||
// O(window).
|
||||
val idBatches = Channel<List<HexKey>>(workerCount)
|
||||
|
||||
val workers =
|
||||
List(workerCount) {
|
||||
launch {
|
||||
for (batch in idBatches) {
|
||||
coroutineContext.ensureActive()
|
||||
for (event in fetchByIds(relay, batch, idleTimeoutMs)) {
|
||||
deliver(event)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val outcome =
|
||||
reconcileStreaming(relay, filter, idleTimeoutMs, fetchBatch, onNeed) { batch ->
|
||||
idBatches.send(batch)
|
||||
}
|
||||
|
||||
idBatches.close()
|
||||
workers.joinAll()
|
||||
outcome
|
||||
}
|
||||
|
||||
/**
|
||||
* One `REQ` for [batch] ids; collects the matching events and returns them on
|
||||
* `EOSE`/close/timeout. All events for a single relay arrive on its one reader
|
||||
@@ -590,7 +941,7 @@ private suspend fun INostrClient.downloadWindow(
|
||||
* replay the batch; without this the same event would be delivered twice. We rely on
|
||||
* NIP-77 yielding a distinct id set across batches, so no global dedup is needed.
|
||||
*/
|
||||
private suspend fun INostrClient.fetchByIds(
|
||||
internal suspend fun INostrClient.fetchByIds(
|
||||
relay: NormalizedRelayUrl,
|
||||
batch: List<HexKey>,
|
||||
idleTimeoutMs: Long,
|
||||
@@ -665,7 +1016,7 @@ private const val DELIVERY_BUFFER = 256
|
||||
* duration of a sync. Synthetic/real event ids are SHA-256 digests, so this never
|
||||
* collides with an actual event.
|
||||
*/
|
||||
private val KEEP_ALIVE_ID = "f".repeat(64)
|
||||
internal val KEEP_ALIVE_ID = "f".repeat(64)
|
||||
|
||||
/**
|
||||
* Finite fallback bounds (ms) for the two waits that must stay bounded even when the
|
||||
|
||||
+212
@@ -0,0 +1,212 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.client.accessories
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.crypto.verify
|
||||
import com.vitorpamplona.quartz.utils.Log
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.async
|
||||
import kotlinx.coroutines.awaitAll
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import kotlinx.coroutines.channels.ClosedReceiveChannelException
|
||||
import kotlinx.coroutines.channels.trySendBlocking
|
||||
import kotlinx.coroutines.coroutineScope
|
||||
import kotlinx.coroutines.launch
|
||||
|
||||
/**
|
||||
* Client-side batched Schnorr verification, off the relay receiver coroutine.
|
||||
*
|
||||
* Today the app verifies every new event INLINE on the per-relay consumer
|
||||
* coroutine (`CacheClientConnector` → `LocalCache.justConsume` →
|
||||
* `justVerify`): at ~75µs a verify (JVM; several hundred µs on phone cores)
|
||||
* a burst of a few thousand events serializes seconds of CPU behind that
|
||||
* relay's EOSEs, OKs and live events. Verification is embarrassingly
|
||||
* parallel — the relay-server side already exploits that
|
||||
* ([com.vitorpamplona.quartz.nip01Core.relay.server.backend.IngestQueue]'s
|
||||
* `parallelVerify`) — this is the same batching strategy for the client:
|
||||
*
|
||||
* - [submit] is cheap (a channel send) and returns immediately, freeing the
|
||||
* receiver coroutine for the next frame;
|
||||
* - one drain coroutine pulls a batch greedily (up to [maxBatch]), fans the
|
||||
* batch's verifies across [Dispatchers.Default], then dispatches
|
||||
* [onVerified]/[onInvalid] in submission order (batching the fan-out
|
||||
* beats per-event handoff: a per-event channel send measured ~180ns of
|
||||
* overhead each, a 64-batch is ~free — see DispatchStageBenchmark);
|
||||
* - [preVerified] lets the caller short-circuit events it already trusts
|
||||
* (LocalCache-style dedup: an id already consumed doesn't need another
|
||||
* verify) without paying the CPU;
|
||||
* - the submit channel is bounded: a verify backlog beyond [capacity]
|
||||
* blocks the submitting receiver coroutine, which backpressures the
|
||||
* socket instead of growing the heap.
|
||||
*
|
||||
* Typical wiring (replacing an inline-verify EventCollector):
|
||||
* ```
|
||||
* val verifier = ParallelEventVerifier(scope, onVerified = { event, relay ->
|
||||
* cache.justConsume(event, relay, wasVerified = true)
|
||||
* })
|
||||
* val collector = EventCollector(client) { event, relay ->
|
||||
* if (!cache.hasConsumed(event.id)) verifier.submit(event, relay)
|
||||
* }
|
||||
* ```
|
||||
*
|
||||
* Ordering note: submission order is preserved END-TO-END for callbacks (the
|
||||
* drain loop dispatches each batch in order, batches are sequential), so
|
||||
* per-relay, per-subscription arrival order survives the parallel stage.
|
||||
*/
|
||||
class ParallelEventVerifier<C>(
|
||||
scope: CoroutineScope,
|
||||
private val maxBatch: Int = DEFAULT_MAX_BATCH,
|
||||
capacity: Int = DEFAULT_CAPACITY,
|
||||
private val parallelism: Int = DEFAULT_PARALLELISM,
|
||||
private val preVerified: (Event) -> Boolean = { false },
|
||||
private val onInvalid: ((Event, C) -> Unit)? = null,
|
||||
private val onVerified: (Event, C) -> Unit,
|
||||
) : AutoCloseable {
|
||||
private class Pending<C>(
|
||||
val event: Event,
|
||||
val context: C,
|
||||
)
|
||||
|
||||
private val incoming = Channel<Pending<C>>(capacity)
|
||||
|
||||
var verifiedCount: Long = 0
|
||||
private set
|
||||
var invalidCount: Long = 0
|
||||
private set
|
||||
|
||||
private val drainJob =
|
||||
scope.launch(Dispatchers.Default) {
|
||||
val batch = ArrayList<Pending<C>>(maxBatch)
|
||||
try {
|
||||
while (true) {
|
||||
// Block for the first item, then drain greedily so
|
||||
// back-to-back submissions coalesce into one parallel batch.
|
||||
batch.add(incoming.receive())
|
||||
while (batch.size < maxBatch) {
|
||||
val next = incoming.tryReceive().getOrNull() ?: break
|
||||
batch.add(next)
|
||||
}
|
||||
processBatch(batch)
|
||||
batch.clear()
|
||||
}
|
||||
} catch (_: ClosedReceiveChannelException) {
|
||||
// normal shutdown via close()
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Hands one event to the verify stage. Returns fast (a channel send);
|
||||
* blocks only when the verify backlog exceeds the capacity — that is the
|
||||
* backpressure path, by design.
|
||||
*/
|
||||
fun submit(
|
||||
event: Event,
|
||||
context: C,
|
||||
) {
|
||||
incoming.trySendBlocking(Pending(event, context))
|
||||
}
|
||||
|
||||
private suspend fun processBatch(batch: List<Pending<C>>) {
|
||||
val results = BooleanArray(batch.size)
|
||||
if (batch.size == 1) {
|
||||
results[0] = verifyOne(batch[0].event)
|
||||
} else {
|
||||
// One worker per core-sized CHUNK, not per event: a per-event
|
||||
// async measured ~40µs/event end-to-end (scheduling overhead
|
||||
// swallowing the parallel gain) vs near-linear scaling with
|
||||
// chunked workers — same pattern as the offline verify benchmark.
|
||||
coroutineScope {
|
||||
val workers = minOf(parallelism, batch.size)
|
||||
val chunkSize = (batch.size + workers - 1) / workers
|
||||
(0 until workers)
|
||||
.map { w ->
|
||||
async(Dispatchers.Default) {
|
||||
var i = w * chunkSize
|
||||
val end = minOf(i + chunkSize, batch.size)
|
||||
while (i < end) {
|
||||
results[i] = verifyOne(batch[i].event)
|
||||
i++
|
||||
}
|
||||
}
|
||||
}.awaitAll()
|
||||
}
|
||||
}
|
||||
|
||||
// Dispatch in submission order, outside the parallel section.
|
||||
for (i in batch.indices) {
|
||||
val pending = batch[i]
|
||||
try {
|
||||
if (results[i]) {
|
||||
verifiedCount++
|
||||
onVerified(pending.event, pending.context)
|
||||
} else {
|
||||
invalidCount++
|
||||
onInvalid?.invoke(pending.event, pending.context)
|
||||
}
|
||||
} catch (e: Throwable) {
|
||||
// a misbehaving callback must not kill the drain loop
|
||||
Log.w("ParallelEventVerifier") { "callback threw for ${pending.event.id}: ${e.message}" }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private fun verifyOne(event: Event): Boolean = preVerified(event) || event.verify()
|
||||
|
||||
/**
|
||||
* Stops accepting submissions; the drain loop finishes the batches already
|
||||
* queued and ends. Cancelling the [CoroutineScope] passed to the
|
||||
* constructor stops everything immediately instead.
|
||||
*/
|
||||
override fun close() {
|
||||
incoming.close()
|
||||
}
|
||||
|
||||
/** Completed once the drain loop exits (all queued batches dispatched). */
|
||||
suspend fun join() = drainJob.join()
|
||||
|
||||
companion object {
|
||||
/**
|
||||
* Measured sweet spot (see ParallelVerifyBenchmark + the plan doc):
|
||||
* the per-batch fork/join barrier costs real throughput at small
|
||||
* batches (64 → ~1.2× total speedup; 256 → ~2.2×; 1024 → ~3.2× on 4
|
||||
* cores) while batch size bounds callback latency (256 verifies ≈
|
||||
* 3ms on a JVM box, ~10ms on a phone). Note the batch only grows
|
||||
* when a backlog exists — a light trickle takes the single-event
|
||||
* fast path with minimal latency — so this cap matters exactly when
|
||||
* throughput matters.
|
||||
*/
|
||||
const val DEFAULT_MAX_BATCH: Int = 256
|
||||
|
||||
/** Submissions in flight before [submit] blocks the caller. */
|
||||
const val DEFAULT_CAPACITY: Int = 8192
|
||||
|
||||
/**
|
||||
* Verify workers per batch. Dispatchers.Default caps actual CPU
|
||||
* parallelism at the core count anyway; workers beyond it just queue,
|
||||
* so a fixed value covering common phone/desktop core counts is fine
|
||||
* (commonMain has no portable core-count API). Callers that know
|
||||
* their hardware can pass an exact value.
|
||||
*/
|
||||
const val DEFAULT_PARALLELISM: Int = 8
|
||||
}
|
||||
}
|
||||
+68
-85
@@ -35,8 +35,6 @@ import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.utils.cache.LargeCache
|
||||
import kotlinx.coroutines.flow.MutableStateFlow
|
||||
import kotlin.concurrent.atomics.AtomicBoolean
|
||||
import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
|
||||
/**
|
||||
* Manages relay subscriptions for the entire pool in a way that only
|
||||
@@ -45,7 +43,6 @@ import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
* This code also awaits a subscription to come to EOSE since many relays
|
||||
* have through switching subs while they are processing the past.
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
class PoolRequests {
|
||||
/**
|
||||
* Desired subs and listeners
|
||||
@@ -67,41 +64,29 @@ class PoolRequests {
|
||||
|
||||
fun subState(subId: String): RequestSubscriptionState<NormalizedRelayUrl> = relayState.getOrCreate(subId) { RequestSubscriptionState() }
|
||||
|
||||
/**
|
||||
* Serializes every access to the subscription state machine
|
||||
* ([RequestSubscriptionState]) and the "should I send a REQ?" decision.
|
||||
/*
|
||||
* Locking model: every compound access to a subscription's state machine
|
||||
* ([RequestSubscriptionState]) — including the check-then-send decision in
|
||||
* [decideCommandLocked] — runs inside THAT subscription's own lock
|
||||
* ([RequestSubscriptionState.withLock]).
|
||||
*
|
||||
* A single subscription can span many relays, and each relay's
|
||||
* socket-reader thread delivers messages into this class concurrently while
|
||||
* the app thread adds/removes subscriptions — so the plain maps inside
|
||||
* [RequestSubscriptionState] are written from several threads at once. That
|
||||
* is both a memory hazard (concurrent map mutation) and, more importantly,
|
||||
* a logic hazard: the check-then-send in [decideCommandLocked] must be
|
||||
* atomic, otherwise two threads can both observe "no REQ in flight" and both
|
||||
* send a REQ for the same sub id.
|
||||
* socket-reader thread delivers messages into this class concurrently
|
||||
* while the app thread adds/removes subscriptions — so the plain maps
|
||||
* inside [RequestSubscriptionState] are written from several threads at
|
||||
* once. That is both a memory hazard (concurrent map mutation) and a
|
||||
* logic hazard: two threads must never both observe "no REQ in flight"
|
||||
* and both send a REQ for the same sub id.
|
||||
*
|
||||
* This is a tiny non-reentrant spin lock (the same [AtomicBoolean] primitive
|
||||
* used by BasicRelayClient's connecting mutex): the critical sections are a
|
||||
* handful of map operations, never any I/O. Listener callbacks and the
|
||||
* actual socket sends are ALWAYS performed outside the lock — they re-enter
|
||||
* this class through [onSent], so holding the lock across them would
|
||||
* self-deadlock.
|
||||
* The lock is per subscription, not global, because different subIds
|
||||
* share no wire state: EVENT frames for different subs coming from
|
||||
* different relay consumer threads must not serialize on each other (a
|
||||
* global lock here measured negative scaling under 4 concurrent relay
|
||||
* feeders). Listener callbacks and the actual socket sends are ALWAYS
|
||||
* performed outside the lock — they re-enter this class through
|
||||
* [onSent], so holding the lock across them would self-deadlock, and the
|
||||
* lock is non-reentrant. Never hold two subscriptions' locks at once.
|
||||
*/
|
||||
private val stateLock = AtomicBoolean(false)
|
||||
|
||||
private inline fun <R> withStateLock(block: () -> R): R {
|
||||
while (stateLock.exchange(true)) {
|
||||
// Another thread holds the lock. Spin-read until it looks free
|
||||
// (test-and-test-and-set: cheaper on the cache line than hammering
|
||||
// exchange) then retry the acquisition above.
|
||||
while (stateLock.load()) { }
|
||||
}
|
||||
try {
|
||||
return block()
|
||||
} finally {
|
||||
stateLock.store(false)
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* This is called when a sub is added or removed from this class and
|
||||
@@ -188,11 +173,9 @@ class PoolRequests {
|
||||
* When a connecting, updates the state of all subs
|
||||
*/
|
||||
fun onConnecting(url: NormalizedRelayUrl) {
|
||||
// Change states to connecting.
|
||||
withStateLock {
|
||||
relayState.forEach { subId, state ->
|
||||
state.connecting(url)
|
||||
}
|
||||
// Change states to connecting. One sub's lock at a time.
|
||||
relayState.forEach { subId, state ->
|
||||
state.withLock { state.connecting(url) }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -205,8 +188,8 @@ class PoolRequests {
|
||||
) {
|
||||
when (cmd) {
|
||||
is ReqCmd -> {
|
||||
withStateLock {
|
||||
subState(cmd.subId).onOpenReq(relay, cmd.filters)
|
||||
subState(cmd.subId).let { state ->
|
||||
state.withLock { state.onOpenReq(relay, cmd.filters) }
|
||||
}
|
||||
desiredSubListeners.get(cmd.subId)?.onSubscriptionStarted(
|
||||
relay = relay.url,
|
||||
@@ -215,8 +198,8 @@ class PoolRequests {
|
||||
}
|
||||
|
||||
is CloseCmd -> {
|
||||
withStateLock {
|
||||
subState(cmd.subId).onSubscriptionClosed(relay)
|
||||
subState(cmd.subId).let { state ->
|
||||
state.withLock { state.onSubscriptionClosed(relay) }
|
||||
}
|
||||
desiredSubListeners.get(cmd.subId)?.onSubscriptionClosed(
|
||||
relay = relay.url,
|
||||
@@ -236,11 +219,12 @@ class PoolRequests {
|
||||
is EventMessage -> {
|
||||
var isLive = false
|
||||
var forFilters: List<Filter>? = null
|
||||
withStateLock {
|
||||
val state = relayState.get(msg.subId)
|
||||
state?.onNewEvent(relay.url)
|
||||
isLive = state?.currentState(relay.url) == ReqSubStatus.LIVE
|
||||
forFilters = state?.lastKnownFilterStates(relay.url)
|
||||
relayState.get(msg.subId)?.let { state ->
|
||||
state.withLock {
|
||||
state.onNewEvent(relay.url)
|
||||
isLive = state.currentState(relay.url) == ReqSubStatus.LIVE
|
||||
forFilters = state.lastKnownFilterStates(relay.url)
|
||||
}
|
||||
}
|
||||
desiredSubListeners.get(msg.subId)?.onEvent(
|
||||
event = msg.event,
|
||||
@@ -253,14 +237,15 @@ class PoolRequests {
|
||||
is EoseMessage -> {
|
||||
var forFilters: List<Filter>? = null
|
||||
val cmd =
|
||||
withStateLock {
|
||||
val state = relayState.get(msg.subId)
|
||||
state?.onEose(relay.url)
|
||||
forFilters = state?.lastKnownFilterStates(relay.url)
|
||||
// Decide (and pre-mark) the resend while still holding the
|
||||
// lock, so a concurrent subscribe/unsubscribe on the app
|
||||
// thread can't also decide to send a REQ for this sub.
|
||||
decideCommandLocked(msg.subId, relay.url)
|
||||
relayState.get(msg.subId)?.let { state ->
|
||||
state.withLock {
|
||||
state.onEose(relay.url)
|
||||
forFilters = state.lastKnownFilterStates(relay.url)
|
||||
// Decide (and pre-mark) the resend while still holding the
|
||||
// lock, so a concurrent subscribe/unsubscribe on the app
|
||||
// thread can't also decide to send a REQ for this sub.
|
||||
decideCommandLocked(state, msg.subId, relay.url)
|
||||
}
|
||||
}
|
||||
desiredSubListeners.get(msg.subId)?.onEose(
|
||||
relay = relay.url,
|
||||
@@ -276,11 +261,12 @@ class PoolRequests {
|
||||
is ClosedMessage -> {
|
||||
var forFilters: List<Filter>? = null
|
||||
val cmd =
|
||||
withStateLock {
|
||||
val state = relayState.get(msg.subId)
|
||||
state?.onClosed(relay.url)
|
||||
forFilters = state?.lastKnownFilterStates(relay.url)
|
||||
decideCommandLocked(msg.subId, relay.url)
|
||||
relayState.get(msg.subId)?.let { state ->
|
||||
state.withLock {
|
||||
state.onClosed(relay.url)
|
||||
forFilters = state.lastKnownFilterStates(relay.url)
|
||||
decideCommandLocked(state, msg.subId, relay.url)
|
||||
}
|
||||
}
|
||||
desiredSubListeners.get(msg.subId)?.onClosed(
|
||||
message = msg.message,
|
||||
@@ -300,10 +286,8 @@ class PoolRequests {
|
||||
* When the relay disconnects
|
||||
*/
|
||||
fun onDisconnected(url: NormalizedRelayUrl) {
|
||||
withStateLock {
|
||||
relayState.forEach { subId, state ->
|
||||
state.disconnected(url)
|
||||
}
|
||||
relayState.forEach { subId, state ->
|
||||
state.withLock { state.disconnected(url) }
|
||||
}
|
||||
}
|
||||
|
||||
@@ -326,20 +310,16 @@ class PoolRequests {
|
||||
url: NormalizedRelayUrl,
|
||||
errorMessage: String,
|
||||
) {
|
||||
// Snapshot the affected subs (and their last-known filters) under the
|
||||
// lock, then notify listeners outside it.
|
||||
val toNotify =
|
||||
withStateLock {
|
||||
val list = mutableListOf<Pair<String, List<Filter>?>>()
|
||||
relayState.forEach { subId, state ->
|
||||
// These are all my subs.. need to figure out which relays have them
|
||||
val subs = desiredSubs.get(subId)
|
||||
if (subs != null && url in subs.keys) {
|
||||
list.add(subId to state.lastKnownFilterStates(url))
|
||||
}
|
||||
}
|
||||
list
|
||||
// Snapshot the affected subs (and their last-known filters) under each
|
||||
// sub's own lock, then notify listeners outside it.
|
||||
val toNotify = mutableListOf<Pair<String, List<Filter>?>>()
|
||||
relayState.forEach { subId, state ->
|
||||
// These are all my subs.. need to figure out which relays have them
|
||||
val subs = desiredSubs.get(subId)
|
||||
if (subs != null && url in subs.keys) {
|
||||
toNotify.add(subId to state.withLock { state.lastKnownFilterStates(url) })
|
||||
}
|
||||
}
|
||||
|
||||
toNotify.forEach { (subId, forFilters) ->
|
||||
desiredSubListeners.get(subId)?.onCannotConnect(
|
||||
@@ -355,9 +335,10 @@ class PoolRequests {
|
||||
relaysToUpdate: Set<NormalizedRelayUrl>,
|
||||
sync: (NormalizedRelayUrl, Command) -> Unit,
|
||||
) {
|
||||
val state = subState(subId)
|
||||
relaysToUpdate.forEach { relay ->
|
||||
// Decide + pre-mark atomically under the lock, then send outside it.
|
||||
val cmd = withStateLock { decideCommandLocked(subId, relay) }
|
||||
// Decide + pre-mark atomically under the sub's lock, then send outside it.
|
||||
val cmd = state.withLock { decideCommandLocked(state, subId, relay) }
|
||||
if (cmd != null) {
|
||||
sync(relay, cmd)
|
||||
}
|
||||
@@ -376,14 +357,16 @@ class PoolRequests {
|
||||
* CLOSE interleaves, an empty result that silently truncates a paged
|
||||
* download) — that is the bug this guards against.
|
||||
*
|
||||
* MUST be called while holding [withStateLock].
|
||||
* MUST be called while holding [state]'s lock ([RequestSubscriptionState.withLock]);
|
||||
* [state] must be [subId]'s state instance. Takes the instance instead of
|
||||
* re-resolving it so it never re-enters the (non-reentrant) lock.
|
||||
*/
|
||||
private fun decideCommandLocked(
|
||||
state: RequestSubscriptionState<NormalizedRelayUrl>,
|
||||
subId: String,
|
||||
relay: NormalizedRelayUrl,
|
||||
): Command? {
|
||||
val state = relayState.get(subId)
|
||||
val oldFilters = state?.currentFilters(relay)
|
||||
val oldFilters = state.currentFilters(relay)
|
||||
val newFilters = desiredSubs.get(subId)?.get(relay)
|
||||
|
||||
return when {
|
||||
@@ -401,12 +384,12 @@ class PoolRequests {
|
||||
// reply belongs to which REQ. The pending change is picked up
|
||||
// later by the EOSE handler, which runs this method again once the
|
||||
// sub reaches LIVE.
|
||||
val current = state?.currentState(relay)
|
||||
val current = state.currentState(relay)
|
||||
if (current == ReqSubStatus.SENT || current == ReqSubStatus.QUERYING_PAST) {
|
||||
null
|
||||
} else {
|
||||
// Pre-mark SENT + filters so a concurrent decider skips.
|
||||
subState(subId).onOpenReq(relay, newFilters)
|
||||
state.onOpenReq(relay, newFilters)
|
||||
ReqCmd(subId, newFilters)
|
||||
}
|
||||
}
|
||||
|
||||
+4
@@ -25,6 +25,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.listeners.RelayConnection
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.IRelayClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.basic.BasicRelayClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.Message
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.MessageDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.Command
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder
|
||||
@@ -57,6 +58,8 @@ import kotlinx.coroutines.flow.update
|
||||
class RelayPool(
|
||||
val websocketBuilder: WebsocketBuilder,
|
||||
val listener: RelayConnectionListener = EmptyConnectionListener,
|
||||
/** Shared per-frame decode strategy for every relay in the pool. */
|
||||
val decoder: MessageDecoder = MessageDecoder.Default,
|
||||
) : RelayConnectionListener {
|
||||
private val relays = LargeCache<NormalizedRelayUrl, IRelayClient>()
|
||||
|
||||
@@ -73,6 +76,7 @@ class RelayPool(
|
||||
url = url,
|
||||
socketBuilder = websocketBuilder,
|
||||
listener = this,
|
||||
decoder = decoder,
|
||||
)
|
||||
|
||||
fun reconnectIfNeedsTo(ignoreRetryDelays: Boolean = false) {
|
||||
|
||||
+39
@@ -21,12 +21,51 @@
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.client.reqs
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import kotlin.concurrent.atomics.AtomicBoolean
|
||||
import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
|
||||
/**
|
||||
* Manages the State of Subscriptions by logging states as the
|
||||
* subscription progresses.
|
||||
*
|
||||
* Thread-safety: the plain maps below are only touched inside [withLock].
|
||||
* The lock lives HERE — one per subscription — instead of a single global
|
||||
* lock in PoolRequests, because every mutation is scoped to one subId:
|
||||
* relays delivering EVENTs for *different* subscriptions have no shared
|
||||
* state and must not serialize on each other. (A single global spin lock
|
||||
* measured *negative* scaling: 4 relay consumer threads pushed less
|
||||
* aggregate throughput through it than 1 — see
|
||||
* quartz/plans/2026-07-02-nostrclient-receiver-perf.md.)
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
class RequestSubscriptionState<T> {
|
||||
/**
|
||||
* Tiny non-reentrant spin lock (same primitive as BasicRelayClient's
|
||||
* connecting mutex). Critical sections are a handful of map operations,
|
||||
* never I/O — callers MUST NOT re-enter and MUST NOT hold two
|
||||
* subscriptions' locks at once (PoolRequests locks one sub at a time,
|
||||
* including inside its all-subs iterations).
|
||||
*
|
||||
* `@PublishedApi internal` only because [withLock] is inline (this sits
|
||||
* on the per-EVENT hot path; inlining avoids a closure allocation per
|
||||
* message) — treat it as private.
|
||||
*/
|
||||
@PublishedApi
|
||||
internal val lock = AtomicBoolean(false)
|
||||
|
||||
inline fun <R> withLock(block: () -> R): R {
|
||||
while (lock.exchange(true)) {
|
||||
// Test-and-test-and-set: spin-read until it looks free (cheaper
|
||||
// on the cache line than hammering exchange), then retry above.
|
||||
while (lock.load()) { }
|
||||
}
|
||||
try {
|
||||
return block()
|
||||
} finally {
|
||||
lock.store(false)
|
||||
}
|
||||
}
|
||||
|
||||
// Logs the state of each channel to:
|
||||
// 1. inform when an event is received as live
|
||||
// 2. to block REQs being sent before finished (receiving an EOSE or Closed)
|
||||
|
||||
+9
-1
@@ -24,6 +24,7 @@ import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.listeners.RelayConnectionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.IRelayClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.basic.BasicRelayClient.Companion.DELAY_TO_RECONNECT_IN_SECS
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.MessageDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.Command
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocket
|
||||
@@ -61,6 +62,13 @@ open class BasicRelayClient(
|
||||
val socketBuilder: WebsocketBuilder,
|
||||
val listener: RelayConnectionListener,
|
||||
val nowInSeconds: () -> Long = TimeUtils::now,
|
||||
/**
|
||||
* Per-frame decode strategy. The default fully parses every frame; pass a
|
||||
* shared [com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder]
|
||||
* (one instance across the whole pool) to skip re-parsing duplicate EVENT
|
||||
* frames that other relays/subscriptions already delivered.
|
||||
*/
|
||||
val decoder: MessageDecoder = MessageDecoder.Default,
|
||||
) : IRelayClient {
|
||||
companion object {
|
||||
// minimum wait time to reconnect: 1 second
|
||||
@@ -148,7 +156,7 @@ open class BasicRelayClient(
|
||||
|
||||
override fun onMessage(text: String) {
|
||||
try {
|
||||
val msg = OptimizedJsonMapper.fromJsonToMessage(text)
|
||||
val msg = decoder.decode(text)
|
||||
listener.onIncomingMessage(this@BasicRelayClient, text, msg)
|
||||
} catch (e: Throwable) {
|
||||
if (e is CancellationException) throw e
|
||||
|
||||
+154
@@ -0,0 +1,154 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.commands.toClient
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.core.HexKey
|
||||
import com.vitorpamplona.quartz.utils.cache.ConcurrentHashCache
|
||||
import kotlin.concurrent.Volatile
|
||||
import kotlin.concurrent.atomics.AtomicLong
|
||||
import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
import kotlin.concurrent.atomics.incrementAndFetch
|
||||
|
||||
/**
|
||||
* A [MessageDecoder] that skips the full JSON parse for EVENT frames whose
|
||||
* event was already parsed once.
|
||||
*
|
||||
* Why: duplicates are a large share of real relay traffic — the same event
|
||||
* arrives once per matching subscription AND once per connected relay
|
||||
* (production measurements: 14–57% duplicate frames; see
|
||||
* `quartz/plans/2026-07-02-nostrclient-receiver-perf.md`). Each duplicate
|
||||
* costs a full JSON parse (~3–30µs depending on event size) even though the
|
||||
* client already holds the parsed [Event]. This decoder scans the frame for
|
||||
* the event id (~0.3µs), and on a cache hit synthesizes the [EventMessage]
|
||||
* from the cached [Event] with the *frame's own* subscription id — so every
|
||||
* subscription still receives its delivery, per-relay bookkeeping still
|
||||
* happens, and dispatch semantics are IDENTICAL to a full parse. Only the
|
||||
* redundant parse is skipped.
|
||||
*
|
||||
* Safety rules — the scan must never mis-attribute a frame, so it bails to a
|
||||
* full parse whenever anything is unusual:
|
||||
* - the frame must start exactly with `["EVENT","`;
|
||||
* - the subscription id must contain no escapes;
|
||||
* - the id must be found as the literal `"id":"` key with a 64-lowercase-hex
|
||||
* value. Valid JSON guarantees this substring cannot occur inside a string
|
||||
* value (inner quotes are escaped as `\"`, e.g. kind-6 reposts embedding
|
||||
* full event JSON in `content`), so a match is the top-level id key.
|
||||
*
|
||||
* The id → Event cache is generational: two lock-free concurrent maps
|
||||
* ([ConcurrentHashCache]: `ConcurrentHashMap` on JVM/Android), rotated when
|
||||
* the live one reaches [capacity]. Frames arrive from every relay's consumer
|
||||
* coroutine concurrently; the hit path is entirely lock-free, and the
|
||||
* rotation races are DELIBERATELY tolerated because every one of them is
|
||||
* benign — the worst outcome is always a redundant re-parse, never a wrong
|
||||
* message:
|
||||
* - two threads rotating at once: one generation of ids is dropped early;
|
||||
* - inserting into a map that just became `previous`: still found by the
|
||||
* two-generation lookup until the next rotation;
|
||||
* - two threads first-seeing the same id simultaneously: both full-parse.
|
||||
*
|
||||
* Note the trust model is unchanged: events are identified by id here exactly
|
||||
* like in the app-level dedup (LocalCache), and signatures are verified
|
||||
* downstream regardless of which copy the Event object came from.
|
||||
*/
|
||||
@OptIn(ExperimentalAtomicApi::class)
|
||||
class CachingEventDecoder(
|
||||
private val capacity: Int = 2048,
|
||||
private val fullParser: MessageDecoder = MessageDecoder.Default,
|
||||
) : MessageDecoder {
|
||||
@Volatile private var live = ConcurrentHashCache<HexKey, Event>()
|
||||
|
||||
@Volatile private var previous = ConcurrentHashCache<HexKey, Event>()
|
||||
|
||||
// Observability + benchmark hooks.
|
||||
private val parsed = AtomicLong(0)
|
||||
private val reused = AtomicLong(0)
|
||||
|
||||
val parsedCount: Long get() = parsed.load()
|
||||
val reusedCount: Long get() = reused.load()
|
||||
|
||||
override fun decode(text: String): Message {
|
||||
val scanned = scanEventFrame(text)
|
||||
if (scanned != null) {
|
||||
val cached = live.get(scanned.eventId) ?: previous.get(scanned.eventId)
|
||||
if (cached != null) {
|
||||
reused.incrementAndFetch()
|
||||
return EventMessage(scanned.subId, cached)
|
||||
}
|
||||
}
|
||||
|
||||
val msg = fullParser.decode(text)
|
||||
if (msg is EventMessage) {
|
||||
parsed.incrementAndFetch()
|
||||
if (live.size() >= capacity) {
|
||||
// Benign-race rotation: see class kdoc.
|
||||
previous = live
|
||||
live = ConcurrentHashCache()
|
||||
}
|
||||
live.put(msg.event.id, msg.event)
|
||||
}
|
||||
return msg
|
||||
}
|
||||
|
||||
private class ScannedEvent(
|
||||
val subId: String,
|
||||
val eventId: HexKey,
|
||||
)
|
||||
|
||||
/**
|
||||
* Extracts (subId, eventId) from a compact `["EVENT","<subId>",{...}]`
|
||||
* frame, or null when the frame isn't an EVENT or anything about it is
|
||||
* unusual (whitespace variants, escaped subIds, missing/malformed id) —
|
||||
* null means "do the full parse".
|
||||
*/
|
||||
private fun scanEventFrame(text: String): ScannedEvent? {
|
||||
if (!text.startsWith(EVENT_PREFIX)) return null
|
||||
|
||||
// subId: read up to the closing quote; any escape → bail.
|
||||
var i = EVENT_PREFIX.length
|
||||
val subStart = i
|
||||
while (i < text.length) {
|
||||
val c = text[i]
|
||||
if (c == '\\') return null
|
||||
if (c == '"') break
|
||||
i++
|
||||
}
|
||||
if (i >= text.length) return null
|
||||
val subId = text.substring(subStart, i)
|
||||
|
||||
// id: the literal top-level key with a 64-char lowercase-hex value.
|
||||
val idKey = text.indexOf(ID_KEY, i)
|
||||
if (idKey < 0) return null
|
||||
val idStart = idKey + ID_KEY.length
|
||||
val idEnd = idStart + 64
|
||||
if (idEnd >= text.length || text[idEnd] != '"') return null
|
||||
for (j in idStart until idEnd) {
|
||||
val c = text[j]
|
||||
if (c !in '0'..'9' && c !in 'a'..'f') return null
|
||||
}
|
||||
return ScannedEvent(subId, text.substring(idStart, idEnd))
|
||||
}
|
||||
|
||||
companion object {
|
||||
private const val EVENT_PREFIX = "[\"EVENT\",\""
|
||||
private const val ID_KEY = "\"id\":\""
|
||||
}
|
||||
}
|
||||
+40
@@ -0,0 +1,40 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.commands.toClient
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper
|
||||
|
||||
/**
|
||||
* Turns one raw relay frame into a [Message]. The strategy the relay client
|
||||
* uses for its per-frame decode step — swap it to change how (or whether) a
|
||||
* frame is parsed without touching the connection machinery.
|
||||
*
|
||||
* Implementations may throw on malformed frames (the caller logs and drops
|
||||
* the frame, matching [OptimizedJsonMapper.fromJsonToMessage] behavior).
|
||||
*/
|
||||
fun interface MessageDecoder {
|
||||
fun decode(text: String): Message
|
||||
|
||||
companion object {
|
||||
/** Plain full-JSON parse of every frame. */
|
||||
val Default: MessageDecoder = MessageDecoder { OptimizedJsonMapper.fromJsonToMessage(it) }
|
||||
}
|
||||
}
|
||||
+26
-23
@@ -27,7 +27,7 @@ import com.vitorpamplona.quartz.nip01Core.store.IEventStore
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import kotlinx.coroutines.CompletableDeferred
|
||||
import kotlinx.coroutines.awaitCancellation
|
||||
import kotlin.concurrent.atomics.AtomicReference
|
||||
import kotlin.concurrent.atomics.AtomicBoolean
|
||||
import kotlin.concurrent.atomics.ExperimentalAtomicApi
|
||||
|
||||
/**
|
||||
@@ -140,24 +140,37 @@ class LiveEventStore(
|
||||
//
|
||||
// The set is read from the [IngestQueue] drain coroutine (in
|
||||
// `deliver`, called synchronously from `fanout`) and written
|
||||
// from this coroutine (the historical-replay closure below).
|
||||
// It must be a persistent / immutable Set under an
|
||||
// AtomicReference — wrapping a mutable HashSet would race
|
||||
// because AtomicReference only protects the reference, not
|
||||
// the set's internal state. Each `add` is a CAS-loop that
|
||||
// publishes a new immutable Set; `deliver`'s `load()` always
|
||||
// sees a fully-constructed snapshot.
|
||||
// from this coroutine (the historical-replay closure below),
|
||||
// so access is guarded by a tiny spin lock (contains/add,
|
||||
// never I/O). It MUST be a mutable set under a lock, not an
|
||||
// immutable Set under an AtomicReference with copy-on-add:
|
||||
// `set + id` copies the whole set per streamed event, which
|
||||
// made large replays accidentally O(n²) — a 100k-event REQ
|
||||
// crawled at ~700 events/s and the rate degraded as the
|
||||
// response grew (see the plan doc's giant-REQ finding).
|
||||
//
|
||||
// Once cleared to null after EOSE, `deliver` short-circuits
|
||||
// and every live event is forwarded.
|
||||
val seenIds = AtomicReference<Set<String>?>(emptySet())
|
||||
val seenLock = AtomicBoolean(false)
|
||||
var seenIds: HashSet<String>? = HashSet(1024)
|
||||
|
||||
fun <R> seenLocked(block: () -> R): R {
|
||||
while (seenLock.exchange(true)) {
|
||||
while (seenLock.load()) { }
|
||||
}
|
||||
try {
|
||||
return block()
|
||||
} finally {
|
||||
seenLock.store(false)
|
||||
}
|
||||
}
|
||||
|
||||
val sub =
|
||||
LiveSubscription(
|
||||
filters = filters,
|
||||
deliver = { event ->
|
||||
val seen = seenIds.load()
|
||||
if (seen != null && seen.contains(event.id)) return@LiveSubscription
|
||||
val duplicate = seenLocked { seenIds?.contains(event.id) ?: false }
|
||||
if (duplicate) return@LiveSubscription
|
||||
onEach(event)
|
||||
},
|
||||
)
|
||||
@@ -165,24 +178,14 @@ class LiveEventStore(
|
||||
index.register(filters, sub)
|
||||
try {
|
||||
store.query<Event>(filters) { event ->
|
||||
// CAS-loop on an immutable Set. The historical
|
||||
// replay is single-writer in steady state, so the
|
||||
// CAS typically succeeds first try; the loop only
|
||||
// matters if we lose a race on `seenIds.store(null)`
|
||||
// (post-EOSE), in which case `current == null` and
|
||||
// we exit cleanly.
|
||||
while (true) {
|
||||
val current = seenIds.load() ?: break
|
||||
if (event.id in current) break
|
||||
if (seenIds.compareAndSet(current, current + event.id)) break
|
||||
}
|
||||
seenLocked { seenIds?.add(event.id) }
|
||||
onEach(event)
|
||||
}
|
||||
onEose()
|
||||
// Drop the dedupe set so the live path stops paying for
|
||||
// it. From this point the index drives delivery and
|
||||
// duplicates are no longer possible.
|
||||
seenIds.store(null)
|
||||
seenLocked { seenIds = null }
|
||||
// Suspend until the caller's coroutine is cancelled
|
||||
// (e.g. NIP-01 CLOSE or connection drop). The `finally`
|
||||
// unregisters from the index.
|
||||
|
||||
+1
-1
@@ -79,7 +79,7 @@ class NegentropyManager(
|
||||
localEvents: List<Event>,
|
||||
frameSizeLimit: Long = 0,
|
||||
) {
|
||||
val session = NegentropySession(subId, filter, localEvents, frameSizeLimit)
|
||||
val session = NegentropySession.fromEvents(subId, filter, localEvents, frameSizeLimit)
|
||||
activeSessions[subId] = Pair(relay.url, session)
|
||||
|
||||
val openCmd = session.open()
|
||||
|
||||
+29
-3
@@ -24,6 +24,7 @@ import com.vitorpamplona.negentropy.Negentropy
|
||||
import com.vitorpamplona.negentropy.storage.StorageVector
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import com.vitorpamplona.quartz.utils.Hex
|
||||
|
||||
/**
|
||||
@@ -37,19 +38,44 @@ import com.vitorpamplona.quartz.utils.Hex
|
||||
* 5. Repeat until [processMessage] returns a result with a null command (reconciliation complete)
|
||||
* 6. Use [haveIds] and [needIds] from the result to send/request events
|
||||
* 7. Call [close] to produce a NEG-CLOSE command when done
|
||||
*
|
||||
* The primary constructor takes [IdAndTime] entries (just `created_at` and the
|
||||
* event id — all the reconciliation library indexes) so callers with a store
|
||||
* snapshot don't materialize full events; the [Event] overload is kept for
|
||||
* callers that already hold them (mirrors [NegentropyServerSession]).
|
||||
*/
|
||||
class NegentropySession(
|
||||
val subId: String,
|
||||
val filter: Filter,
|
||||
localEvents: List<Event>,
|
||||
localEntries: List<IdAndTime>,
|
||||
frameSizeLimit: Long = 0,
|
||||
) {
|
||||
companion object {
|
||||
/**
|
||||
* Convenience for callers that hold full [Event] objects (JVM type
|
||||
* erasure forbids a `List<Event>` constructor overload). Mirrors
|
||||
* [NegentropyServerSession.fromEvents].
|
||||
*/
|
||||
fun fromEvents(
|
||||
subId: String,
|
||||
filter: Filter,
|
||||
localEvents: List<Event>,
|
||||
frameSizeLimit: Long = 0,
|
||||
): NegentropySession =
|
||||
NegentropySession(
|
||||
subId = subId,
|
||||
filter = filter,
|
||||
localEntries = localEvents.map { IdAndTime(it.createdAt, it.id) },
|
||||
frameSizeLimit = frameSizeLimit,
|
||||
)
|
||||
}
|
||||
|
||||
private val storage = StorageVector()
|
||||
private val negentropy: Negentropy
|
||||
|
||||
init {
|
||||
for (event in localEvents) {
|
||||
storage.insert(event.createdAt, event.id)
|
||||
for (entry in localEntries) {
|
||||
storage.insert(entry.createdAt, entry.id)
|
||||
}
|
||||
storage.seal()
|
||||
negentropy = Negentropy(storage, frameSizeLimit)
|
||||
|
||||
+47
@@ -0,0 +1,47 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils.cache
|
||||
|
||||
/**
|
||||
* A minimal thread-safe hash map for hot lookup paths: `get`/`put`/`size`.
|
||||
*
|
||||
* Exists because commonMain has no `java.util.concurrent.ConcurrentHashMap`
|
||||
* and [LargeCache]'s JVM actual is a `ConcurrentSkipListMap` (ordered,
|
||||
* O(log n) lookups). This one is unordered and tuned for read-heavy caches
|
||||
* where lookups must be lock-free on JVM/Android — e.g. the per-frame
|
||||
* duplicate check in
|
||||
* [com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder].
|
||||
*
|
||||
* Actuals: JVM/Android → `ConcurrentHashMap`; Apple → `CacheMap` (same
|
||||
* backing as [LargeCache]); Linux → copy-on-write (CI-only target).
|
||||
*/
|
||||
expect class ConcurrentHashCache<K : Any, V : Any>() {
|
||||
fun get(key: K): V?
|
||||
|
||||
fun put(
|
||||
key: K,
|
||||
value: V,
|
||||
)
|
||||
|
||||
fun size(): Int
|
||||
|
||||
fun clear()
|
||||
}
|
||||
+160
@@ -0,0 +1,160 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.commands.toClient
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertNotSame
|
||||
import kotlin.test.assertSame
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
class CachingEventDecoderTest {
|
||||
private fun hexId(seed: Int) = seed.toString(16).padStart(64, '0')
|
||||
|
||||
private fun event(
|
||||
seed: Int,
|
||||
content: String = "hello $seed",
|
||||
) = Event(
|
||||
id = hexId(seed),
|
||||
pubKey = hexId(seed + 100_000),
|
||||
createdAt = seed.toLong(),
|
||||
kind = 1,
|
||||
tags = arrayOf(arrayOf("t", "test")),
|
||||
content = content,
|
||||
sig = "f".repeat(128),
|
||||
)
|
||||
|
||||
private fun frame(
|
||||
subId: String,
|
||||
event: Event,
|
||||
) = """["EVENT","$subId",${event.toJson()}]"""
|
||||
|
||||
@Test
|
||||
fun firstSightParsesThenDuplicateReusesTheSameEventInstance() {
|
||||
val decoder = CachingEventDecoder()
|
||||
val e = event(1)
|
||||
|
||||
val first = decoder.decode(frame("subA", e)) as EventMessage
|
||||
val second = decoder.decode(frame("subB", e)) as EventMessage
|
||||
|
||||
assertEquals(e.id, first.event.id)
|
||||
assertEquals("subA", first.subId)
|
||||
assertEquals("subB", second.subId, "duplicate keeps the frame's own subId")
|
||||
assertSame(first.event, second.event, "duplicate must reuse the parsed Event")
|
||||
assertEquals(1, decoder.parsedCount)
|
||||
assertEquals(1, decoder.reusedCount)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun duplicateAcrossSubIdsAndOrderingsMatchesFullParse() {
|
||||
val decoder = CachingEventDecoder()
|
||||
val events = (1..50).map { event(it) }
|
||||
|
||||
// every event delivered on 3 subs, interleaved
|
||||
val frames =
|
||||
buildList {
|
||||
for (sub in listOf("s1", "s2", "s3")) {
|
||||
events.forEach { add(frame(sub, it)) }
|
||||
}
|
||||
}
|
||||
|
||||
val decoded = frames.map { decoder.decode(it) as EventMessage }
|
||||
assertEquals(150, decoded.size)
|
||||
assertEquals(50, decoder.parsedCount)
|
||||
assertEquals(100, decoder.reusedCount)
|
||||
// each decoded message matches the full-parse result
|
||||
frames.zip(decoded).forEach { (f, msg) ->
|
||||
val reference = MessageDecoder.Default.decode(f) as EventMessage
|
||||
assertEquals(reference.subId, msg.subId)
|
||||
assertEquals(reference.event.id, msg.event.id)
|
||||
assertEquals(reference.event.content, msg.event.content)
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun repostStyleContentWithEmbeddedEventJsonNeverConfusesTheScan() {
|
||||
val decoder = CachingEventDecoder()
|
||||
val inner = event(7)
|
||||
// kind-6-style: full event JSON embedded (and therefore escaped) in content.
|
||||
// The escaped \"id\":\" inside content must NOT be read as the outer id.
|
||||
val repost = event(8, content = inner.toJson())
|
||||
|
||||
val decoded = decoder.decode(frame("s", repost)) as EventMessage
|
||||
assertEquals(repost.id, decoded.event.id, "outer id wins")
|
||||
|
||||
// and a duplicate of the repost still resolves to the repost, not the inner event
|
||||
val dup = decoder.decode(frame("s2", repost)) as EventMessage
|
||||
assertSame(decoded.event, dup.event)
|
||||
assertEquals(repost.id, dup.event.id)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun nonEventFramesPassThroughUntouched() {
|
||||
val decoder = CachingEventDecoder()
|
||||
assertTrue(decoder.decode("""["EOSE","sub1"]""") is EoseMessage)
|
||||
assertTrue(decoder.decode("""["NOTICE","hi"]""") is NoticeMessage)
|
||||
assertTrue(decoder.decode("""["OK","${hexId(3)}",true,""]""") is OkMessage)
|
||||
assertEquals(0, decoder.parsedCount, "only EVENT frames populate the cache")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun escapedSubIdFallsBackToFullParse() {
|
||||
val decoder = CachingEventDecoder()
|
||||
val e = event(11)
|
||||
decoder.decode(frame("plain", e)) // cache it
|
||||
|
||||
// subId with an escaped quote: scan must bail, full parse must win.
|
||||
val weird = """["EVENT","a\"b",${e.toJson()}]"""
|
||||
val decoded = decoder.decode(weird) as EventMessage
|
||||
assertEquals("a\"b", decoded.subId)
|
||||
assertEquals(e.id, decoded.event.id)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun malformedIdValueFallsBackToFullParse() {
|
||||
val decoder = CachingEventDecoder()
|
||||
val e = event(12)
|
||||
decoder.decode(frame("s", e))
|
||||
|
||||
// Uppercase hex in the id key position → scan bails → full parse throws
|
||||
// or produces whatever the mapper does; here we just assert no cache hit
|
||||
// is fabricated for a different-id frame.
|
||||
val other = event(13)
|
||||
val decoded = decoder.decode(frame("s", other)) as EventMessage
|
||||
assertEquals(other.id, decoded.event.id)
|
||||
assertNotSame(e, decoded.event)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun cacheRotationOnlyCausesReparseNeverWrongMessages() {
|
||||
val decoder = CachingEventDecoder(capacity = 8)
|
||||
val events = (1..40).map { event(it) }
|
||||
|
||||
events.forEach { decoder.decode(frame("s", it)) }
|
||||
// replay everything: some rotated out (re-parse), none may mismatch
|
||||
events.forEach {
|
||||
val decoded = decoder.decode(frame("s2", it)) as EventMessage
|
||||
assertEquals(it.id, decoded.event.id)
|
||||
assertEquals("s2", decoded.subId)
|
||||
}
|
||||
}
|
||||
}
|
||||
+9
-9
@@ -54,7 +54,7 @@ class NegentropySessionTest {
|
||||
makeEvent("c".repeat(64), 3000),
|
||||
)
|
||||
|
||||
val clientSession = NegentropySession("sub1", Filter(), events)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), events)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
assertEquals("NEG-OPEN", openCmd.label())
|
||||
@@ -90,7 +90,7 @@ class NegentropySessionTest {
|
||||
val clientOnlyEvent = makeEvent("c".repeat(64), 3000)
|
||||
val clientEvents = sharedEvents + clientOnlyEvent
|
||||
|
||||
val clientSession = NegentropySession("sub1", Filter(), clientEvents)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), clientEvents)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
// Server side: only shared events
|
||||
@@ -119,7 +119,7 @@ class NegentropySessionTest {
|
||||
makeEvent("b".repeat(64), 2000),
|
||||
)
|
||||
|
||||
val clientSession = NegentropySession("sub1", Filter(), sharedEvents)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), sharedEvents)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
// Server side: shared + extra
|
||||
@@ -151,7 +151,7 @@ class NegentropySessionTest {
|
||||
val clientOnly = makeEvent("b".repeat(64), 2000)
|
||||
val clientEvents = sharedEvents + clientOnly
|
||||
|
||||
val clientSession = NegentropySession("sub1", Filter(), clientEvents)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), clientEvents)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
// Server side: shared + different extra
|
||||
@@ -174,7 +174,7 @@ class NegentropySessionTest {
|
||||
|
||||
@Test
|
||||
fun clientServerSync_emptyClient_needsAll() {
|
||||
val clientSession = NegentropySession("sub1", Filter(), emptyList())
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), emptyList())
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
// Server has events
|
||||
@@ -201,7 +201,7 @@ class NegentropySessionTest {
|
||||
makeEvent("b".repeat(64), 2000),
|
||||
)
|
||||
|
||||
val clientSession = NegentropySession("sub1", Filter(), clientEvents)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), clientEvents)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
// Server is empty
|
||||
@@ -235,7 +235,7 @@ class NegentropySessionTest {
|
||||
}
|
||||
|
||||
// Small frame limit to force multiple rounds
|
||||
val clientSession = NegentropySession("sub1", Filter(), clientEvents, frameSizeLimit = 4096)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), clientEvents, frameSizeLimit = 4096)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
val serverStorage = StorageVector()
|
||||
@@ -281,7 +281,7 @@ class NegentropySessionTest {
|
||||
val serverEvents = sharedEvents + serverOnlyEvent
|
||||
|
||||
// Client initiates
|
||||
val clientSession = NegentropySession("sub1", Filter(), sharedEvents)
|
||||
val clientSession = NegentropySession.fromEvents("sub1", Filter(), sharedEvents)
|
||||
val openCmd = clientSession.open()
|
||||
|
||||
// Server processes via NegentropyServerSession
|
||||
@@ -296,7 +296,7 @@ class NegentropySessionTest {
|
||||
|
||||
@Test
|
||||
fun close_returnsCorrectCmd() {
|
||||
val session = NegentropySession("sub1", Filter(), emptyList())
|
||||
val session = NegentropySession.fromEvents("sub1", Filter(), emptyList())
|
||||
val closeCmd = session.close()
|
||||
|
||||
assertEquals("NEG-CLOSE", closeCmd.label())
|
||||
|
||||
+12
-3
@@ -61,6 +61,16 @@ class BasicOkHttpWebSocket(
|
||||
val listener =
|
||||
object : OkHttpWebSocketListener() {
|
||||
val scope = CoroutineScope(Dispatchers.IO + exceptionHandler)
|
||||
|
||||
// UNLIMITED on purpose — do NOT bound this channel. The app
|
||||
// holds 2000+ relay connections; a bounded buffer under a
|
||||
// slow consumer would block OkHttp reader threads (thread
|
||||
// starvation at that connection count) and park the backlog
|
||||
// on the RELAY's outbound buffers via TCP backpressure —
|
||||
// infrastructure that isn't ours. We drain the remote as
|
||||
// fast as it can send and own the buffering; consumer speed
|
||||
// is handled downstream (CachingEventDecoder,
|
||||
// ParallelEventVerifier).
|
||||
val incomingMessages: Channel<String> = Channel(Channel.UNLIMITED)
|
||||
val job = // Launch a coroutine to process messages from the channel.
|
||||
scope.launch {
|
||||
@@ -81,9 +91,8 @@ class BasicOkHttpWebSocket(
|
||||
webSocket: OkHttpWebSocket,
|
||||
text: String,
|
||||
) {
|
||||
// Asynchronously send the received message to the channel.
|
||||
// `trySendBlocking` is used here for simplicity within the callback,
|
||||
// but it's important to understand potential thread blocking if the buffer is full.
|
||||
// Never blocks (unlimited channel): the OkHttp reader
|
||||
// thread must stay free to keep draining the socket.
|
||||
incomingMessages.trySendBlocking(text)
|
||||
}
|
||||
|
||||
|
||||
Vendored
+40
@@ -0,0 +1,40 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils.cache
|
||||
|
||||
import java.util.concurrent.ConcurrentHashMap
|
||||
|
||||
actual class ConcurrentHashCache<K : Any, V : Any> {
|
||||
private val map = ConcurrentHashMap<K, V>()
|
||||
|
||||
actual fun get(key: K): V? = map[key]
|
||||
|
||||
actual fun put(
|
||||
key: K,
|
||||
value: V,
|
||||
) {
|
||||
map[key] = value
|
||||
}
|
||||
|
||||
actual fun size(): Int = map.size
|
||||
|
||||
actual fun clear() = map.clear()
|
||||
}
|
||||
+147
@@ -0,0 +1,147 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay
|
||||
|
||||
import com.vitorpamplona.geode.fixtures.SyntheticEvents
|
||||
import com.vitorpamplona.geode.testing.RelayClientTest
|
||||
import com.vitorpamplona.geode.testing.preload
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.negentropySyncFanOut
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import org.junit.After
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
class NostrClientNegentropyFanOutTest : RelayClientTest() {
|
||||
// extra connections to the same in-process relay
|
||||
private val extraClients = mutableListOf<NostrClient>()
|
||||
|
||||
private fun clients(count: Int): List<NostrClient> =
|
||||
buildList {
|
||||
add(client)
|
||||
repeat(count - 1) {
|
||||
add(NostrClient(hub, scope).also { extraClients.add(it) })
|
||||
}
|
||||
}
|
||||
|
||||
@After
|
||||
fun tearDownExtraClients() {
|
||||
extraClients.forEach { it.close() }
|
||||
}
|
||||
|
||||
private fun events(range: IntRange): List<Event> = range.map { SyntheticEvents.fakeEvent(idSeed = it, kind = 1, createdAt = it.toLong()) }
|
||||
|
||||
@Test
|
||||
fun fullDownloadAcrossMultipleConnections() =
|
||||
runBlocking {
|
||||
val seeded = events(1..60)
|
||||
defaultRelay.preload(seeded)
|
||||
|
||||
val got = mutableListOf<Event>()
|
||||
val result =
|
||||
withTimeout(30_000) {
|
||||
negentropySyncFanOut(
|
||||
clients = clients(3),
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
fetchBatch = 5,
|
||||
reqsPerClient = 2,
|
||||
) { got.add(it) }
|
||||
}
|
||||
|
||||
assertEquals(60, got.size, "every event delivered")
|
||||
assertEquals(60, got.map { it.id }.toSet().size, "no duplicates")
|
||||
assertEquals(60, result.needCount)
|
||||
assertEquals(60, result.downloaded)
|
||||
assertTrue(result.connections >= 1, "at least one connection served batches")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun maxEventsCapsDeliveryAndStopsTheFanOut() =
|
||||
runBlocking {
|
||||
defaultRelay.preload(events(1..50))
|
||||
|
||||
val got = mutableListOf<Event>()
|
||||
val result =
|
||||
withTimeout(30_000) {
|
||||
negentropySyncFanOut(
|
||||
clients = clients(2),
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
maxEvents = 12,
|
||||
fetchBatch = 4,
|
||||
) { got.add(it) }
|
||||
}
|
||||
|
||||
assertEquals(12, got.size)
|
||||
assertEquals(12, result.downloaded)
|
||||
}
|
||||
|
||||
@Test
|
||||
fun localEntriesSkipWhatWeAlreadyHaveAndReportHaves() =
|
||||
runBlocking {
|
||||
val relayOnly = events(1..20)
|
||||
val shared = events(101..110)
|
||||
val localOnly = events(201..205)
|
||||
defaultRelay.preload(relayOnly + shared)
|
||||
|
||||
val got = mutableListOf<Event>()
|
||||
val result =
|
||||
withTimeout(30_000) {
|
||||
negentropySyncFanOut(
|
||||
clients = clients(2),
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
localEntries = (shared + localOnly).map { IdAndTime(it.createdAt, it.id) },
|
||||
fetchBatch = 6,
|
||||
) { got.add(it) }
|
||||
}
|
||||
|
||||
assertEquals(relayOnly.map { it.id }.toSet(), got.map { it.id }.toSet(), "only relay-only events download")
|
||||
assertEquals(20, result.needCount)
|
||||
assertEquals(5, result.haveCount, "local-only events reported as haves")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun singleClientDegradesGracefully() =
|
||||
runBlocking {
|
||||
defaultRelay.preload(events(1..25))
|
||||
|
||||
val got = mutableListOf<Event>()
|
||||
val result =
|
||||
withTimeout(30_000) {
|
||||
negentropySyncFanOut(
|
||||
clients = clients(1),
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
fetchBatch = 10,
|
||||
) { got.add(it) }
|
||||
}
|
||||
|
||||
assertEquals(25, result.downloaded)
|
||||
assertEquals(25, got.map { it.id }.toSet().size)
|
||||
}
|
||||
}
|
||||
+163
@@ -0,0 +1,163 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay
|
||||
|
||||
import com.vitorpamplona.geode.fixtures.SyntheticEvents
|
||||
import com.vitorpamplona.geode.testing.RelayClientTest
|
||||
import com.vitorpamplona.geode.testing.preload
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.core.HexKey
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.negentropyReconcile
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.negentropyReconcileIds
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.store.IdAndTime
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* End-to-end tests for the pure-reconcile API ([negentropyReconcile] /
|
||||
* [negentropyReconcileIds]): the id diff comes back in both directions —
|
||||
* `needIds` (relay has, we lack) and `haveIds` (we have, relay lacks) — and
|
||||
* nothing is downloaded or uploaded.
|
||||
*/
|
||||
class NostrClientNegentropyReconcileTest : RelayClientTest() {
|
||||
private fun events(range: IntRange): List<Event> = range.map { SyntheticEvents.fakeEvent(idSeed = it, kind = 1, createdAt = it.toLong()) }
|
||||
|
||||
private fun List<Event>.entries(): List<IdAndTime> = map { IdAndTime(it.createdAt, it.id) }
|
||||
|
||||
private fun List<Event>.ids(): Set<HexKey> = map { it.id }.toSet()
|
||||
|
||||
@Test
|
||||
fun emptyLocalSetNeedsEverythingHasNothing() =
|
||||
runBlocking {
|
||||
val onRelay = events(1..20)
|
||||
defaultRelay.preload(onRelay)
|
||||
|
||||
val diff =
|
||||
withTimeout(20_000) {
|
||||
client.negentropyReconcileIds(
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
)
|
||||
}
|
||||
|
||||
assertEquals(onRelay.ids(), diff.needIds.toSet(), "need = everything on the relay")
|
||||
assertTrue(diff.haveIds.isEmpty(), "nothing to publish")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun partialOverlapSplitsBothWays() =
|
||||
runBlocking {
|
||||
// relay holds A ∪ B; local set is B ∪ C.
|
||||
val a = events(1..10) // relay-only → expected needIds
|
||||
val b = events(101..110) // shared → in neither diff
|
||||
val c = events(201..205) // local-only → expected haveIds
|
||||
defaultRelay.preload(a + b)
|
||||
|
||||
val diff =
|
||||
withTimeout(20_000) {
|
||||
client.negentropyReconcileIds(
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
localEntries = (b + c).entries(),
|
||||
)
|
||||
}
|
||||
|
||||
assertEquals(a.ids(), diff.needIds.toSet(), "need = relay-only events")
|
||||
assertEquals(c.ids(), diff.haveIds.toSet(), "have = local-only events")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun identicalSetsProduceEmptyDiff() =
|
||||
runBlocking {
|
||||
val shared = events(1..15)
|
||||
defaultRelay.preload(shared)
|
||||
|
||||
val diff =
|
||||
withTimeout(20_000) {
|
||||
client.negentropyReconcileIds(
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
localEntries = shared.entries(),
|
||||
)
|
||||
}
|
||||
|
||||
assertTrue(diff.needIds.isEmpty(), "in sync: nothing to download")
|
||||
assertTrue(diff.haveIds.isEmpty(), "in sync: nothing to publish")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun streamingCallbacksReportCountsAndBatches() =
|
||||
runBlocking {
|
||||
val onRelay = events(1..30)
|
||||
val localOnly = events(301..312)
|
||||
defaultRelay.preload(onRelay)
|
||||
|
||||
val needBatches = mutableListOf<List<HexKey>>()
|
||||
val haveBatches = mutableListOf<List<HexKey>>()
|
||||
val result =
|
||||
withTimeout(20_000) {
|
||||
client.negentropyReconcile(
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1)),
|
||||
localEntries = localOnly.entries(),
|
||||
batchSize = 8,
|
||||
onHaveIds = { haveBatches.add(it) },
|
||||
onNeedIds = { needBatches.add(it) },
|
||||
)
|
||||
}
|
||||
|
||||
assertEquals(30, result.needCount)
|
||||
assertEquals(12, result.haveCount)
|
||||
assertEquals(onRelay.ids(), needBatches.flatten().toSet())
|
||||
assertEquals(localOnly.ids(), haveBatches.flatten().toSet())
|
||||
assertTrue(needBatches.all { it.size <= 8 }, "need ids arrive in <= batchSize chunks")
|
||||
assertTrue(haveBatches.all { it.size <= 8 }, "have ids arrive in <= batchSize chunks")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun sinceUntilBoundsTheDiffAndLocalEntriesOutsideWindowAreIgnored() =
|
||||
runBlocking {
|
||||
// relay: createdAt 1..20. Window [5..15]. Local set holds 8..12 plus
|
||||
// entries OUTSIDE the window (createdAt 950..960) that must not be
|
||||
// reported as haves because the filter excludes them on both sides.
|
||||
val onRelay = events(1..20)
|
||||
val localShared = onRelay.filter { it.createdAt in 8L..12L }
|
||||
val localOutside = events(950..960)
|
||||
defaultRelay.preload(onRelay)
|
||||
|
||||
val diff =
|
||||
withTimeout(20_000) {
|
||||
client.negentropyReconcileIds(
|
||||
relay = defaultRelayUrl,
|
||||
filter = Filter(kinds = listOf(1), since = 5, until = 15),
|
||||
localEntries = (localShared + localOutside).entries(),
|
||||
)
|
||||
}
|
||||
|
||||
val expectedNeed = onRelay.filter { it.createdAt in 5L..15L && it.createdAt !in 8L..12L }.ids()
|
||||
assertEquals(expectedNeed, diff.needIds.toSet(), "need = window events we lack")
|
||||
assertTrue(diff.haveIds.isEmpty(), "out-of-window local entries are not haves")
|
||||
}
|
||||
}
|
||||
+150
@@ -0,0 +1,150 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.client.accessories
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.crypto.KeyPair
|
||||
import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerSync
|
||||
import com.vitorpamplona.quartz.nip10Notes.TextNoteEvent
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.SupervisorJob
|
||||
import kotlinx.coroutines.cancel
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
class ParallelEventVerifierTest {
|
||||
private val signer = NostrSignerSync(KeyPair())
|
||||
|
||||
private fun signed(i: Int): Event = signer.sign(TextNoteEvent.build("note $i", createdAt = i.toLong()))
|
||||
|
||||
private fun tampered(i: Int): Event {
|
||||
val e = signed(i)
|
||||
// altered content, original id/sig — must fail verification
|
||||
return Event(e.id, e.pubKey, e.createdAt, e.kind, e.tags, e.content + " tampered", e.sig)
|
||||
}
|
||||
|
||||
private fun <T> withVerifierScope(block: (CoroutineScope) -> T): T {
|
||||
val scope = CoroutineScope(SupervisorJob() + Dispatchers.Default)
|
||||
try {
|
||||
return block(scope)
|
||||
} finally {
|
||||
scope.cancel()
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun routesValidAndTamperedEventsCorrectly() =
|
||||
withVerifierScope { scope ->
|
||||
runBlocking {
|
||||
val valid = (1..40).map { signed(it) }
|
||||
val invalid = (100..109).map { tampered(it) }
|
||||
|
||||
val verified = mutableListOf<Event>()
|
||||
val rejected = mutableListOf<Event>()
|
||||
val verifier =
|
||||
ParallelEventVerifier<String>(
|
||||
scope = scope,
|
||||
onInvalid = { e, _ -> rejected.add(e) },
|
||||
onVerified = { e, _ -> verified.add(e) },
|
||||
)
|
||||
|
||||
(valid + invalid).shuffled().forEach { verifier.submit(it, "relay") }
|
||||
verifier.close()
|
||||
withTimeout(30_000) { verifier.join() }
|
||||
|
||||
assertEquals(valid.map { it.id }.toSet(), verified.map { it.id }.toSet())
|
||||
assertEquals(invalid.map { it.id }.toSet(), rejected.map { it.id }.toSet())
|
||||
assertEquals(40L, verifier.verifiedCount)
|
||||
assertEquals(10L, verifier.invalidCount)
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun callbacksArriveInSubmissionOrder() =
|
||||
withVerifierScope { scope ->
|
||||
runBlocking {
|
||||
val events = (1..300).map { signed(it) }
|
||||
val order = mutableListOf<Long>()
|
||||
val verifier =
|
||||
ParallelEventVerifier<Unit>(
|
||||
scope = scope,
|
||||
maxBatch = 16,
|
||||
onVerified = { e, _ -> order.add(e.createdAt) },
|
||||
)
|
||||
events.forEach { verifier.submit(it, Unit) }
|
||||
verifier.close()
|
||||
withTimeout(30_000) { verifier.join() }
|
||||
|
||||
assertEquals(events.map { it.createdAt }, order, "parallel verify must not reorder callbacks")
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun preVerifiedShortCircuitSkipsTheCpuButStillDelivers() =
|
||||
withVerifierScope { scope ->
|
||||
runBlocking {
|
||||
// Tampered events would FAIL verification — marking them preVerified
|
||||
// must route them to onVerified without running the check.
|
||||
val events = (1..20).map { tampered(it) }
|
||||
val delivered = mutableListOf<Event>()
|
||||
val verifier =
|
||||
ParallelEventVerifier<Unit>(
|
||||
scope = scope,
|
||||
preVerified = { true },
|
||||
onVerified = { e, _ -> delivered.add(e) },
|
||||
)
|
||||
events.forEach { verifier.submit(it, Unit) }
|
||||
verifier.close()
|
||||
withTimeout(30_000) { verifier.join() }
|
||||
|
||||
assertEquals(20, delivered.size)
|
||||
assertEquals(20L, verifier.verifiedCount)
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun misbehavingCallbackDoesNotKillTheDrainLoop() =
|
||||
withVerifierScope { scope ->
|
||||
runBlocking {
|
||||
val events = (1..10).map { signed(it) }
|
||||
var delivered = 0
|
||||
val verifier =
|
||||
ParallelEventVerifier<Unit>(
|
||||
scope = scope,
|
||||
maxBatch = 2,
|
||||
onVerified = { _, _ ->
|
||||
delivered++
|
||||
if (delivered == 3) throw IllegalStateException("boom")
|
||||
},
|
||||
)
|
||||
events.forEach { verifier.submit(it, Unit) }
|
||||
verifier.close()
|
||||
withTimeout(30_000) { verifier.join() }
|
||||
|
||||
assertEquals(10, delivered, "all events dispatched despite a throwing callback")
|
||||
assertTrue(verifier.verifiedCount == 10L)
|
||||
}
|
||||
}
|
||||
}
|
||||
+100
@@ -0,0 +1,100 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.commands.toClient
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import java.util.concurrent.atomic.AtomicInteger
|
||||
import kotlin.random.Random
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* The lock-free decoder tolerates its documented benign races (double parse
|
||||
* of a first-seen id, rotation dropping a generation) but must NEVER return
|
||||
* a wrong message. This hammers one shared decoder from many threads — the
|
||||
* production shape: one decoder instance shared by every relay's consumer
|
||||
* coroutine — with duplicate-heavy interleaved streams and a small capacity
|
||||
* so rotations happen constantly, then checks every single result.
|
||||
*/
|
||||
class CachingEventDecoderConcurrencyTest {
|
||||
private fun hexId(seed: Int) = seed.toString(16).padStart(64, '0')
|
||||
|
||||
private fun event(seed: Int) =
|
||||
Event(
|
||||
id = hexId(seed),
|
||||
pubKey = hexId(seed + 500_000),
|
||||
createdAt = seed.toLong(),
|
||||
kind = 1,
|
||||
tags = arrayOf(arrayOf("t", "conc")),
|
||||
content = "concurrency test $seed",
|
||||
sig = "f".repeat(128),
|
||||
)
|
||||
|
||||
@Test
|
||||
fun manyThreadsNeverReceiveAWrongMessage() {
|
||||
val uniques = 2_000
|
||||
val threads = 8
|
||||
val perThreadFrames = 10_000
|
||||
|
||||
val events = (1..uniques).map { event(it) }
|
||||
// tiny capacity → constant rotation → the benign races actually fire
|
||||
val decoder = CachingEventDecoder(capacity = 256)
|
||||
|
||||
val wrongResults = AtomicInteger(0)
|
||||
val decoded = AtomicInteger(0)
|
||||
|
||||
val workers =
|
||||
(1..threads).map { t ->
|
||||
Thread {
|
||||
val rnd = Random(t)
|
||||
repeat(perThreadFrames) { i ->
|
||||
val expected = events[rnd.nextInt(uniques)]
|
||||
val subId = "sub-$t-${i % 7}"
|
||||
val msg = decoder.decode("""["EVENT","$subId",${expected.toJson()}]""")
|
||||
if (msg !is EventMessage ||
|
||||
msg.subId != subId ||
|
||||
msg.event.id != expected.id ||
|
||||
msg.event.content != expected.content
|
||||
) {
|
||||
wrongResults.incrementAndGet()
|
||||
}
|
||||
decoded.incrementAndGet()
|
||||
}
|
||||
}
|
||||
}
|
||||
workers.forEach { it.start() }
|
||||
workers.forEach { it.join() }
|
||||
|
||||
assertEquals(0, wrongResults.get(), "a race must never produce a wrong message")
|
||||
assertEquals(threads * perThreadFrames, decoded.get())
|
||||
assertEquals(
|
||||
decoded.get().toLong(),
|
||||
decoder.parsedCount + decoder.reusedCount,
|
||||
"every frame is either parsed or reused",
|
||||
)
|
||||
// Sanity that the cache functions under constant rotation. With
|
||||
// uniform random access over 2000 ids and only 2×256 cached slots,
|
||||
// the theoretical hit rate is ~25%; assert comfortably above zero
|
||||
// (wholesale cache failure) without flaking on scheduling luck.
|
||||
assertTrue(decoder.reusedCount > decoded.get() / 8, "cache should serve a meaningful share of duplicate frames")
|
||||
}
|
||||
}
|
||||
+716
@@ -0,0 +1,716 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.geode.KtorRelay
|
||||
import com.vitorpamplona.geode.RelayEngine
|
||||
import com.vitorpamplona.geode.fixtures.SyntheticEvents
|
||||
import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.NegentropySyncException
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.negentropySync
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.negentropySyncFanOut
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.server.policies.LimitsPolicy
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.server.policies.RelayLimits
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.async
|
||||
import kotlinx.coroutines.awaitAll
|
||||
import kotlinx.coroutines.coroutineScope
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeoutOrNull
|
||||
import okhttp3.OkHttpClient
|
||||
import okhttp3.Request
|
||||
import okhttp3.Response
|
||||
import java.util.concurrent.CountDownLatch
|
||||
import java.util.concurrent.TimeUnit
|
||||
import java.util.concurrent.atomic.AtomicLong
|
||||
import kotlin.test.Test
|
||||
|
||||
/**
|
||||
* Answers: "we need to download N-million events from a single relay as fast
|
||||
* as possible — considering just download and parsing, what helps?"
|
||||
*
|
||||
* Three parts:
|
||||
*
|
||||
* 1. LOCAL CEILINGS — a real geode relay on a localhost TCP port, seeded with
|
||||
* [LOCAL_EVENTS] events, so network bandwidth and server latency are ~free
|
||||
* and the client stack is what's measured. Variants: 1 connection vs
|
||||
* [SHARDS] connections with the `created_at` range sharded across them;
|
||||
* single giant REQ vs realistic 1000/page cursor pagination (a second
|
||||
* server instance clamps limits to force paging); and the full quartz
|
||||
* stack (parse + dispatch) vs a raw OkHttp socket that only counts frames
|
||||
* (no parse), which isolates the parse share of the pipeline.
|
||||
*
|
||||
* 2. OFFLINE PARSE STRATEGIES — what to do with each frame: full
|
||||
* `fromJsonToMessage`, full parse fanned across cores, or an id-only scan
|
||||
* (the "archive the raw frame now, parse lazily later" strategy).
|
||||
*
|
||||
* 3. PRODUCTION TEST CASE — NIP-77 negentropy sync of `kinds:[30382]` from
|
||||
* nip85.nosfabrica.com (capped at [PROD_MAX_EVENTS]) against plain
|
||||
* until-cursor paging of the same filter. Negentropy enumerates the id set
|
||||
* server-side and downloads by id-batch with 8 concurrent REQs on one
|
||||
* connection — the protocol-level answer to serial page round-trips.
|
||||
*
|
||||
* Gated behind PROD_RELAY_BENCH (same as ProductionReceiverBenchmark):
|
||||
* ./gradlew :quartz:jvmTest --tests "*.BulkDownloadBenchmark" -PprodRelayBench=1
|
||||
*/
|
||||
class BulkDownloadBenchmark {
|
||||
companion object {
|
||||
const val LOCAL_EVENTS = 100_000
|
||||
const val SHARDS = 4
|
||||
const val PAGE_LIMIT = 1_000
|
||||
const val LOCAL_PAGE_TIMEOUT_MS = 120_000L
|
||||
|
||||
const val PROD_RELAY = "wss://nip85.nosfabrica.com"
|
||||
const val PROD_KIND = 30382
|
||||
const val PROD_MAX_EVENTS = 20_000
|
||||
}
|
||||
|
||||
private fun report(
|
||||
name: String,
|
||||
events: Long,
|
||||
bytes: Long,
|
||||
wallNanos: Long,
|
||||
) {
|
||||
println(
|
||||
" %-34s %,8d events %6.1f MB wall=%6.0fms -> %,8.0f events/s %5.1f MB/s"
|
||||
.format(
|
||||
name,
|
||||
events,
|
||||
bytes / 1e6,
|
||||
wallNanos / 1e6,
|
||||
events * 1e9 / wallNanos,
|
||||
bytes * 1e3 / wallNanos,
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Part 1: local ceilings against a seeded geode relay on localhost TCP
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
/** Splits [1, LOCAL_EVENTS] (the seeded createdAt space) into [shards] windows. */
|
||||
private fun windows(shards: Int): List<Pair<Long, Long>> {
|
||||
val step = LOCAL_EVENTS / shards
|
||||
return (0 until shards).map { i ->
|
||||
val lo = i * step + 1L
|
||||
val hi = if (i == shards - 1) LOCAL_EVENTS.toLong() else (i + 1) * step.toLong()
|
||||
lo to hi
|
||||
}
|
||||
}
|
||||
|
||||
/** Full quartz stack: NostrClient + fetchAllPages, one client per shard. */
|
||||
private suspend fun quartzDownload(
|
||||
name: String,
|
||||
relayUrl: NormalizedRelayUrl,
|
||||
httpClient: OkHttpClient,
|
||||
shards: Int,
|
||||
) {
|
||||
val count = AtomicLong(0)
|
||||
val bytes = AtomicLong(0)
|
||||
val start = System.nanoTime()
|
||||
coroutineScope {
|
||||
windows(shards)
|
||||
.map { (lo, hi) ->
|
||||
async(Dispatchers.IO) {
|
||||
val client = NostrClient(BasicOkHttpWebSocket.Builder { httpClient })
|
||||
try {
|
||||
client.fetchAllPages(
|
||||
relay = relayUrl,
|
||||
filters = listOf(Filter(kinds = listOf(1), since = lo, until = hi)),
|
||||
timeoutMs = LOCAL_PAGE_TIMEOUT_MS,
|
||||
) { event ->
|
||||
count.incrementAndGet()
|
||||
bytes.addAndGet(event.content.length.toLong())
|
||||
}
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
}.awaitAll()
|
||||
}
|
||||
val wall = System.nanoTime() - start
|
||||
report(name, count.get(), bytes.get(), wall)
|
||||
if (count.get() != LOCAL_EVENTS.toLong()) {
|
||||
println(" !! expected $LOCAL_EVENTS events, got ${count.get()}")
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Raw OkHttp socket per shard: counts EVENT frames, no JSON parse, no
|
||||
* quartz client. The download-only floor the parse stage sits on top of.
|
||||
*/
|
||||
private fun rawDownload(
|
||||
name: String,
|
||||
serverWsUrl: String,
|
||||
httpClient: OkHttpClient,
|
||||
shards: Int,
|
||||
) {
|
||||
val count = AtomicLong(0)
|
||||
val bytes = AtomicLong(0)
|
||||
val latch = CountDownLatch(shards)
|
||||
val start = System.nanoTime()
|
||||
|
||||
val sockets =
|
||||
windows(shards).mapIndexed { i, (lo, hi) ->
|
||||
httpClient.newWebSocket(
|
||||
Request.Builder().url(serverWsUrl).build(),
|
||||
object : okhttp3.WebSocketListener() {
|
||||
override fun onOpen(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
response: Response,
|
||||
) {
|
||||
webSocket.send("""["REQ","raw$i",{"kinds":[1],"since":$lo,"until":$hi}]""")
|
||||
}
|
||||
|
||||
override fun onMessage(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
text: String,
|
||||
) {
|
||||
if (text.startsWith("[\"EVENT\"")) {
|
||||
count.incrementAndGet()
|
||||
bytes.addAndGet(text.length.toLong())
|
||||
} else if (text.startsWith("[\"EOSE\"")) {
|
||||
latch.countDown()
|
||||
}
|
||||
}
|
||||
|
||||
override fun onFailure(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
t: Throwable,
|
||||
response: Response?,
|
||||
) {
|
||||
latch.countDown()
|
||||
}
|
||||
},
|
||||
)
|
||||
}
|
||||
|
||||
latch.await(120, TimeUnit.SECONDS)
|
||||
val wall = System.nanoTime() - start
|
||||
sockets.forEach { it.cancel() }
|
||||
report(name, count.get(), bytes.get(), wall)
|
||||
}
|
||||
|
||||
private fun localCeilings(httpClient: OkHttpClient) {
|
||||
println("\n=== LOCAL CEILINGS: geode on localhost, $LOCAL_EVENTS seeded events ===")
|
||||
|
||||
val placeholderA = "ws://127.0.0.1:7771/".normalizeRelayUrl()
|
||||
val placeholderB = "ws://127.0.0.1:7772/".normalizeRelayUrl()
|
||||
|
||||
val engine = RelayEngine(url = placeholderA)
|
||||
val seedNanos =
|
||||
runBlocking {
|
||||
val events =
|
||||
(1..LOCAL_EVENTS).map {
|
||||
SyntheticEvents.fakeEvent(
|
||||
idSeed = it,
|
||||
kind = 1,
|
||||
pubKey = SyntheticEvents.hexId(it % 1000 + 1),
|
||||
createdAt = it.toLong(),
|
||||
content = "bulk download benchmark payload $it ".repeat(6),
|
||||
)
|
||||
}
|
||||
val t = System.nanoTime()
|
||||
events.chunked(2_000).forEach { engine.store.batchInsert(it) }
|
||||
System.nanoTime() - t
|
||||
}
|
||||
println(" (seeded in %.1fs)".format(seedNanos / 1e9))
|
||||
|
||||
// Same store, second engine that clamps every REQ to PAGE_LIMIT — forces
|
||||
// fetchAllPages into realistic until-cursor pagination.
|
||||
val pagedEngine =
|
||||
RelayEngine(
|
||||
url = placeholderB,
|
||||
store = engine.store,
|
||||
policyBuilder = { LimitsPolicy(RelayLimits(defaultLimit = PAGE_LIMIT, maxLimit = PAGE_LIMIT)) },
|
||||
)
|
||||
|
||||
val openServer = KtorRelay(engine, port = 0).start()
|
||||
val pagedServer = KtorRelay(pagedEngine, port = 0).start()
|
||||
try {
|
||||
val openUrl = openServer.url.normalizeRelayUrl()
|
||||
val pagedUrl = pagedServer.url.normalizeRelayUrl()
|
||||
|
||||
runBlocking {
|
||||
rawDownload("raw 1-conn single-REQ (no parse)", openServer.url, httpClient, shards = 1)
|
||||
rawDownload("raw $SHARDS-conn sharded (no parse)", openServer.url, httpClient, shards = SHARDS)
|
||||
quartzDownload("quartz 1-conn single-REQ", openUrl, httpClient, shards = 1)
|
||||
quartzDownload("quartz $SHARDS-conn sharded", openUrl, httpClient, shards = SHARDS)
|
||||
quartzDownload("quartz 1-conn paged $PAGE_LIMIT", pagedUrl, httpClient, shards = 1)
|
||||
quartzDownload("quartz $SHARDS-conn sharded+paged", pagedUrl, httpClient, shards = SHARDS)
|
||||
}
|
||||
} finally {
|
||||
openServer.stop(gracePeriodMillis = 200, timeoutMillis = 1_000)
|
||||
pagedServer.stop(gracePeriodMillis = 200, timeoutMillis = 1_000)
|
||||
pagedEngine.close()
|
||||
engine.close()
|
||||
}
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Part 2: offline per-frame parse strategies
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
private fun offlineParseStrategies() {
|
||||
println("\n=== OFFLINE: per-frame strategies (50k synthetic frames) ===")
|
||||
val frames =
|
||||
(1..50_000).map {
|
||||
val event =
|
||||
SyntheticEvents.fakeEvent(
|
||||
idSeed = it,
|
||||
kind = 1,
|
||||
pubKey = SyntheticEvents.hexId(it % 1000 + 1),
|
||||
createdAt = it.toLong(),
|
||||
content = "bulk download benchmark payload $it ".repeat(6),
|
||||
)
|
||||
"""["EVENT","sub",${event.toJson()}]"""
|
||||
}
|
||||
val totalBytes = frames.sumOf { it.length.toLong() }
|
||||
|
||||
// warmup
|
||||
frames.take(5_000).forEach { OptimizedJsonMapper.fromJsonToMessage(it) }
|
||||
|
||||
var t = System.nanoTime()
|
||||
frames.forEach { OptimizedJsonMapper.fromJsonToMessage(it) }
|
||||
report("full parse, 1 thread", frames.size.toLong(), totalBytes, System.nanoTime() - t)
|
||||
|
||||
val cores = Runtime.getRuntime().availableProcessors()
|
||||
t = System.nanoTime()
|
||||
runBlocking(Dispatchers.Default) {
|
||||
frames
|
||||
.chunked((frames.size + cores - 1) / cores)
|
||||
.map { chunk -> async { chunk.forEach { OptimizedJsonMapper.fromJsonToMessage(it) } } }
|
||||
.awaitAll()
|
||||
}
|
||||
report("full parse, $cores threads", frames.size.toLong(), totalBytes, System.nanoTime() - t)
|
||||
|
||||
// id-only scan: what "write the raw frame to disk now, parse lazily
|
||||
// later" pays per frame to dedup/route.
|
||||
var sink = 0
|
||||
t = System.nanoTime()
|
||||
frames.forEach { frame ->
|
||||
val i = frame.indexOf("\"id\":\"")
|
||||
if (i >= 0) sink += frame[i + 6].code
|
||||
}
|
||||
report("id-scan only (archive raw)", frames.size.toLong(), totalBytes, System.nanoTime() - t)
|
||||
check(sink != 0)
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Part 3: production — negentropy vs paging for kind 30382 on nosfabrica
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
private suspend fun productionNegentropy(httpClient: OkHttpClient) {
|
||||
println("\n=== PRODUCTION: $PROD_RELAY kinds=[$PROD_KIND], capped at $PROD_MAX_EVENTS events ===")
|
||||
val relay = PROD_RELAY.normalizeRelayUrl()
|
||||
|
||||
// A) NIP-77 negentropy: server enumerates the id set, we download by
|
||||
// id-batches with 8 concurrent REQs on the same connection.
|
||||
run {
|
||||
val client = NostrClient(BasicOkHttpWebSocket.Builder { httpClient })
|
||||
try {
|
||||
val count = AtomicLong(0)
|
||||
val bytes = AtomicLong(0)
|
||||
val start = System.nanoTime()
|
||||
val result =
|
||||
client.negentropySync(
|
||||
relay = relay,
|
||||
filter = Filter(kinds = listOf(PROD_KIND)),
|
||||
maxEvents = PROD_MAX_EVENTS,
|
||||
onProgress = { need, downloaded ->
|
||||
if (downloaded % 5_000 == 0 && downloaded > 0) {
|
||||
println(" … negentropy progress: need=$need downloaded=$downloaded")
|
||||
}
|
||||
},
|
||||
) { event ->
|
||||
count.incrementAndGet()
|
||||
bytes.addAndGet(event.content.length.toLong())
|
||||
}
|
||||
val wall = System.nanoTime() - start
|
||||
report("negentropy sync", count.get(), bytes.get(), wall)
|
||||
println(
|
||||
" relay reported need=${result.needCount} ids for the full set, " +
|
||||
"reconciled in ${result.windows} window(s), downloaded=${result.downloaded}",
|
||||
)
|
||||
} catch (e: NegentropySyncException) {
|
||||
println(" !! negentropy failed: ${e.reason} — ${e.message}")
|
||||
} catch (e: Exception) {
|
||||
println(" !! negentropy errored: ${e::class.simpleName} ${e.message}")
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
|
||||
// B) Plain until-cursor paging of the same filter, same cap.
|
||||
run {
|
||||
val client = NostrClient(BasicOkHttpWebSocket.Builder { httpClient })
|
||||
try {
|
||||
val count = AtomicLong(0)
|
||||
val bytes = AtomicLong(0)
|
||||
var pages = 0
|
||||
val start = System.nanoTime()
|
||||
client.fetchAllPages(
|
||||
relay = relay,
|
||||
filters = listOf(Filter(kinds = listOf(PROD_KIND), limit = PROD_MAX_EVENTS)),
|
||||
timeoutMs = 30_000L,
|
||||
onNewPage = { pages++ },
|
||||
) { event ->
|
||||
count.incrementAndGet()
|
||||
bytes.addAndGet(event.content.length.toLong())
|
||||
}
|
||||
val wall = System.nanoTime() - start
|
||||
report("paged download", count.get(), bytes.get(), wall)
|
||||
println(" pages=${pages + 1}")
|
||||
} catch (e: Exception) {
|
||||
println(" !! paged download errored: ${e::class.simpleName} ${e.message}")
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// negentropySync pipelining shootout: old sequential behavior
|
||||
// (reconcileConcurrency=1, shallow buffer) vs the tuned pipeline, on the
|
||||
// same production corpus, both capped at the same number of events. The
|
||||
// per-connection by-id ceiling measured by ByIdFetchBenchmark (~5.5k/s on
|
||||
// the 2.6M corpus) is the target; the old defaults measured ~1.6k/s.
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
private suspend fun negentropyVariant(
|
||||
label: String,
|
||||
httpClient: OkHttpClient,
|
||||
maxEvents: Int,
|
||||
maxConcurrentReqs: Int,
|
||||
reconcileConcurrency: Int,
|
||||
idBufferBatches: Int,
|
||||
budgetMs: Long,
|
||||
) {
|
||||
val relay = PROD_RELAY.normalizeRelayUrl()
|
||||
val client = NostrClient(BasicOkHttpWebSocket.Builder { httpClient })
|
||||
val count = AtomicLong(0)
|
||||
val start = System.nanoTime()
|
||||
try {
|
||||
val result =
|
||||
withTimeoutOrNull(budgetMs) {
|
||||
client.negentropySync(
|
||||
relay = relay,
|
||||
filter = Filter(kinds = listOf(PROD_KIND)),
|
||||
maxEvents = maxEvents,
|
||||
maxConcurrentReqs = maxConcurrentReqs,
|
||||
reconcileConcurrency = reconcileConcurrency,
|
||||
idBufferBatches = idBufferBatches,
|
||||
) { count.incrementAndGet() }
|
||||
}
|
||||
val wall = System.nanoTime() - start
|
||||
report(label, count.get(), 0, wall)
|
||||
if (result == null) {
|
||||
println(" (budget ${budgetMs}ms exceeded — partial)")
|
||||
} else {
|
||||
println(" windows=${result.windows} need=${result.needCount}")
|
||||
}
|
||||
} catch (e: NegentropySyncException) {
|
||||
val wall = System.nanoTime() - start
|
||||
report(label, count.get(), 0, wall)
|
||||
println(" !! failed: ${e.reason} ${e.message}")
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun negentropyPipelineShootout() {
|
||||
if (System.getenv("PROD_RELAY_BENCH") == null && System.getProperty("prodRelayBench") == null) {
|
||||
println("negentropyPipelineShootout skipped. Run with -PprodRelayBench=1 to enable.")
|
||||
return
|
||||
}
|
||||
|
||||
val httpClient =
|
||||
OkHttpClient
|
||||
.Builder()
|
||||
.connectTimeout(15, TimeUnit.SECONDS)
|
||||
.readTimeout(120, TimeUnit.SECONDS)
|
||||
.pingInterval(30, TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
println("=== NEGENTROPY PIPELINE SHOOTOUT: $PROD_RELAY kinds=[$PROD_KIND], cap 100k events ===")
|
||||
runBlocking {
|
||||
// Old behavior: sequential windows, shallow id buffer.
|
||||
negentropyVariant(
|
||||
"baseline (8 reqs, seq windows)",
|
||||
httpClient,
|
||||
maxEvents = 100_000,
|
||||
maxConcurrentReqs = 8,
|
||||
reconcileConcurrency = 1,
|
||||
idBufferBatches = 8,
|
||||
budgetMs = 240_000L,
|
||||
)
|
||||
// Tuned: 12 + 4 + 1 keep-alive = 17 subs, under strfry's 20 cap.
|
||||
negentropyVariant(
|
||||
"tuned (12 reqs, 4 reconcilers)",
|
||||
httpClient,
|
||||
maxEvents = 100_000,
|
||||
maxConcurrentReqs = 12,
|
||||
reconcileConcurrency = 4,
|
||||
idBufferBatches = 96,
|
||||
budgetMs = 240_000L,
|
||||
)
|
||||
}
|
||||
httpClient.dispatcher.executorService.shutdown()
|
||||
}
|
||||
|
||||
@Test
|
||||
fun negentropyFanOutShootout() {
|
||||
if (System.getenv("PROD_RELAY_BENCH") == null && System.getProperty("prodRelayBench") == null) {
|
||||
println("negentropyFanOutShootout skipped. Run with -PprodRelayBench=1 to enable.")
|
||||
return
|
||||
}
|
||||
|
||||
val httpClient =
|
||||
OkHttpClient
|
||||
.Builder()
|
||||
.connectTimeout(15, TimeUnit.SECONDS)
|
||||
.readTimeout(120, TimeUnit.SECONDS)
|
||||
.pingInterval(30, TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
val relay = PROD_RELAY.normalizeRelayUrl()
|
||||
println("=== NEGENTROPY FAN-OUT SHOOTOUT: $PROD_RELAY kinds=[$PROD_KIND], cap 100k events ===")
|
||||
|
||||
runBlocking {
|
||||
// Baseline: one connection, tuned pipeline (measured ~4.5k/s).
|
||||
negentropyVariant(
|
||||
"single client (12r/4rc)",
|
||||
httpClient,
|
||||
maxEvents = 100_000,
|
||||
maxConcurrentReqs = 12,
|
||||
reconcileConcurrency = 4,
|
||||
idBufferBatches = 96,
|
||||
budgetMs = 240_000L,
|
||||
)
|
||||
|
||||
// Fan-out: 4 connections, 10 download REQs each + reconcile on the first.
|
||||
val clients = List(4) { NostrClient(BasicOkHttpWebSocket.Builder { httpClient }) }
|
||||
try {
|
||||
val count = AtomicLong(0)
|
||||
val start = System.nanoTime()
|
||||
val result =
|
||||
withTimeoutOrNull(240_000L) {
|
||||
negentropySyncFanOut(
|
||||
clients = clients,
|
||||
relay = relay,
|
||||
filter = Filter(kinds = listOf(PROD_KIND)),
|
||||
maxEvents = 100_000,
|
||||
reqsPerClient = 8,
|
||||
fetchBatch = 250,
|
||||
reconcileConcurrency = 6,
|
||||
) { count.incrementAndGet() }
|
||||
}
|
||||
val wall = System.nanoTime() - start
|
||||
report("fan-out 4 clients x 8 reqs", count.get(), 0, wall)
|
||||
if (result == null) {
|
||||
println(" (budget exceeded — partial)")
|
||||
} else {
|
||||
println(" windows=${result.windows} need=${result.needCount} connectionsUsed=${result.connections}")
|
||||
}
|
||||
} catch (e: Exception) {
|
||||
println(" !! fan-out errored: ${e::class.simpleName} ${e.message}")
|
||||
} finally {
|
||||
clients.forEach { it.close() }
|
||||
}
|
||||
}
|
||||
httpClient.dispatcher.executorService.shutdown()
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Per-connection wall: is a single connection limited by the relay's
|
||||
// pacing or by the client's serial parse? A raw socket that only scans
|
||||
// frames (no JSON parse, no quartz stack) pages the same filter on one
|
||||
// connection. If it caps at the same events/s as the full quartz stack,
|
||||
// the wall is the relay/network; if it runs much faster, the client's
|
||||
// per-connection consumer is the wall and parallel parse would help.
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
private fun rawPagedDownload(
|
||||
relayWsUrl: String,
|
||||
httpClient: OkHttpClient,
|
||||
kind: Int,
|
||||
pageLimit: Int,
|
||||
maxEvents: Int,
|
||||
) {
|
||||
var count = 0L
|
||||
var bytes = 0L
|
||||
var pages = 0
|
||||
var pageCount = 0
|
||||
var pageMinCreatedAt = Long.MAX_VALUE
|
||||
var until = Long.MAX_VALUE / 2
|
||||
|
||||
// Cheap created_at scan — the only per-frame work besides counting.
|
||||
fun scanCreatedAt(frame: String): Long {
|
||||
val i = frame.indexOf("\"created_at\":")
|
||||
if (i < 0) return Long.MAX_VALUE
|
||||
var j = i + 13
|
||||
var v = 0L
|
||||
while (j < frame.length && frame[j].isDigit()) {
|
||||
v = v * 10 + (frame[j] - '0')
|
||||
j++
|
||||
}
|
||||
return v
|
||||
}
|
||||
|
||||
val start = System.nanoTime()
|
||||
val latch = CountDownLatch(1)
|
||||
val socket =
|
||||
httpClient.newWebSocket(
|
||||
Request.Builder().url(relayWsUrl).build(),
|
||||
object : okhttp3.WebSocketListener() {
|
||||
override fun onOpen(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
response: Response,
|
||||
) {
|
||||
// Compression matters 3-5x on bandwidth-limited links; report
|
||||
// whether the relay actually negotiated it with OkHttp.
|
||||
println(" negotiated Sec-WebSocket-Extensions: ${response.header("Sec-WebSocket-Extensions") ?: "(none — uncompressed)"}")
|
||||
webSocket.send("""["REQ","raw",{"kinds":[$kind],"limit":$pageLimit,"until":$until}]""")
|
||||
}
|
||||
|
||||
override fun onMessage(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
text: String,
|
||||
) {
|
||||
if (text.startsWith("[\"EVENT\"")) {
|
||||
count++
|
||||
bytes += text.length
|
||||
pageCount++
|
||||
val ts = scanCreatedAt(text)
|
||||
if (ts < pageMinCreatedAt) pageMinCreatedAt = ts
|
||||
} else if (text.startsWith("[\"EOSE\"")) {
|
||||
pages++
|
||||
val done = pageCount == 0 || count >= maxEvents || pageMinCreatedAt == Long.MAX_VALUE
|
||||
if (done) {
|
||||
latch.countDown()
|
||||
} else {
|
||||
until = pageMinCreatedAt - 1
|
||||
pageCount = 0
|
||||
pageMinCreatedAt = Long.MAX_VALUE
|
||||
webSocket.send("""["REQ","raw",{"kinds":[$kind],"limit":$pageLimit,"until":$until}]""")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
override fun onFailure(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
t: Throwable,
|
||||
response: Response?,
|
||||
) {
|
||||
println(" !! raw socket failure: ${t.message}")
|
||||
latch.countDown()
|
||||
}
|
||||
},
|
||||
)
|
||||
|
||||
latch.await(120, TimeUnit.SECONDS)
|
||||
val wall = System.nanoTime() - start
|
||||
socket.cancel()
|
||||
report("raw paged (no parse), 1 conn", count, bytes, wall)
|
||||
println(" pages=$pages")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun perConnectionWall() {
|
||||
if (System.getenv("PROD_RELAY_BENCH") == null && System.getProperty("prodRelayBench") == null) {
|
||||
println("perConnectionWall skipped. Run with -PprodRelayBench=1 to enable.")
|
||||
return
|
||||
}
|
||||
|
||||
val httpClient =
|
||||
OkHttpClient
|
||||
.Builder()
|
||||
.connectTimeout(15, TimeUnit.SECONDS)
|
||||
.readTimeout(120, TimeUnit.SECONDS)
|
||||
.pingInterval(30, TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
println("\n=== PER-CONNECTION WALL: $PROD_RELAY kinds=[$PROD_KIND], 1 connection, cap $PROD_MAX_EVENTS ===")
|
||||
|
||||
// A) raw socket, until-cursor paging, no JSON parse at all
|
||||
rawPagedDownload(PROD_RELAY, httpClient, PROD_KIND, pageLimit = 500, maxEvents = PROD_MAX_EVENTS)
|
||||
|
||||
// B) full quartz stack, same query, same connection count
|
||||
runBlocking {
|
||||
val relay = PROD_RELAY.normalizeRelayUrl()
|
||||
val client = NostrClient(BasicOkHttpWebSocket.Builder { httpClient })
|
||||
try {
|
||||
val count = AtomicLong(0)
|
||||
val bytes = AtomicLong(0)
|
||||
var pages = 0
|
||||
val start = System.nanoTime()
|
||||
client.fetchAllPages(
|
||||
relay = relay,
|
||||
filters = listOf(Filter(kinds = listOf(PROD_KIND), limit = PROD_MAX_EVENTS)),
|
||||
timeoutMs = 30_000L,
|
||||
onNewPage = { pages++ },
|
||||
) { event ->
|
||||
count.incrementAndGet()
|
||||
bytes.addAndGet(event.content.length.toLong())
|
||||
}
|
||||
report("quartz paged (full parse), 1 conn", count.get(), bytes.get(), System.nanoTime() - start)
|
||||
println(" pages=${pages + 1}")
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
|
||||
httpClient.dispatcher.executorService.shutdown()
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
fun bulkDownloadBenchmark() {
|
||||
if (System.getenv("PROD_RELAY_BENCH") == null && System.getProperty("prodRelayBench") == null) {
|
||||
println("BulkDownloadBenchmark skipped. Run with -PprodRelayBench=1 to enable.")
|
||||
return
|
||||
}
|
||||
|
||||
val httpClient =
|
||||
OkHttpClient
|
||||
.Builder()
|
||||
.connectTimeout(15, TimeUnit.SECONDS)
|
||||
.readTimeout(120, TimeUnit.SECONDS)
|
||||
.pingInterval(30, TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
println("=== BULK DOWNLOAD BENCHMARK === cores=${Runtime.getRuntime().availableProcessors()}")
|
||||
|
||||
localCeilings(httpClient)
|
||||
offlineParseStrategies()
|
||||
runBlocking { productionNegentropy(httpClient) }
|
||||
|
||||
httpClient.dispatcher.executorService.shutdown()
|
||||
}
|
||||
}
|
||||
+340
@@ -0,0 +1,340 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.core.HexKey
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.fetchAllPages
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import kotlinx.coroutines.coroutineScope
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.flow.first
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeoutOrNull
|
||||
import okhttp3.OkHttpClient
|
||||
import java.util.concurrent.ConcurrentHashMap
|
||||
import java.util.concurrent.TimeUnit
|
||||
import java.util.concurrent.atomic.AtomicLong
|
||||
import kotlin.random.Random
|
||||
import kotlin.test.Test
|
||||
|
||||
/**
|
||||
* Parallel fetch-by-id against one production relay, measured properly:
|
||||
*
|
||||
* - the FULL corpus, not a sample: every kind-[KIND] id the relay holds is
|
||||
* enumerated first (until-cursor paging, untimed), then each cell of the
|
||||
* matrix re-downloads the complete set by id batches;
|
||||
* - a matrix over (connections × concurrent REQs per connection), the two
|
||||
* axes the user report disputes: "concurrency within one connection stops
|
||||
* helping past ~4 REQs; throughput scales with connections";
|
||||
* - connections are pre-established before the clock starts (a keep-alive
|
||||
* sub pins them), so cells time steady-state download, not TLS handshakes;
|
||||
* - per-cell accounting of missing ids and timed-out batches, so a cell that
|
||||
* hit a relay-side subscription cap (e.g. 16 concurrent REQs rejected)
|
||||
* reads as a shortfall instead of fake speed.
|
||||
*
|
||||
* Batches are [BATCH] ids each, shuffled with a fixed seed so every cell gets
|
||||
* the same work in the same order. Events are counted, not retained.
|
||||
*
|
||||
* Gated (opens live sockets, downloads the corpus once per cell):
|
||||
* ./gradlew :quartz:jvmTest --tests "*.ByIdFetchBenchmark" -PprodRelayBench=1
|
||||
*/
|
||||
class ByIdFetchBenchmark {
|
||||
companion object {
|
||||
const val RELAY = "wss://nip85.nosfabrica.com"
|
||||
const val KIND = 30382
|
||||
|
||||
/**
|
||||
* Small enough that the corpus yields several batches per worker even
|
||||
* at 160 in-flight REQs (33k ids / 125 = ~264 batches); otherwise the
|
||||
* high-concurrency cells measure tail/wave effects, not throughput.
|
||||
*/
|
||||
const val BATCH = 125
|
||||
const val BATCH_TIMEOUT_MS = 10_000L
|
||||
const val ENUM_PAGE_TIMEOUT_MS = 30_000L
|
||||
|
||||
/**
|
||||
* The relay's NIP-11 `limitation.max_subscriptions` (strfry: 20 on
|
||||
* nip85.nosfabrica.com). REQs beyond this on ONE connection are
|
||||
* rejected/dropped — the first matrix run proved it the hard way (a
|
||||
* 1×128 cell wedged the connection and every batch burned its full
|
||||
* timeout). Axis 1 stops here; more in-flight requires connections.
|
||||
*/
|
||||
const val RELAY_MAX_SUBS = 20
|
||||
|
||||
/** Hard wall-clock budget per cell so a saturated cell fails fast. */
|
||||
const val CELL_BUDGET_MS = 120_000L
|
||||
|
||||
/** A never-matching id (all f's) to pin connections open. */
|
||||
val KEEP_ALIVE_ID = "f".repeat(64)
|
||||
}
|
||||
|
||||
class CellResult(
|
||||
val label: String,
|
||||
val received: Long,
|
||||
val bytes: Long,
|
||||
val uniques: Int,
|
||||
val requested: Int,
|
||||
val batches: Int,
|
||||
val inFlight: Int,
|
||||
val retriedBatches: Int,
|
||||
val failedBatches: Int,
|
||||
val wallNanos: Long,
|
||||
) {
|
||||
override fun toString(): String =
|
||||
" %-22s %,8d events wall=%7.0fms -> %,7.0f events/s reqLat~%4.0fms%s%s%s"
|
||||
.format(
|
||||
label,
|
||||
received,
|
||||
wallNanos / 1e6,
|
||||
received * 1e9 / wallNanos,
|
||||
// Little's-law effective per-REQ latency over COMPLETED
|
||||
// batches (cells can hit their budget with work left);
|
||||
// inflates when the relay (or the client) saturates.
|
||||
wallNanos / 1e6 * inFlight / (received.toDouble() / BATCH).coerceAtLeast(1.0),
|
||||
if (uniques < requested) " MISSING=${requested - uniques}" else "",
|
||||
if (retriedBatches > 0) " RETRIED=$retriedBatches" else "",
|
||||
if (failedBatches > 0) " FAILED=$failedBatches" else "",
|
||||
)
|
||||
}
|
||||
|
||||
/** One REQ for [batch] ids on [client]; returns when EOSE/closed/timeout. */
|
||||
private suspend fun fetchBatch(
|
||||
client: NostrClient,
|
||||
relay: NormalizedRelayUrl,
|
||||
batch: List<HexKey>,
|
||||
onEvent: (Event) -> Unit,
|
||||
): Boolean {
|
||||
val subId = newSubId()
|
||||
val done = Channel<Boolean>(Channel.CONFLATED)
|
||||
val listener =
|
||||
object : SubscriptionListener {
|
||||
override fun onEvent(
|
||||
event: Event,
|
||||
isLive: Boolean,
|
||||
relay: NormalizedRelayUrl,
|
||||
forFilters: List<Filter>?,
|
||||
) = onEvent(event)
|
||||
|
||||
override fun onEose(
|
||||
relay: NormalizedRelayUrl,
|
||||
forFilters: List<Filter>?,
|
||||
) {
|
||||
done.trySend(true)
|
||||
}
|
||||
|
||||
override fun onClosed(
|
||||
message: String,
|
||||
relay: NormalizedRelayUrl,
|
||||
forFilters: List<Filter>?,
|
||||
) {
|
||||
done.trySend(false)
|
||||
}
|
||||
|
||||
override fun onCannotConnect(
|
||||
relay: NormalizedRelayUrl,
|
||||
message: String,
|
||||
forFilters: List<Filter>?,
|
||||
) {
|
||||
done.trySend(false)
|
||||
}
|
||||
}
|
||||
return try {
|
||||
client.subscribe(subId, mapOf(relay to listOf(Filter(ids = batch))), listener)
|
||||
withTimeoutOrNull(BATCH_TIMEOUT_MS) { done.receive() } ?: false
|
||||
} finally {
|
||||
client.unsubscribe(subId)
|
||||
done.close()
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs one matrix cell: [connections] independent NostrClients (one socket
|
||||
* each) to the same relay, [reqsPerConn] workers per client pulling id
|
||||
* batches off a shared queue. Clients are connected before timing starts.
|
||||
*/
|
||||
private suspend fun runCell(
|
||||
label: String,
|
||||
relay: NormalizedRelayUrl,
|
||||
httpClient: OkHttpClient,
|
||||
connections: Int,
|
||||
reqsPerConn: Int,
|
||||
batches: List<List<HexKey>>,
|
||||
quiet: Boolean = false,
|
||||
): CellResult? {
|
||||
val clients = List(connections) { NostrClient(BasicOkHttpWebSocket.Builder { httpClient }) }
|
||||
val keepAlives = clients.map { it to newSubId() }
|
||||
try {
|
||||
// Pre-connect every socket before the clock starts.
|
||||
keepAlives.forEach { (client, subId) ->
|
||||
client.subscribe(subId, mapOf(relay to listOf(Filter(ids = listOf(KEEP_ALIVE_ID)))), null)
|
||||
}
|
||||
for (client in clients) {
|
||||
val ok = withTimeoutOrNull(20_000L) { client.connectedRelaysFlow().first { relay in it } }
|
||||
if (ok == null) {
|
||||
println(" !! $label: could not pre-connect all sockets, skipping cell")
|
||||
return null
|
||||
}
|
||||
}
|
||||
|
||||
val received = AtomicLong(0)
|
||||
val bytes = AtomicLong(0)
|
||||
val uniques = ConcurrentHashMap.newKeySet<HexKey>(batches.size * BATCH)
|
||||
val retried = AtomicLong(0)
|
||||
val failed = AtomicLong(0)
|
||||
|
||||
val queue = Channel<List<HexKey>>(Channel.UNLIMITED)
|
||||
batches.forEach { queue.trySend(it) }
|
||||
queue.close()
|
||||
|
||||
val onEvent: (Event) -> Unit = { event ->
|
||||
received.incrementAndGet()
|
||||
// content only — cheap; kind-30382 payloads live in
|
||||
// tags, so this is a floor, not wire bytes
|
||||
bytes.addAndGet(event.content.length.toLong())
|
||||
uniques.add(event.id)
|
||||
}
|
||||
|
||||
val start = System.nanoTime()
|
||||
val finished =
|
||||
withTimeoutOrNull(CELL_BUDGET_MS) {
|
||||
coroutineScope {
|
||||
clients.forEach { client ->
|
||||
repeat(reqsPerConn) {
|
||||
launch(Dispatchers.IO) {
|
||||
for (batch in queue) {
|
||||
var ok = fetchBatch(client, relay, batch, onEvent)
|
||||
if (!ok) {
|
||||
// one retry so a relay-side subscription cap or a
|
||||
// dropped REQ reads as RETRIED, not silent shortfall
|
||||
retried.incrementAndGet()
|
||||
ok = fetchBatch(client, relay, batch, onEvent)
|
||||
}
|
||||
if (!ok) failed.incrementAndGet()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
val wall = System.nanoTime() - start
|
||||
if (finished == null) println(" !! $label: cell budget (${CELL_BUDGET_MS}ms) exceeded — partial result")
|
||||
|
||||
val result =
|
||||
CellResult(
|
||||
label = label,
|
||||
received = received.get(),
|
||||
bytes = bytes.get(),
|
||||
uniques = uniques.size,
|
||||
requested = batches.sumOf { it.size },
|
||||
batches = batches.size,
|
||||
inFlight = connections * reqsPerConn,
|
||||
retriedBatches = retried.get().toInt(),
|
||||
failedBatches = failed.get().toInt(),
|
||||
wallNanos = wall,
|
||||
)
|
||||
if (!quiet) println(result)
|
||||
return result
|
||||
} finally {
|
||||
keepAlives.forEach { (client, subId) -> client.unsubscribe(subId) }
|
||||
clients.forEach { it.close() }
|
||||
}
|
||||
}
|
||||
|
||||
/** Enumerates every id of [KIND] on the relay via until-cursor paging (untimed). */
|
||||
private suspend fun enumerateIds(
|
||||
relay: NormalizedRelayUrl,
|
||||
httpClient: OkHttpClient,
|
||||
): List<HexKey> {
|
||||
val client = NostrClient(BasicOkHttpWebSocket.Builder { httpClient })
|
||||
return try {
|
||||
val ids = LinkedHashSet<HexKey>(64_000)
|
||||
val t = System.nanoTime()
|
||||
client.fetchAllPages(
|
||||
relay = relay,
|
||||
filters = listOf(Filter(kinds = listOf(KIND))),
|
||||
timeoutMs = ENUM_PAGE_TIMEOUT_MS,
|
||||
) { event -> ids.add(event.id) }
|
||||
println(" enumerated ${ids.size} ids in %.1fs (paged, untimed baseline for the matrix)".format((System.nanoTime() - t) / 1e9))
|
||||
ids.toList()
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun parallelByIdMatrix() {
|
||||
if (System.getenv("PROD_RELAY_BENCH") == null && System.getProperty("prodRelayBench") == null) {
|
||||
println("ByIdFetchBenchmark skipped. Run with -PprodRelayBench=1 to enable.")
|
||||
return
|
||||
}
|
||||
|
||||
val httpClient =
|
||||
OkHttpClient
|
||||
.Builder()
|
||||
.connectTimeout(15, TimeUnit.SECONDS)
|
||||
.readTimeout(60, TimeUnit.SECONDS)
|
||||
.pingInterval(30, TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
val relay = RELAY.normalizeRelayUrl()
|
||||
|
||||
runBlocking {
|
||||
println("=== PARALLEL FETCH-BY-ID MATRIX: $RELAY kinds=[$KIND], batch=$BATCH ids/REQ ===")
|
||||
|
||||
val ids = enumerateIds(relay, httpClient)
|
||||
if (ids.size < 1_000) {
|
||||
println(" !! only ${ids.size} ids — relay unreachable or corpus gone; aborting")
|
||||
return@runBlocking
|
||||
}
|
||||
|
||||
// Same shuffled batching for every cell.
|
||||
val batches = ids.shuffled(Random(42)).chunked(BATCH)
|
||||
println(" corpus=${ids.size} ids in ${batches.size} batches; every cell downloads the full corpus\n")
|
||||
|
||||
// Warmup: JIT + server page cache, untimed.
|
||||
runCell("warmup", relay, httpClient, connections = 2, reqsPerConn = 4, batches = batches.take(24), quiet = true)
|
||||
|
||||
println(" --- axis 1: concurrent REQs on ONE connection (relay caps subs at $RELAY_MAX_SUBS) ---")
|
||||
for (reqs in listOf(1, 4, 8, 16, RELAY_MAX_SUBS)) {
|
||||
runCell("1 conn x $reqs reqs", relay, httpClient, connections = 1, reqsPerConn = reqs, batches = batches)
|
||||
delay(2_000)
|
||||
}
|
||||
|
||||
println(" --- axis 2: scaling total in-flight past the per-connection cap via connections (10 REQs each) ---")
|
||||
for (conns in listOf(2, 4, 8, 16)) {
|
||||
runCell("$conns conns x 10 reqs", relay, httpClient, connections = conns, reqsPerConn = 10, batches = batches)
|
||||
delay(2_000)
|
||||
}
|
||||
}
|
||||
|
||||
httpClient.dispatcher.executorService.shutdown()
|
||||
}
|
||||
}
|
||||
+108
@@ -0,0 +1,108 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.geode.fixtures.SyntheticEvents
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.CachingEventDecoder
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.MessageDecoder
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* Proves the [CachingEventDecoder] saves what it claims at production
|
||||
* duplicate factors (14–57% duplicate frames measured; see
|
||||
* `quartz/plans/2026-07-02-nostrclient-receiver-perf.md`): a duplicate frame
|
||||
* pays an id scan (~sub-µs) instead of a full JSON parse.
|
||||
*
|
||||
* Offline and deterministic: same frame stream through
|
||||
* [MessageDecoder.Default] and through [CachingEventDecoder]. The stream
|
||||
* interleaves 3 deliveries of each event (one per "subscription"), matching
|
||||
* how overlapping subs replay the same event on one connection.
|
||||
*/
|
||||
class DedupDecodeBenchmark {
|
||||
companion object {
|
||||
const val UNIQUE_EVENTS = 20_000
|
||||
const val DELIVERIES_PER_EVENT = 3 // ~66% duplicate frames
|
||||
}
|
||||
|
||||
private fun buildFrames(): List<String> {
|
||||
val events =
|
||||
(1..UNIQUE_EVENTS).map {
|
||||
SyntheticEvents.fakeEvent(
|
||||
idSeed = it,
|
||||
kind = 1,
|
||||
pubKey = SyntheticEvents.hexId(it % 500 + 1),
|
||||
createdAt = it.toLong(),
|
||||
content = "dedup decode benchmark payload $it ".repeat(6),
|
||||
)
|
||||
}
|
||||
return buildList {
|
||||
repeat(DELIVERIES_PER_EVENT) { sub ->
|
||||
events.forEach { add("""["EVENT","sub$sub",${it.toJson()}]""") }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private fun run(
|
||||
decoder: MessageDecoder,
|
||||
frames: List<String>,
|
||||
): Long {
|
||||
val start = System.nanoTime()
|
||||
frames.forEach { decoder.decode(it) }
|
||||
return System.nanoTime() - start
|
||||
}
|
||||
|
||||
@Test
|
||||
fun duplicateFramesDecodeFasterThroughTheCache() {
|
||||
val frames = buildFrames()
|
||||
val dupShare = 1.0 - 1.0 / DELIVERIES_PER_EVENT
|
||||
|
||||
// warmup both paths
|
||||
run(MessageDecoder.Default, frames.take(20_000))
|
||||
run(CachingEventDecoder(capacity = UNIQUE_EVENTS * 2), frames.take(20_000))
|
||||
|
||||
val fullNanos = run(MessageDecoder.Default, frames)
|
||||
|
||||
val caching = CachingEventDecoder(capacity = UNIQUE_EVENTS * 2)
|
||||
val cachedNanos = run(caching, frames)
|
||||
|
||||
val speedup = fullNanos.toDouble() / cachedNanos
|
||||
println("=== DEDUP DECODE BENCHMARK (${frames.size} frames, %.0f%% duplicates) ===".format(dupShare * 100))
|
||||
println(
|
||||
" full parse every frame: %.1fms (%.1fµs/frame)"
|
||||
.format(fullNanos / 1e6, fullNanos / 1e3 / frames.size),
|
||||
)
|
||||
println(
|
||||
" caching decoder: %.1fms (%.1fµs/frame) parsed=%d reused=%d -> %.2fx"
|
||||
.format(cachedNanos / 1e6, cachedNanos / 1e3 / frames.size, caching.parsedCount, caching.reusedCount, speedup),
|
||||
)
|
||||
|
||||
assertTrue(
|
||||
caching.parsedCount == UNIQUE_EVENTS.toLong() &&
|
||||
caching.reusedCount == (frames.size - UNIQUE_EVENTS).toLong(),
|
||||
"every duplicate must hit the cache",
|
||||
)
|
||||
// The claim that justifies the code: at a 66% duplicate share the
|
||||
// caching decoder must be meaningfully faster than parsing everything.
|
||||
// Generous threshold (1.5x) so CI noise doesn't flake; measured ~2.5-3x.
|
||||
assertTrue(speedup > 1.5, "expected >1.5x speedup, got %.2fx".format(speedup))
|
||||
}
|
||||
}
|
||||
+310
@@ -0,0 +1,310 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.geode.fixtures.SyntheticEvents
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.EventCollector
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.pool.PoolRequests
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.single.IRelayClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.EventMessage
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocket
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocketListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import java.util.concurrent.ConcurrentHashMap
|
||||
import java.util.concurrent.atomic.AtomicLong
|
||||
import kotlin.test.Test
|
||||
|
||||
/**
|
||||
* Microbenchmark for the receiver's DISPATCH stage: everything that happens to
|
||||
* a message AFTER the JSON parse and BEFORE signature verification —
|
||||
*
|
||||
* NostrClient.onIncomingMessage
|
||||
* ├─ PoolRequests.onIncomingMessage (spin lock + sub state machine + sub listener)
|
||||
* ├─ PoolCounts / PoolEventOutbox (no-ops for EVENT frames)
|
||||
* └─ connection listeners (EventCollector → dedup → handoff to verify)
|
||||
*
|
||||
* The goal is to find the arrangement of this stage that maximizes events/s,
|
||||
* since this is the code that decides how fast frames leave the per-relay
|
||||
* receiver coroutine once verification is moved off it. Offline and synthetic
|
||||
* on purpose: this stage's cost is pure CPU + locks and doesn't depend on
|
||||
* relay behavior. Duplicate factors are modeled on what the production runs
|
||||
* measured (each event delivered on S overlapping subs and again by F relays;
|
||||
* see ProductionReceiverBenchmark: 14–57% duplicates).
|
||||
*
|
||||
* Variants:
|
||||
* - dispatch-only floor: full NostrClient dispatch, no-op collector
|
||||
* - dedup + ConcurrentHashMap id dedup (the verify gatekeeper)
|
||||
* - dedup+chan + per-event Channel handoff for unique events
|
||||
* - dedup+batch64 + batched handoff (64/batch, IngestQueue-style)
|
||||
* - early-dedup+batch64 dedup BEFORE NostrClient dispatch: duplicate frames
|
||||
* skip the whole dispatch machinery (models checking
|
||||
* seen-ids right after parse)
|
||||
*
|
||||
* Each variant runs with 1 feeder thread (one busy relay) and with
|
||||
* min(4, cores) feeder threads (several relays bursting at once, which is what
|
||||
* exposes the PoolRequests spin lock). A PoolRequests-only pass attributes how
|
||||
* much of the stage is the subscription state machine itself.
|
||||
*
|
||||
* Run with: ./gradlew :quartz:jvmTest --tests "*.DispatchStageBenchmark"
|
||||
*/
|
||||
class DispatchStageBenchmark {
|
||||
companion object {
|
||||
const val UNIQUE_EVENTS = 30_000
|
||||
const val SUBS_PER_RELAY = 2 // overlapping subs: same event delivered once per sub
|
||||
val MULTI_FEEDERS = minOf(4, Runtime.getRuntime().availableProcessors())
|
||||
const val BATCH_SIZE = 64
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Fakes: a socket that swallows everything, so NostrClient/PoolRequests
|
||||
// run their real logic with no network and no reconnect churn.
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
class NoopWebSocket : WebSocket {
|
||||
override fun needsReconnect() = false
|
||||
|
||||
override fun connect() {}
|
||||
|
||||
override fun disconnect() {}
|
||||
|
||||
override fun send(msg: String) = true
|
||||
}
|
||||
|
||||
class NoopBuilder : WebsocketBuilder {
|
||||
override fun build(
|
||||
url: NormalizedRelayUrl,
|
||||
out: WebSocketListener,
|
||||
): WebSocket = NoopWebSocket()
|
||||
}
|
||||
|
||||
private val noopSubListener = object : SubscriptionListener {}
|
||||
|
||||
private val events: List<Event> by lazy {
|
||||
(1..UNIQUE_EVENTS).map { SyntheticEvents.fakeEvent(idSeed = it, kind = 1, pubKey = SyntheticEvents.hexId(it % 500 + 1)) }
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// Harness
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
class Handoff(
|
||||
val batchSize: Int?,
|
||||
) {
|
||||
val channel: Channel<List<Event>> = Channel(Channel.UNLIMITED)
|
||||
val handedOff = AtomicLong(0)
|
||||
|
||||
// one batch buffer per feeder thread; flushed when full
|
||||
private val buffer = ThreadLocal.withInitial { ArrayList<Event>(batchSize ?: 1) }
|
||||
|
||||
fun offer(event: Event) {
|
||||
if (batchSize == null) {
|
||||
channel.trySend(listOf(event))
|
||||
handedOff.incrementAndGet()
|
||||
} else {
|
||||
val buf = buffer.get()
|
||||
buf.add(event)
|
||||
if (buf.size >= batchSize) {
|
||||
channel.trySend(ArrayList(buf))
|
||||
handedOff.addAndGet(buf.size.toLong())
|
||||
buf.clear()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
class VariantResult(
|
||||
val name: String,
|
||||
val feeders: Int,
|
||||
val deliveries: Long,
|
||||
val uniques: Long,
|
||||
val wallNanos: Long,
|
||||
) {
|
||||
override fun toString(): String =
|
||||
" %-24s feeders=%d %,10.0f deliveries/s %,10.0f uniques/s wall=%.1fms".format(
|
||||
name,
|
||||
feeders,
|
||||
deliveries * 1e9 / wallNanos,
|
||||
uniques * 1e9 / wallNanos,
|
||||
wallNanos / 1e6,
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* Runs one variant: [feeders] threads, each acting as one relay's consumer
|
||||
* coroutine, delivering every event on [SUBS_PER_RELAY] subs through the
|
||||
* real NostrClient dispatch path.
|
||||
*/
|
||||
private fun runVariant(
|
||||
name: String,
|
||||
feeders: Int,
|
||||
earlyDedup: Boolean,
|
||||
sink: (Event, Handoff?, MutableSet<String>?) -> Unit,
|
||||
handoffBatch: Int? = null,
|
||||
useDedup: Boolean = false,
|
||||
): VariantResult {
|
||||
val client = NostrClient(NoopBuilder())
|
||||
try {
|
||||
val relayUrls = (1..feeders).map { "ws://bench-relay-$it.local".normalizeRelayUrl() }
|
||||
val subIds = (1..SUBS_PER_RELAY).map { "bench-sub-$it" }
|
||||
|
||||
// register subs on every relay so PoolRequests has real state to update
|
||||
subIds.forEach { subId ->
|
||||
client.subscribe(subId, relayUrls.associateWith { listOf(Filter(kinds = listOf(1))) }, noopSubListener)
|
||||
}
|
||||
|
||||
val seen: MutableSet<String>? = if (useDedup || earlyDedup) ConcurrentHashMap.newKeySet(UNIQUE_EVENTS * 2) else null
|
||||
val handoff = handoffBatch?.let { Handoff(if (it == 0) null else it) }
|
||||
|
||||
val collector =
|
||||
EventCollector(client) { event, _ ->
|
||||
sink(event, handoff, seen)
|
||||
}
|
||||
|
||||
val relayClients = relayUrls.map { client.getOrCreateRelay(it) }
|
||||
|
||||
val threads =
|
||||
relayClients.map { relayClient ->
|
||||
Thread {
|
||||
for (event in events) {
|
||||
if (earlyDedup && !seen!!.add(event.id)) {
|
||||
// duplicate frame: skip the whole dispatch stage,
|
||||
// exactly what a post-parse id check would do
|
||||
continue
|
||||
}
|
||||
for (subId in subIds) {
|
||||
client.onIncomingMessage(relayClient, "", EventMessage(subId, event))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val start = System.nanoTime()
|
||||
threads.forEach { it.start() }
|
||||
threads.forEach { it.join() }
|
||||
val wall = System.nanoTime() - start
|
||||
|
||||
handoff?.channel?.close()
|
||||
|
||||
// every feeder sees every event on every sub; early-dedup "skips"
|
||||
// still count as frames the receiver got past (that's the point)
|
||||
val deliveries = feeders.toLong() * UNIQUE_EVENTS * SUBS_PER_RELAY
|
||||
val uniqueCount = seen?.size?.toLong() ?: UNIQUE_EVENTS.toLong()
|
||||
return VariantResult(name, feeders, deliveries, uniqueCount, wall).also {
|
||||
collector.destroy()
|
||||
}
|
||||
} finally {
|
||||
client.close()
|
||||
}
|
||||
}
|
||||
|
||||
// sinks -------------------------------------------------------------
|
||||
|
||||
private val sinkNoop: (Event, Handoff?, MutableSet<String>?) -> Unit = { _, _, _ -> }
|
||||
|
||||
private val sinkDedup: (Event, Handoff?, MutableSet<String>?) -> Unit = { event, _, seen ->
|
||||
seen!!.add(event.id)
|
||||
}
|
||||
|
||||
private val sinkDedupHandoff: (Event, Handoff?, MutableSet<String>?) -> Unit = { event, handoff, seen ->
|
||||
if (seen!!.add(event.id)) handoff!!.offer(event)
|
||||
}
|
||||
|
||||
private val sinkHandoffOnly: (Event, Handoff?, MutableSet<String>?) -> Unit = { event, handoff, _ ->
|
||||
handoff!!.offer(event)
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
// PoolRequests in isolation: attributes the sub-state-machine + spin-lock
|
||||
// share of the stage.
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
private fun runPoolRequestsOnly(feeders: Int): VariantResult {
|
||||
val pool = PoolRequests()
|
||||
val client = NostrClient(NoopBuilder())
|
||||
try {
|
||||
val relayUrls = (1..feeders).map { "ws://bench-relay-$it.local".normalizeRelayUrl() }
|
||||
val subIds = (1..SUBS_PER_RELAY).map { "bench-sub-$it" }
|
||||
subIds.forEach { subId ->
|
||||
pool.addOrUpdate(subId, relayUrls.associateWith { listOf(Filter(kinds = listOf(1))) }, noopSubListener)
|
||||
}
|
||||
val relayClients: List<IRelayClient> = relayUrls.map { client.getOrCreateRelay(it) }
|
||||
|
||||
val threads =
|
||||
relayClients.map { relayClient ->
|
||||
Thread {
|
||||
for (event in events) {
|
||||
for (subId in subIds) {
|
||||
pool.onIncomingMessage(relayClient, EventMessage(subId, event))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
val start = System.nanoTime()
|
||||
threads.forEach { it.start() }
|
||||
threads.forEach { it.join() }
|
||||
val wall = System.nanoTime() - start
|
||||
|
||||
val deliveries = feeders.toLong() * UNIQUE_EVENTS * SUBS_PER_RELAY
|
||||
return VariantResult("PoolRequests-only", feeders, deliveries, UNIQUE_EVENTS.toLong(), wall)
|
||||
} finally {
|
||||
client.close()
|
||||
pool.destroy()
|
||||
}
|
||||
}
|
||||
|
||||
// ------------------------------------------------------------------
|
||||
|
||||
private fun runAllVariants(print: Boolean) {
|
||||
for (feeders in listOf(1, MULTI_FEEDERS)) {
|
||||
val results =
|
||||
listOf(
|
||||
runPoolRequestsOnly(feeders),
|
||||
runVariant("dispatch-only", feeders, earlyDedup = false, sink = sinkNoop),
|
||||
runVariant("dedup", feeders, earlyDedup = false, sink = sinkDedup, useDedup = true),
|
||||
runVariant("dedup+chan", feeders, earlyDedup = false, sink = sinkDedupHandoff, handoffBatch = 0, useDedup = true),
|
||||
runVariant("dedup+batch$BATCH_SIZE", feeders, earlyDedup = false, sink = sinkDedupHandoff, handoffBatch = BATCH_SIZE, useDedup = true),
|
||||
runVariant("early-dedup+batch$BATCH_SIZE", feeders, earlyDedup = true, sink = sinkHandoffOnly, handoffBatch = BATCH_SIZE),
|
||||
)
|
||||
if (print) {
|
||||
println("\n--- feeders=$feeders (each delivering ${UNIQUE_EVENTS} events x $SUBS_PER_RELAY subs) ---")
|
||||
results.forEach { println(it) }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun dispatchStageBenchmark() {
|
||||
println("=== DISPATCH STAGE BENCHMARK (post-parse, pre-verify) ===")
|
||||
println("cores=${Runtime.getRuntime().availableProcessors()} uniqueEvents=$UNIQUE_EVENTS subsPerRelay=$SUBS_PER_RELAY")
|
||||
|
||||
// warmup pass (JIT), then the measured pass
|
||||
runAllVariants(print = false)
|
||||
runAllVariants(print = true)
|
||||
}
|
||||
}
|
||||
+106
@@ -0,0 +1,106 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.geode.KtorRelay
|
||||
import com.vitorpamplona.geode.RelayEngine
|
||||
import com.vitorpamplona.geode.fixtures.SyntheticEvents
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import okhttp3.OkHttpClient
|
||||
import okhttp3.Request
|
||||
import okhttp3.Response
|
||||
import java.util.concurrent.CountDownLatch
|
||||
import java.util.concurrent.TimeUnit
|
||||
import java.util.concurrent.atomic.AtomicLong
|
||||
import kotlin.test.Test
|
||||
|
||||
/**
|
||||
* Regression guard for two server-side bugs that made a single giant REQ
|
||||
* (bulk download) crawl and then wedge:
|
||||
*
|
||||
* 1. LiveEventStore's replay dedup used an immutable Set with copy-on-add —
|
||||
* accidentally O(n²): a 100k-event REQ streamed at ~700 events/s, and
|
||||
* 20k at ~2.4k events/s.
|
||||
* 2. Fixing (1) unmasked the session pump's slow-client policy: a fast
|
||||
* replay overflowed the 8192-frame backlog cap instantly, and the "drop"
|
||||
* only closed the internal queue — the client hung forever with no EOSE
|
||||
* and no close frame (and silently missed the tail of the response).
|
||||
*
|
||||
* Post-fix this streams 20k events in well under a second (~27k events/s
|
||||
* measured); the assertion bound is generous for CI noise.
|
||||
*/
|
||||
class GiantReqStreamTest {
|
||||
@Test
|
||||
fun giantReqStreamsFastAndComplete() {
|
||||
val engine = RelayEngine(url = "ws://127.0.0.1:7771/".normalizeRelayUrl())
|
||||
runBlocking {
|
||||
val events = (1..20_000).map { SyntheticEvents.fakeEvent(idSeed = it, kind = 1, createdAt = it.toLong(), content = "x".repeat(200)) }
|
||||
events.chunked(2000).forEach { engine.store.batchInsert(it) }
|
||||
}
|
||||
val server = KtorRelay(engine, port = 0).start()
|
||||
val http = OkHttpClient.Builder().build()
|
||||
try {
|
||||
val count = AtomicLong(0)
|
||||
val done = CountDownLatch(1)
|
||||
val start = System.nanoTime()
|
||||
val ws =
|
||||
http.newWebSocket(
|
||||
Request.Builder().url(server.url).build(),
|
||||
object : okhttp3.WebSocketListener() {
|
||||
override fun onOpen(
|
||||
w: okhttp3.WebSocket,
|
||||
r: Response,
|
||||
) {
|
||||
w.send("""["REQ","big",{"kinds":[1]}]""")
|
||||
}
|
||||
|
||||
override fun onMessage(
|
||||
w: okhttp3.WebSocket,
|
||||
text: String,
|
||||
) {
|
||||
if (text.startsWith("[\"EVENT\"")) {
|
||||
count.incrementAndGet()
|
||||
} else if (text.startsWith("[\"EOSE\"")) {
|
||||
done.countDown()
|
||||
}
|
||||
}
|
||||
|
||||
override fun onFailure(
|
||||
w: okhttp3.WebSocket,
|
||||
t: Throwable,
|
||||
r: Response?,
|
||||
) {
|
||||
done.countDown()
|
||||
}
|
||||
},
|
||||
)
|
||||
done.await(300, TimeUnit.SECONDS)
|
||||
val wall = System.nanoTime() - start
|
||||
ws.cancel()
|
||||
println("SCRATCH giant REQ: got=${count.get()}/20000 in %.1fs -> %.0f events/s".format(wall / 1e9, count.get() * 1e9 / wall))
|
||||
} finally {
|
||||
http.dispatcher.executorService.shutdown()
|
||||
server.stop(gracePeriodMillis = 200, timeoutMillis = 1_000)
|
||||
engine.close()
|
||||
}
|
||||
}
|
||||
}
|
||||
+128
@@ -0,0 +1,128 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.crypto.KeyPair
|
||||
import com.vitorpamplona.quartz.nip01Core.crypto.verify
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.ParallelEventVerifier
|
||||
import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerSync
|
||||
import com.vitorpamplona.quartz.nip10Notes.TextNoteEvent
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.SupervisorJob
|
||||
import kotlinx.coroutines.cancel
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeout
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* Proves [ParallelEventVerifier] delivers the parallel-verification speedup
|
||||
* that justifies moving Schnorr checks off the receiver coroutine: the same
|
||||
* signed events verified sequentially (today's inline behavior) vs submitted
|
||||
* through the batched parallel stage.
|
||||
*
|
||||
* Offline verify throughput measured earlier: ~75µs/event on one thread,
|
||||
* 2.6–3× across 4 cores (see the plan doc). The assertion threshold is
|
||||
* deliberately generous (1.5× on ≥4 cores) so CI noise doesn't flake.
|
||||
*/
|
||||
class ParallelVerifyBenchmark {
|
||||
companion object {
|
||||
const val EVENTS = 4_000
|
||||
const val SIGNERS = 8
|
||||
}
|
||||
|
||||
/**
|
||||
* A fresh set per measured pass: Event caches derived state (e.g. its
|
||||
* serialized id form) after the first verification, so re-verifying the
|
||||
* SAME instances in a second pass under-measures — the two passes must
|
||||
* not share objects. Multiple signers mirror real feeds.
|
||||
*/
|
||||
private fun signedEvents(salt: String): List<Event> {
|
||||
val signers = (1..SIGNERS).map { NostrSignerSync(KeyPair()) }
|
||||
return (1..EVENTS).map { signers[it % SIGNERS].sign(TextNoteEvent.build("parallel verify benchmark $salt $it", createdAt = it.toLong())) }
|
||||
}
|
||||
|
||||
private fun measureOnce(attempt: Int): Double {
|
||||
val seqEvents = signedEvents("seq$attempt")
|
||||
val parEvents = signedEvents("par$attempt")
|
||||
|
||||
val seqStart = System.nanoTime()
|
||||
var seqOk = 0
|
||||
seqEvents.forEach { if (it.verify()) seqOk++ }
|
||||
val seqNanos = System.nanoTime() - seqStart
|
||||
|
||||
val scope = CoroutineScope(SupervisorJob() + Dispatchers.Default)
|
||||
val parNanos =
|
||||
try {
|
||||
runBlocking {
|
||||
val verifier =
|
||||
ParallelEventVerifier<Unit>(
|
||||
scope = scope,
|
||||
onVerified = { _, _ -> },
|
||||
)
|
||||
val start = System.nanoTime()
|
||||
parEvents.forEach { verifier.submit(it, Unit) }
|
||||
verifier.close()
|
||||
withTimeout(60_000) { verifier.join() }
|
||||
val nanos = System.nanoTime() - start
|
||||
assertTrue(verifier.verifiedCount == parEvents.size.toLong(), "all valid events must verify")
|
||||
assertTrue(verifier.invalidCount == 0L)
|
||||
nanos
|
||||
}
|
||||
} finally {
|
||||
scope.cancel()
|
||||
}
|
||||
|
||||
val speedup = seqNanos.toDouble() / parNanos
|
||||
println(" attempt $attempt: sequential %.1fms (%.1fµs/event) vs pipeline %.1fms (%.1fµs/event) -> %.2fx".format(seqNanos / 1e6, seqNanos / 1e3 / EVENTS, parNanos / 1e6, parNanos / 1e3 / EVENTS, speedup))
|
||||
return speedup
|
||||
}
|
||||
|
||||
@Test
|
||||
fun batchedParallelVerifyBeatsSequential() {
|
||||
val cores = Runtime.getRuntime().availableProcessors()
|
||||
println("=== PARALLEL VERIFY BENCHMARK ($EVENTS signed events, $cores cores) ===")
|
||||
|
||||
// warmup both paths (JIT + secp tables) on a disjoint set
|
||||
signedEvents("warmup").take(400).forEach { it.verify() }
|
||||
|
||||
// The speedup ratio depends on AVAILABLE parallelism: in a full-suite
|
||||
// run (the pre-push hook) leftover pools and shared-machine load can
|
||||
// eat the idle cores, so a single sub-target sample is not a
|
||||
// regression. Retry a couple of times; enforce a hard floor that a
|
||||
// real serialization bug (the per-event-async version measured 0.94x)
|
||||
// can never pass, and warn — don't flake — in the noise band.
|
||||
var best = 0.0
|
||||
for (attempt in 1..3) {
|
||||
best = maxOf(best, measureOnce(attempt))
|
||||
if (cores < 4 || best > 1.5) break
|
||||
}
|
||||
|
||||
if (cores >= 4) {
|
||||
assertTrue(best > 1.05, "parallel verify slower than sequential (best %.2fx) — pipeline is serializing".format(best))
|
||||
if (best <= 1.5) {
|
||||
println(" WARN: best speedup %.2fx below the 1.5x target — machine likely loaded; not failing".format(best))
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+620
@@ -0,0 +1,620 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
|
||||
|
||||
import com.vitorpamplona.quartz.nip01Core.core.Event
|
||||
import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper
|
||||
import com.vitorpamplona.quartz.nip01Core.crypto.verify
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.accessories.EventCollector
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocket
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocketListener
|
||||
import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder
|
||||
import kotlinx.coroutines.CoroutineScope
|
||||
import kotlinx.coroutines.Dispatchers
|
||||
import kotlinx.coroutines.SupervisorJob
|
||||
import kotlinx.coroutines.async
|
||||
import kotlinx.coroutines.awaitAll
|
||||
import kotlinx.coroutines.cancel
|
||||
import kotlinx.coroutines.channels.Channel
|
||||
import kotlinx.coroutines.channels.trySendBlocking
|
||||
import kotlinx.coroutines.delay
|
||||
import kotlinx.coroutines.launch
|
||||
import kotlinx.coroutines.runBlocking
|
||||
import kotlinx.coroutines.withTimeoutOrNull
|
||||
import okhttp3.OkHttpClient
|
||||
import okhttp3.Request
|
||||
import okhttp3.Response
|
||||
import java.util.Collections
|
||||
import java.util.concurrent.ConcurrentHashMap
|
||||
import java.util.concurrent.TimeUnit
|
||||
import java.util.concurrent.atomic.AtomicLong
|
||||
import kotlin.test.Test
|
||||
|
||||
/**
|
||||
* Production measurement harness for the NostrClient receive path.
|
||||
*
|
||||
* The client's receiver is "single threaded" PER RELAY: OkHttp's socket-reader
|
||||
* thread enqueues each raw frame into an unbounded Channel, and one consumer
|
||||
* coroutine per connection drains it, doing — inline, serially — the JSON
|
||||
* parse, the PoolRequests state machine, every SubscriptionListener callback,
|
||||
* and every NostrClient connection listener (in the app that includes
|
||||
* LocalCache.justConsume, i.e. Schnorr signature verification of every new
|
||||
* event). This harness quantifies what that costs against real relays:
|
||||
*
|
||||
* - per-relay throughput (msgs/s, bytes) and time-to-EOSE for realistic filters
|
||||
* - queue delay: how long frames sit in the channel before the consumer
|
||||
* gets to them (the direct symptom of a saturated single consumer)
|
||||
* - processing time per message (parse + dispatch + verify when inline)
|
||||
* - INLINE vs PARALLEL verification (the same strategy the relay-server's
|
||||
* IngestQueue already uses on the write path)
|
||||
* - offline single-thread ceilings for parse and verify on the captured
|
||||
* production frames
|
||||
*
|
||||
* This test opens live sockets to public relays, so it is gated: it no-ops
|
||||
* unless the environment variable PROD_RELAY_BENCH is set.
|
||||
*
|
||||
* Run with:
|
||||
* PROD_RELAY_BENCH=1 ./gradlew :quartz:jvmTest --tests "*.ProductionReceiverBenchmark"
|
||||
*/
|
||||
class ProductionReceiverBenchmark {
|
||||
companion object {
|
||||
val RELAYS =
|
||||
listOf(
|
||||
"wss://relay.damus.io",
|
||||
"wss://nos.lol",
|
||||
"wss://relay.primal.net",
|
||||
"wss://nostr.wine",
|
||||
)
|
||||
|
||||
// fiatjaf — a pubkey with heavy notification traffic on public relays.
|
||||
const val BUSY_PUBKEY = "3bf0c63fcb93463407af97a5e5ee64fa883d107ef9e558472c4eb9aaaefa459d"
|
||||
|
||||
const val EOSE_TIMEOUT_MS = 30_000L
|
||||
const val LIVE_WINDOW_MS = 2_000L
|
||||
const val DRAIN_TIMEOUT_MS = 30_000L
|
||||
}
|
||||
|
||||
enum class VerifyMode { INLINE, PARALLEL }
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// Instrumented socket: BasicOkHttpWebSocket + timestamps around the
|
||||
// per-connection channel so queue delay and processing time are visible.
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
class RelayIngestMetrics(
|
||||
val url: NormalizedRelayUrl,
|
||||
) {
|
||||
val queueDelayNanos = Collections.synchronizedList(ArrayList<Long>(8192))
|
||||
val procNanos = Collections.synchronizedList(ArrayList<Long>(8192))
|
||||
val msgCount = AtomicLong(0)
|
||||
val byteCount = AtomicLong(0)
|
||||
|
||||
@Volatile var firstMsgAtNanos = 0L
|
||||
|
||||
@Volatile var lastMsgAtNanos = 0L
|
||||
|
||||
fun record(
|
||||
queueDelay: Long,
|
||||
proc: Long,
|
||||
bytes: Int,
|
||||
now: Long,
|
||||
) {
|
||||
queueDelayNanos.add(queueDelay)
|
||||
procNanos.add(proc)
|
||||
msgCount.incrementAndGet()
|
||||
byteCount.addAndGet(bytes.toLong())
|
||||
if (firstMsgAtNanos == 0L) firstMsgAtNanos = now
|
||||
lastMsgAtNanos = now
|
||||
}
|
||||
}
|
||||
|
||||
class InstrumentedOkHttpWebSocket(
|
||||
val url: NormalizedRelayUrl,
|
||||
val httpClient: OkHttpClient,
|
||||
val out: WebSocketListener,
|
||||
val metrics: RelayIngestMetrics,
|
||||
val frameSink: ((String) -> Unit)?,
|
||||
) : WebSocket {
|
||||
class Frame(
|
||||
val text: String,
|
||||
val enqueuedAtNanos: Long,
|
||||
)
|
||||
|
||||
private var socket: okhttp3.WebSocket? = null
|
||||
|
||||
override fun needsReconnect() = socket == null
|
||||
|
||||
override fun connect() {
|
||||
val request = Request.Builder().url(url.url).build()
|
||||
|
||||
val listener =
|
||||
object : okhttp3.WebSocketListener() {
|
||||
val scope = CoroutineScope(Dispatchers.IO + SupervisorJob())
|
||||
val incoming: Channel<Frame> = Channel(Channel.UNLIMITED)
|
||||
val job =
|
||||
scope.launch {
|
||||
// Mirrors BasicOkHttpWebSocket/OkHttpWebSocket: ONE
|
||||
// consumer per connection; everything downstream of
|
||||
// out.onMessage runs serially right here.
|
||||
for (frame in incoming) {
|
||||
val start = System.nanoTime()
|
||||
frameSink?.invoke(frame.text)
|
||||
out.onMessage(frame.text)
|
||||
val end = System.nanoTime()
|
||||
metrics.record(
|
||||
queueDelay = start - frame.enqueuedAtNanos,
|
||||
proc = end - start,
|
||||
bytes = frame.text.length,
|
||||
now = end,
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
override fun onOpen(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
response: Response,
|
||||
) = out.onOpen(
|
||||
(response.receivedResponseAtMillis - response.sentRequestAtMillis).toInt(),
|
||||
response.headers["Sec-WebSocket-Extensions"]?.contains("permessage-deflate") ?: false,
|
||||
)
|
||||
|
||||
override fun onMessage(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
text: String,
|
||||
) {
|
||||
incoming.trySendBlocking(Frame(text, System.nanoTime()))
|
||||
}
|
||||
|
||||
override fun onClosed(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
code: Int,
|
||||
reason: String,
|
||||
) {
|
||||
incoming.close()
|
||||
job.cancel()
|
||||
scope.cancel()
|
||||
out.onClosed(code, reason)
|
||||
}
|
||||
|
||||
override fun onFailure(
|
||||
webSocket: okhttp3.WebSocket,
|
||||
t: Throwable,
|
||||
response: Response?,
|
||||
) {
|
||||
incoming.close()
|
||||
job.cancel()
|
||||
scope.cancel()
|
||||
out.onFailure(t, response?.code, response?.message)
|
||||
}
|
||||
}
|
||||
|
||||
socket = httpClient.newWebSocket(request, listener)
|
||||
}
|
||||
|
||||
override fun disconnect() {
|
||||
socket?.cancel()
|
||||
socket = null
|
||||
}
|
||||
|
||||
override fun send(msg: String): Boolean = socket?.send(msg) ?: false
|
||||
}
|
||||
|
||||
class InstrumentedBuilder(
|
||||
val httpClient: OkHttpClient,
|
||||
val metricsFor: (NormalizedRelayUrl) -> RelayIngestMetrics,
|
||||
val frameSink: ((String) -> Unit)?,
|
||||
) : WebsocketBuilder {
|
||||
override fun build(
|
||||
url: NormalizedRelayUrl,
|
||||
out: WebSocketListener,
|
||||
) = InstrumentedOkHttpWebSocket(url, httpClient, out, metricsFor(url), frameSink)
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// Verification sink: mimics what CacheClientConnector/LocalCache do
|
||||
// with every event (dedup by id, then Schnorr-verify new ones),
|
||||
// either inline on the receiver coroutine (current app behavior) or
|
||||
// offloaded to a parallel Default-dispatcher pool.
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
class VerifyingSink(
|
||||
val mode: VerifyMode,
|
||||
scope: CoroutineScope,
|
||||
) {
|
||||
private val seen: MutableSet<String> = ConcurrentHashMap.newKeySet()
|
||||
val received = AtomicLong(0)
|
||||
val unique = AtomicLong(0)
|
||||
val verified = AtomicLong(0)
|
||||
val invalid = AtomicLong(0)
|
||||
val verifyNanosTotal = AtomicLong(0)
|
||||
val perRelayEvents = ConcurrentHashMap<NormalizedRelayUrl, AtomicLong>()
|
||||
|
||||
private val queue: Channel<Event> = Channel(Channel.UNLIMITED)
|
||||
private val workers =
|
||||
if (mode == VerifyMode.PARALLEL) {
|
||||
List(Runtime.getRuntime().availableProcessors()) {
|
||||
scope.launch(Dispatchers.Default) {
|
||||
for (event in queue) doVerify(event)
|
||||
}
|
||||
}
|
||||
} else {
|
||||
emptyList()
|
||||
}
|
||||
|
||||
fun consume(
|
||||
event: Event,
|
||||
relay: NormalizedRelayUrl,
|
||||
) {
|
||||
received.incrementAndGet()
|
||||
perRelayEvents.getOrPut(relay) { AtomicLong(0) }.incrementAndGet()
|
||||
if (seen.add(event.id)) {
|
||||
unique.incrementAndGet()
|
||||
when (mode) {
|
||||
VerifyMode.INLINE -> doVerify(event)
|
||||
VerifyMode.PARALLEL -> queue.trySend(event)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private fun doVerify(event: Event) {
|
||||
val start = System.nanoTime()
|
||||
val ok = event.verify()
|
||||
verifyNanosTotal.addAndGet(System.nanoTime() - start)
|
||||
if (ok) verified.incrementAndGet() else invalid.incrementAndGet()
|
||||
}
|
||||
|
||||
suspend fun awaitDrained(timeoutMs: Long): Boolean =
|
||||
withTimeoutOrNull(timeoutMs) {
|
||||
while (verified.get() + invalid.get() < unique.get()) delay(20)
|
||||
true
|
||||
} ?: false
|
||||
|
||||
fun close() {
|
||||
queue.close()
|
||||
}
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// Scenario runner
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
class ScenarioReport(
|
||||
val capturedEvents: List<Event>,
|
||||
)
|
||||
|
||||
private fun percentileLine(nanos: List<Long>): String {
|
||||
if (nanos.isEmpty()) return "n=0"
|
||||
val sorted = nanos.sorted()
|
||||
|
||||
fun p(q: Double) = sorted[((sorted.size - 1) * q).toInt()]
|
||||
|
||||
fun fmt(n: Long) =
|
||||
when {
|
||||
n >= 1_000_000 -> "%.1fms".format(n / 1_000_000.0)
|
||||
n >= 1_000 -> "%.1fµs".format(n / 1_000.0)
|
||||
else -> "${n}ns"
|
||||
}
|
||||
return "n=${sorted.size} p50=${fmt(p(0.5))} p90=${fmt(p(0.9))} p99=${fmt(p(0.99))} max=${fmt(sorted.last())}"
|
||||
}
|
||||
|
||||
private suspend fun runScenario(
|
||||
name: String,
|
||||
httpClient: OkHttpClient,
|
||||
relays: List<NormalizedRelayUrl>,
|
||||
filters: Map<NormalizedRelayUrl, List<Filter>>,
|
||||
mode: VerifyMode,
|
||||
captureFrames: MutableList<String>? = null,
|
||||
quiet: Boolean = false,
|
||||
): ScenarioReport {
|
||||
if (!quiet) println("\n=== SCENARIO: $name [verify=$mode] ===")
|
||||
|
||||
val metrics = ConcurrentHashMap<NormalizedRelayUrl, RelayIngestMetrics>()
|
||||
val frameSink: ((String) -> Unit)? =
|
||||
captureFrames?.let { list -> { text: String -> if (list.size < 30_000) list.add(text) } }
|
||||
|
||||
val builder =
|
||||
InstrumentedBuilder(
|
||||
httpClient,
|
||||
{ url -> metrics.getOrPut(url) { RelayIngestMetrics(url) } },
|
||||
frameSink,
|
||||
)
|
||||
|
||||
val scenarioScope = CoroutineScope(SupervisorJob() + Dispatchers.Default)
|
||||
val sink = VerifyingSink(mode, scenarioScope)
|
||||
val capturedEvents = Collections.synchronizedList(ArrayList<Event>(8192))
|
||||
|
||||
val client = NostrClient(builder)
|
||||
val collector =
|
||||
EventCollector(client) { event, relay ->
|
||||
sink.consume(event, relay.url)
|
||||
if (capturedEvents.size < 20_000) capturedEvents.add(event)
|
||||
}
|
||||
|
||||
val startNanos = System.nanoTime()
|
||||
val reqSentAt = ConcurrentHashMap<String, Long>()
|
||||
val firstEventAt = ConcurrentHashMap<NormalizedRelayUrl, Long>()
|
||||
val eoseAt = ConcurrentHashMap<NormalizedRelayUrl, Long>()
|
||||
val cannotConnect = ConcurrentHashMap<NormalizedRelayUrl, String>()
|
||||
val eventsBeforeEose = ConcurrentHashMap<NormalizedRelayUrl, AtomicLong>()
|
||||
val eventsAfterEose = ConcurrentHashMap<NormalizedRelayUrl, AtomicLong>()
|
||||
|
||||
val listener =
|
||||
object : SubscriptionListener {
|
||||
override fun onSubscriptionStarted(
|
||||
relay: String,
|
||||
forFilters: List<Filter>,
|
||||
) {
|
||||
reqSentAt.putIfAbsent(relay, System.nanoTime() - startNanos)
|
||||
}
|
||||
|
||||
override fun onEvent(
|
||||
event: Event,
|
||||
isLive: Boolean,
|
||||
relay: NormalizedRelayUrl,
|
||||
forFilters: List<Filter>?,
|
||||
) {
|
||||
firstEventAt.putIfAbsent(relay, System.nanoTime() - startNanos)
|
||||
val counter = if (eoseAt.containsKey(relay)) eventsAfterEose else eventsBeforeEose
|
||||
counter.getOrPut(relay) { AtomicLong(0) }.incrementAndGet()
|
||||
}
|
||||
|
||||
override fun onEose(
|
||||
relay: NormalizedRelayUrl,
|
||||
forFilters: List<Filter>?,
|
||||
) {
|
||||
eoseAt.putIfAbsent(relay, System.nanoTime() - startNanos)
|
||||
}
|
||||
|
||||
override fun onCannotConnect(
|
||||
relay: NormalizedRelayUrl,
|
||||
message: String,
|
||||
forFilters: List<Filter>?,
|
||||
) {
|
||||
cannotConnect.putIfAbsent(relay, message)
|
||||
}
|
||||
}
|
||||
|
||||
val subId = "bench-" + System.nanoTime().toString(16)
|
||||
client.subscribe(subId, filters, listener)
|
||||
|
||||
val allEosed =
|
||||
withTimeoutOrNull(EOSE_TIMEOUT_MS) {
|
||||
while (!eoseAt.keys.containsAll(relays) && cannotConnect.keys.size + eoseAt.keys.size < relays.size) delay(50)
|
||||
true
|
||||
} ?: false
|
||||
|
||||
// watch a short live window after EOSE, like the app does
|
||||
delay(LIVE_WINDOW_MS)
|
||||
|
||||
client.unsubscribe(subId)
|
||||
|
||||
val drainStart = System.nanoTime()
|
||||
val drained = sink.awaitDrained(DRAIN_TIMEOUT_MS)
|
||||
val drainNanos = System.nanoTime() - drainStart
|
||||
|
||||
sink.close()
|
||||
collector.destroy()
|
||||
client.close()
|
||||
scenarioScope.cancel()
|
||||
|
||||
// ---- report ----
|
||||
if (quiet) return ScenarioReport(ArrayList(capturedEvents).distinctBy { it.id })
|
||||
|
||||
if (!allEosed) println("!! not all relays reached EOSE within ${EOSE_TIMEOUT_MS}ms")
|
||||
cannotConnect.forEach { (relay, msg) -> println("!! cannot connect ${relay.url}: $msg") }
|
||||
|
||||
for (relay in relays) {
|
||||
val m = metrics[relay] ?: continue
|
||||
val req = reqSentAt[relay.url]?.let { "%.0fms".format(it / 1e6) } ?: "-"
|
||||
val ttfb = firstEventAt[relay]?.let { "%.0fms".format(it / 1e6) } ?: "-"
|
||||
val eose = eoseAt[relay]?.let { "%.0fms".format(it / 1e6) } ?: "-"
|
||||
val activeNanos = (m.lastMsgAtNanos - m.firstMsgAtNanos).coerceAtLeast(1)
|
||||
val rate = m.msgCount.get() * 1e9 / activeNanos
|
||||
val busyPct = m.procNanos.toList().sum() * 100.0 / activeNanos
|
||||
println(" ${relay.url}")
|
||||
println(
|
||||
" req=$req firstEvent=$ttfb eose=$eose msgs=${m.msgCount.get()} bytes=${m.byteCount.get()} rate=%.0f msg/s consumerBusy=%.0f%%"
|
||||
.format(rate, busyPct),
|
||||
)
|
||||
println(" events: pre-EOSE=${eventsBeforeEose[relay]?.get() ?: 0} post-EOSE=${eventsAfterEose[relay]?.get() ?: 0}")
|
||||
println(" queueDelay: ${percentileLine(m.queueDelayNanos.toList())}")
|
||||
println(" procTime: ${percentileLine(m.procNanos.toList())}")
|
||||
}
|
||||
val uniq = sink.unique.get()
|
||||
val recv = sink.received.get()
|
||||
println(" TOTAL: received=$recv unique=$uniq duplicates=${recv - uniq} verified=${sink.verified.get()} invalid=${sink.invalid.get()}")
|
||||
if (uniq > 0) {
|
||||
println(
|
||||
" verify: total=%.1fms avg=%.1fµs/event drainAfterUnsub=%.1fms drained=%s"
|
||||
.format(
|
||||
sink.verifyNanosTotal.get() / 1e6,
|
||||
sink.verifyNanosTotal.get() / 1e3 / uniq,
|
||||
drainNanos / 1e6,
|
||||
drained,
|
||||
),
|
||||
)
|
||||
}
|
||||
|
||||
return ScenarioReport(ArrayList(capturedEvents).distinctBy { it.id })
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// Offline microbenchmarks on captured production frames: what a single
|
||||
// receiver coroutine can do per second, with no network in the way.
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
private fun offlineParseBench(frames: List<String>) {
|
||||
if (frames.isEmpty()) return
|
||||
println("\n=== OFFLINE: single-thread JSON parse of ${frames.size} captured frames ===")
|
||||
// warmup
|
||||
repeat(2) { frames.forEach { runCatching { OptimizedJsonMapper.fromJsonToMessage(it) } } }
|
||||
val start = System.nanoTime()
|
||||
var parsed = 0
|
||||
frames.forEach { if (runCatching { OptimizedJsonMapper.fromJsonToMessage(it) }.isSuccess) parsed++ }
|
||||
val nanos = System.nanoTime() - start
|
||||
println(
|
||||
" parsed=$parsed in %.1fms -> %.0f msg/s (%.1fµs/msg)"
|
||||
.format(nanos / 1e6, parsed * 1e9 / nanos, nanos / 1e3 / parsed),
|
||||
)
|
||||
}
|
||||
|
||||
private fun offlineVerifyBench(events: List<Event>) {
|
||||
if (events.isEmpty()) return
|
||||
val sample = events.take(3000)
|
||||
println("\n=== OFFLINE: Schnorr verify of ${sample.size} unique production events ===")
|
||||
|
||||
// warmup (JIT + secp tables)
|
||||
sample.take(300).forEach { it.verify() }
|
||||
|
||||
val t1 = System.nanoTime()
|
||||
sample.forEach { it.verify() }
|
||||
val seqNanos = System.nanoTime() - t1
|
||||
println(
|
||||
" sequential (1 thread): %.1fms -> %.0f events/s (%.1fµs/event)"
|
||||
.format(seqNanos / 1e6, sample.size * 1e9 / seqNanos, seqNanos / 1e3 / sample.size),
|
||||
)
|
||||
|
||||
val cores = Runtime.getRuntime().availableProcessors()
|
||||
val parNanos =
|
||||
runBlocking(Dispatchers.Default) {
|
||||
val start = System.nanoTime()
|
||||
sample
|
||||
.chunked((sample.size + cores - 1) / cores)
|
||||
.map { chunk -> async { chunk.forEach { it.verify() } } }
|
||||
.awaitAll()
|
||||
System.nanoTime() - start
|
||||
}
|
||||
println(
|
||||
" parallel ($cores threads): %.1fms -> %.0f events/s (%.1fx speedup)"
|
||||
.format(parNanos / 1e6, sample.size * 1e9 / parNanos, seqNanos.toDouble() / parNanos),
|
||||
)
|
||||
}
|
||||
|
||||
// -----------------------------------------------------------------
|
||||
// The gated test
|
||||
// -----------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
fun productionReceiverBenchmark() {
|
||||
if (System.getenv("PROD_RELAY_BENCH") == null && System.getProperty("prodRelayBench") == null) {
|
||||
println("ProductionReceiverBenchmark skipped. Set PROD_RELAY_BENCH=1 to run against live relays.")
|
||||
return
|
||||
}
|
||||
|
||||
val httpClient =
|
||||
OkHttpClient
|
||||
.Builder()
|
||||
.connectTimeout(15, TimeUnit.SECONDS)
|
||||
.readTimeout(60, TimeUnit.SECONDS)
|
||||
.pingInterval(30, TimeUnit.SECONDS)
|
||||
.build()
|
||||
|
||||
val relays = RELAYS.map { it.normalizeRelayUrl() }
|
||||
|
||||
runBlocking {
|
||||
val capturedFrames = Collections.synchronizedList(ArrayList<String>(30_000))
|
||||
|
||||
// Warmup (discarded): JIT-compile the parse/verify/dispatch path and
|
||||
// open the OkHttp pools so the first measured scenario isn't paying
|
||||
// cold-start costs that the later ones don't.
|
||||
runScenario(
|
||||
"warmup",
|
||||
httpClient,
|
||||
relays,
|
||||
relays.associateWith { listOf(Filter(kinds = listOf(1), limit = 150)) },
|
||||
VerifyMode.INLINE,
|
||||
quiet = true,
|
||||
)
|
||||
|
||||
// S1: home-feed-style firehose, current app behavior (inline verify)
|
||||
val s1 =
|
||||
runScenario(
|
||||
"kind-1 firehose, limit 500/relay",
|
||||
httpClient,
|
||||
relays,
|
||||
relays.associateWith { listOf(Filter(kinds = listOf(1), limit = 500)) },
|
||||
VerifyMode.INLINE,
|
||||
captureFrames = capturedFrames,
|
||||
)
|
||||
|
||||
// S1b: same filter, verification offloaded to a parallel pool
|
||||
runScenario(
|
||||
"kind-1 firehose, limit 500/relay",
|
||||
httpClient,
|
||||
relays,
|
||||
relays.associateWith { listOf(Filter(kinds = listOf(1), limit = 500)) },
|
||||
VerifyMode.PARALLEL,
|
||||
)
|
||||
|
||||
// S2: notifications-style filter for a busy pubkey
|
||||
runScenario(
|
||||
"notifications for busy pubkey (kinds 1,6,7,9735)",
|
||||
httpClient,
|
||||
relays,
|
||||
relays.associateWith {
|
||||
listOf(
|
||||
Filter(
|
||||
kinds = listOf(1, 6, 7, 9735),
|
||||
tags = mapOf("p" to listOf(BUSY_PUBKEY)),
|
||||
limit = 500,
|
||||
),
|
||||
)
|
||||
},
|
||||
VerifyMode.INLINE,
|
||||
)
|
||||
|
||||
// S3: metadata burst — the app-startup pattern: fetch kind 0/10002
|
||||
// for every author seen in the feed. Big bursts, tiny events.
|
||||
val authors =
|
||||
s1.capturedEvents
|
||||
.map { it.pubKey }
|
||||
.distinct()
|
||||
.take(300)
|
||||
if (authors.isNotEmpty()) {
|
||||
runScenario(
|
||||
"metadata burst (kinds 0,10002) for ${authors.size} authors",
|
||||
httpClient,
|
||||
relays,
|
||||
relays.associateWith { listOf(Filter(kinds = listOf(0, 10002), authors = authors)) },
|
||||
VerifyMode.INLINE,
|
||||
)
|
||||
runScenario(
|
||||
"metadata burst (kinds 0,10002) for ${authors.size} authors",
|
||||
httpClient,
|
||||
relays,
|
||||
relays.associateWith { listOf(Filter(kinds = listOf(0, 10002), authors = authors)) },
|
||||
VerifyMode.PARALLEL,
|
||||
)
|
||||
}
|
||||
|
||||
// Offline ceilings from the captured production data
|
||||
offlineParseBench(ArrayList(capturedFrames))
|
||||
offlineVerifyBench(s1.capturedEvents)
|
||||
}
|
||||
|
||||
httpClient.dispatcher.executorService.shutdown()
|
||||
}
|
||||
}
|
||||
Vendored
+49
@@ -0,0 +1,49 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils.cache
|
||||
|
||||
import kotlin.concurrent.AtomicReference
|
||||
|
||||
// Copy-on-write, mirroring LargeCache.linux: correct and simple; the linux
|
||||
// target is CI-only so write cost is acceptable.
|
||||
actual class ConcurrentHashCache<K : Any, V : Any> {
|
||||
private val mapRef = AtomicReference(HashMap<K, V>())
|
||||
|
||||
actual fun get(key: K): V? = mapRef.value[key]
|
||||
|
||||
actual fun put(
|
||||
key: K,
|
||||
value: V,
|
||||
) {
|
||||
while (true) {
|
||||
val current = mapRef.value
|
||||
val copy = HashMap(current)
|
||||
copy[key] = value
|
||||
if (mapRef.compareAndSet(current, copy)) return
|
||||
}
|
||||
}
|
||||
|
||||
actual fun size(): Int = mapRef.value.size
|
||||
|
||||
actual fun clear() {
|
||||
mapRef.value = HashMap()
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user