From 32b15d55d03eecc9ccd1f65d679f8f738b1cc739 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 21:22:31 +0000 Subject: [PATCH 01/21] =?UTF-8?q?feat(quartz):=20NostrServer.ingest=20?= =?UTF-8?q?=E2=80=94=20local=20write=20path=20with=20per-submission=20veri?= =?UTF-8?q?fy=20skip?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds a public local-ingest entry to NostrServer for events that don't arrive over a client connection (mirror/sync workers, import jobs). It routes through the same group-commit IngestQueue and live fanout as a client publish, and each Submission can opt out of the parallel Schnorr-verify hook — the relay-to-relay trust model, for events streamed from an upstream that already verified them (verify profiles at ~8% of busy ingest CPU). Library defaults are unchanged: skipVerify defaults to false everywhere and nothing in the client-publish path can set it. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../nip01Core/relay/server/NostrServer.kt | 24 ++- .../relay/server/backend/IngestQueue.kt | 29 +++- .../relay/server/backend/LiveEventStore.kt | 17 +- .../relay/server/NostrServerIngestTest.kt | 162 ++++++++++++++++++ 4 files changed, 224 insertions(+), 8 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt index 1ec4386619..81326a5bbf 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt @@ -20,6 +20,7 @@ */ package com.vitorpamplona.quartz.nip01Core.relay.server +import com.vitorpamplona.quartz.nip01Core.core.Event import com.vitorpamplona.quartz.nip01Core.crypto.verify import com.vitorpamplona.quartz.nip01Core.relay.server.backend.IngestQueue import com.vitorpamplona.quartz.nip01Core.relay.server.backend.LiveEventStore @@ -95,7 +96,28 @@ class NostrServer( }, ) - override val backend: SessionBackend = LiveEventStore(store, ingest) + private val liveStore = LiveEventStore(store, ingest) + + override val backend: SessionBackend = liveStore + + /** + * Local ingestion path for events that did not arrive over a client + * connection — e.g. a mirror worker streaming a trusted upstream + * relay, or an import job. Routes through the same group-commit + * [IngestQueue] and live fanout as a client EVENT publish, but skips + * the per-connection policy chain (there is no connection). + * + * [skipVerify] exempts the event from the parallel signature-verify + * hook — the relay-to-relay trust model: pass `true` only for events + * from an explicitly configured upstream that already verified them + * (Schnorr verify profiles at ~8% of busy ingest CPU). The default + * `false` keeps verify-everything semantics. + */ + suspend fun ingest( + event: Event, + skipVerify: Boolean = false, + onComplete: (IEventStore.InsertOutcome) -> Unit, + ) = liveStore.submit(event, skipVerify, onComplete) init { // Deferred-FTS catch-up worker: tokenizes in the gaps between diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/IngestQueue.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/IngestQueue.kt index 4e32fba5ad..39f1191882 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/IngestQueue.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/IngestQueue.kt @@ -27,7 +27,6 @@ import kotlinx.coroutines.CoroutineScope import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.SupervisorJob import kotlinx.coroutines.async -import kotlinx.coroutines.awaitAll import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.channels.ClosedReceiveChannelException @@ -115,9 +114,15 @@ class IngestQueue( /** * One outstanding ingest request: the event to insert plus the * callback the writer fires once the row's outcome is known. + * [skipVerify] exempts this row from the [verify] hook — the + * relay-to-relay trust model: set by local ingestion paths for + * events streamed from an explicitly configured upstream relay + * that already verified them (see + * [com.vitorpamplona.quartz.nip01Core.relay.server.NostrServer.ingest]). */ class Submission( val event: Event, + val skipVerify: Boolean, val onComplete: (IEventStore.InsertOutcome) -> Unit, ) @@ -159,11 +164,12 @@ class IngestQueue( */ suspend fun submit( event: Event, + skipVerify: Boolean = false, onComplete: (IEventStore.InsertOutcome) -> Unit, ) { ensureWriterStarted() pending.addAndFetch(1) - incoming.send(Submission(event, onComplete)) + incoming.send(Submission(event, skipVerify, onComplete)) } private fun ensureWriterStarted() { @@ -234,15 +240,26 @@ class IngestQueue( * configured (skip the stage entirely). For multi-event batches * each verify runs as its own `async(Default)` so they spread * across CPU cores; single-event batches short-circuit to a - * direct call to avoid coroutine-scope overhead. + * direct call to avoid coroutine-scope overhead. Rows flagged + * [Submission.skipVerify] (trusted publishers) pass without + * invoking the hook. */ private suspend fun verifyBatch(batch: List): BooleanArray? { val hook = verify ?: return null - if (batch.size == 1) return BooleanArray(1) { hook(batch[0].event) } + if (batch.size == 1) { + val sub = batch[0] + return BooleanArray(1) { sub.skipVerify || hook(sub.event) } + } + if (batch.all { it.skipVerify }) return BooleanArray(batch.size) { true } return coroutineScope { batch - .map { sub -> async(Dispatchers.Default) { hook(sub.event) } } - .awaitAll() + .map { sub -> + if (sub.skipVerify) { + null + } else { + async(Dispatchers.Default) { hook(sub.event) } + } + }.map { it?.await() ?: true } .toBooleanArray() } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt index 89f1fe98fe..6bed5eae8d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt @@ -91,8 +91,23 @@ class LiveEventStore( override suspend fun submit( event: Event, onComplete: (IEventStore.InsertOutcome) -> Unit, + ) = submit(event, skipVerify = false, onComplete = onComplete) + + /** + * [submit] variant for locally-originated traffic (mirror/sync + * workers rather than client connections). [skipVerify] exempts + * this event from the [IngestQueue]'s signature-verification hook — + * the relay-to-relay trust model: set it only for events streamed + * from a configured upstream relay that already verified them. + * Accepted events fan out to live subscribers exactly like a + * client publish. + */ + suspend fun submit( + event: Event, + skipVerify: Boolean, + onComplete: (IEventStore.InsertOutcome) -> Unit, ) { - ingest.submit(event) { outcome -> + ingest.submit(event, skipVerify) { outcome -> if (outcome is IEventStore.InsertOutcome.Accepted) { writeGeneration.addAndFetch(1L) fanout(event) diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt new file mode 100644 index 0000000000..085edb3098 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt @@ -0,0 +1,162 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.server + +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.crypto.KeyPair +import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy +import com.vitorpamplona.quartz.nip01Core.store.IEventStore +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.DeterministicSigner +import kotlinx.coroutines.CompletableDeferred +import kotlinx.coroutines.CoroutineDispatcher +import kotlinx.coroutines.ExperimentalCoroutinesApi +import kotlinx.coroutines.test.UnconfinedTestDispatcher +import kotlinx.coroutines.test.runTest +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * The local-ingest path ([NostrServer.ingest]) and its relay-to-relay + * trust switch: `skipVerify = true` (a mirror streaming from a trusted + * upstream) must land forged-signature events, while the default keeps + * verify-everything semantics through the IngestQueue's parallel hook. + */ +@OptIn(ExperimentalCoroutinesApi::class) +class NostrServerIngestTest { + private val signer = DeterministicSigner(KeyPair()) + + private fun signedEvent(content: String = "hello"): Event = signer.sign(createdAt = 1000L, kind = 1, tags = emptyArray(), content = content) + + /** A structurally valid event whose signature verifies against nothing. */ + private fun forgedEvent(id: Int = 1): Event = + Event( + id = id.toString().padStart(64, '0'), + pubKey = signer.pubKey, + createdAt = 1000L + id, + kind = 1, + tags = emptyArray(), + content = "forged", + sig = "f".repeat(128), + ) + + private fun createServer(dispatcher: CoroutineDispatcher): NostrServer = + NostrServer( + store = EventStore(null), + policyBuilder = { EmptyPolicy }, + parentContext = dispatcher, + parallelVerify = true, + ) + + private suspend fun NostrServer.ingestOutcome( + event: Event, + skipVerify: Boolean, + ): IEventStore.InsertOutcome { + val outcome = CompletableDeferred() + ingest(event, skipVerify) { outcome.complete(it) } + return outcome.await() + } + + @Test + fun forgedEventIsRejectedByDefault() = + runTest { + val server = createServer(UnconfinedTestDispatcher(testScheduler)) + + val outcome = server.ingestOutcome(forgedEvent(), skipVerify = false) + + assertTrue(outcome is IEventStore.InsertOutcome.Rejected) + assertTrue(outcome.reason.contains("signature")) + + server.close() + } + + @Test + fun forgedEventLandsWhenTrusted() = + runTest { + val server = createServer(UnconfinedTestDispatcher(testScheduler)) + + val outcome = server.ingestOutcome(forgedEvent(), skipVerify = true) + + assertEquals(IEventStore.InsertOutcome.Accepted, outcome) + + server.close() + } + + @Test + fun validEventLandsEitherWay() = + runTest { + val server = createServer(UnconfinedTestDispatcher(testScheduler)) + + assertEquals( + IEventStore.InsertOutcome.Accepted, + server.ingestOutcome(signedEvent("via verify"), skipVerify = false), + ) + assertEquals( + IEventStore.InsertOutcome.Accepted, + server.ingestOutcome(signedEvent("via trust"), skipVerify = true), + ) + + server.close() + } + + @Test + fun trustIsPerSubmissionNotPerQueue() = + runTest { + // A trusted mirror and an untrusted publisher share the same + // IngestQueue; the skip must apply row-by-row, never leak from + // one submission to the next. + val server = createServer(UnconfinedTestDispatcher(testScheduler)) + + assertEquals( + IEventStore.InsertOutcome.Accepted, + server.ingestOutcome(forgedEvent(1), skipVerify = true), + ) + val untrusted = server.ingestOutcome(forgedEvent(2), skipVerify = false) + assertTrue(untrusted is IEventStore.InsertOutcome.Rejected) + + server.close() + } + + @Test + fun trustedIngestFansOutToLiveSubscribers() = + runTest { + // Mirror traffic must feed live REQs exactly like a client + // publish — the skip changes verification, not delivery. + val dispatcher = UnconfinedTestDispatcher(testScheduler) + val server = createServer(dispatcher) + + val received = mutableListOf() + val session = server.connect { received.add(it) } + session.receive("""["REQ","live",{"kinds":[1]}]""") + assertTrue(received.any { it.startsWith("[\"EOSE\"") }) + + val forged = forgedEvent() + assertEquals( + IEventStore.InsertOutcome.Accepted, + server.ingestOutcome(forged, skipVerify = true), + ) + + assertTrue(received.any { it.startsWith("[\"EVENT\",\"live\"") && it.contains(forged.id) }) + + server.close() + } +} From 50e259b495c5ac12bf9a74be26cb496ce9708036 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 21:22:46 +0000 Subject: [PATCH 02/21] feat(geode): [[mirror]] upstream streaming with relay-to-relay trust MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Backlog item 1 of the relay performance campaign: skip signature verification for events ingested from explicitly configured trusted upstream relays, strfry-router style. geode now dials each [[mirror]] url from the config, subscribes to everything newer than now - backfill_seconds, and feeds the stream through NostrServer.ingest (same group-commit writer + live fanout as client publishes). The NostrClient underneath owns reconnects, backoff and REQ re-sync; duplicate replays after a reconnect are rejected by the store's unique-id constraint and only surface as counters. trusted = true is the per-upstream trust switch: events from that connection skip Schnorr verify. The trusted identity is the URL this relay dialed (TLS-authenticated for wss://), never anything an inbound peer claims. Default is false — mirror-but-verify — and a relay with no [[mirror]] entries behaves exactly as before. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- geode/build.gradle.kts | 5 + geode/config.example.toml | 18 ++ .../kotlin/com/vitorpamplona/geode/Main.kt | 37 +++- .../geode/config/StaticConfig.kt | 23 +++ .../geode/mirror/MirrorWorker.kt | 190 ++++++++++++++++++ .../geode/config/StaticConfigTest.kt | 30 +++ .../geode/mirror/MirrorWorkerTest.kt | 165 +++++++++++++++ 7 files changed, 466 insertions(+), 2 deletions(-) create mode 100644 geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt create mode 100644 geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt diff --git a/geode/build.gradle.kts b/geode/build.gradle.kts index bc41967f3d..981d2a0abc 100644 --- a/geode/build.gradle.kts +++ b/geode/build.gradle.kts @@ -85,6 +85,11 @@ dependencies { implementation(libs.jackson.module.kotlin) implementation(libs.kotlinx.serialization.json) + // Outbound WebSockets for the [[mirror]] upstream streams (quartz's + // BasicOkHttpWebSocket transport). Same OkHttp the rest of the repo + // already ships (Apache-2.0). + implementation(libs.okhttp) + // Bundled SQLite driver — Relay's default in-memory EventStore creates // an in-memory DB at runtime. implementation(libs.androidx.sqlite.bundled.jvm) diff --git a/geode/config.example.toml b/geode/config.example.toml index 73f7adca1c..98ad0fc2d4 100644 --- a/geode/config.example.toml +++ b/geode/config.example.toml @@ -86,6 +86,24 @@ require_auth = false # kind_whitelist = [0, 1, 3, 7, 1059, 30023] # kind_blacklist = [4] +# Mirror upstream relays (strfry-router style, "down" direction): the +# relay dials each [[mirror]] url, subscribes to everything newer than +# now - backfill_seconds, and ingests the stream alongside client +# publishes. Reconnects and re-subscribes automatically. +# +# `trusted = true` is the relay-to-relay trust switch: events from that +# upstream skip Schnorr signature verification (the upstream already +# verified its own ingest; re-verifying burns ~8% of ingest CPU). The +# trusted identity is the URL *this* relay dialed — TLS-authenticated +# for wss:// — so an inbound client can never claim it. Default false: +# mirror-but-verify. Only trust relays you operate or whose ingest +# discipline you'd stake your own db on. +# +# [[mirror]] +# url = "wss://upstream.example.com/" +# trusted = true +# backfill_seconds = 3600 + [admin] # NIP-86 relay management API. When `pubkeys` is non-empty, the relay # accepts HTTP POST application/nostr+json+rpc on the same URL, diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt index 963269af0d..89dc1c989f 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt @@ -24,6 +24,8 @@ import com.vitorpamplona.geode.config.BannedEntry import com.vitorpamplona.geode.config.RuntimeConfig import com.vitorpamplona.geode.config.RuntimeConfigData import com.vitorpamplona.geode.config.StaticConfig +import com.vitorpamplona.geode.mirror.MirrorUpstream +import com.vitorpamplona.geode.mirror.MirrorWorker import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy @@ -172,10 +174,37 @@ fun main(args: Array) { callGroupSize = config.network.call_group_size, ).start() + // `[[mirror]]` upstreams: dial each configured relay and stream its + // events into the local store. `trusted = true` entries skip Schnorr + // verification for that connection (relay-to-relay trust) — only + // meaningful while verify is on; with --no-verify nothing verifies + // anyway. Never mirror ourselves: a self-URL would echo every local + // publish back forever. + val upstreams = + config.mirror.map { + MirrorUpstream( + url = it.url.normalizeRelayUrl(), + trusted = it.trusted, + backfillSeconds = it.backfill_seconds, + ) + } + require(upstreams.none { it.url == advertisedUrl }) { + "[[mirror]] must not list this relay's own URL ($advertisedUrl)" + } + val mirror = + if (upstreams.isEmpty()) { + null + } else { + MirrorWorker(upstreams, relay.server).also { it.start() } + } + Runtime.getRuntime().addShutdownHook( Thread { - // Each step wrapped so a throw in `server.stop()` doesn't - // skip `relay.close()` (which closes the SQLite store). + // Each step wrapped so a throw in any stage doesn't skip + // `relay.close()` (which closes the SQLite store). The mirror + // goes first: stop pulling new events before the queue and + // store beneath it shut down. + runCatching { mirror?.close() } runCatching { server.stop() } runCatching { relay.close() } }, @@ -183,6 +212,10 @@ fun main(args: Array) { println("geode listening on ${server.url}") println("NIP-11 info doc: curl -H 'Accept: application/nostr+json' http://$advertisedHost:$port$path") + if (upstreams.isNotEmpty()) { + val trusted = upstreams.count { it.trusted } + println("mirroring ${upstreams.size} upstream relay(s), $trusted trusted (signature verification skipped)") + } // Park the main thread; shutdown hook handles teardown. Thread.currentThread().join() diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt index b9c9f6aacd..ba7291330a 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt @@ -45,6 +45,8 @@ data class StaticConfig( val authorization: AuthorizationSection = AuthorizationSection(), val admin: AdminSection = AdminSection(), val negentropy: NegentropySection = NegentropySection(), + /** `[[mirror]]` entries — upstream relays this relay streams from. */ + val mirror: List = emptyList(), ) { fun resolveInfo(fullTextSearch: Boolean = true): RelayInfo = RelayInfo( @@ -160,6 +162,27 @@ data class StaticConfig( val kind_blacklist: List = emptyList(), ) + /** + * One upstream relay to mirror, declared as a `[[mirror]]` TOML array + * entry. The relay dials [url] itself, subscribes to everything newer + * than `now - backfill_seconds`, and ingests the stream through the + * same group-commit writer as client publishes (reconnects and + * re-subscribes automatically). + * + * [trusted] is the relay-to-relay trust switch (strfry's model): + * `true` skips Schnorr signature verification for events arriving on + * this connection — sound only when the upstream verifies its own + * ingest, which is why it defaults to `false` (mirror-but-verify). + * The identity being trusted is the URL this relay dialed (TLS- + * authenticated for `wss://`), never anything a peer claims. + */ + data class MirrorSection( + val url: String, + val trusted: Boolean = false, + /** How far back the initial subscription reaches. 0 = live-only. */ + val backfill_seconds: Long = 0L, + ) + /** * NIP-86 admin. [pubkeys] non-empty opens the POST endpoint at the * relay path; only NIP-98 tokens signed by these pubkeys dispatch. diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt new file mode 100644 index 0000000000..f6f5b18468 --- /dev/null +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt @@ -0,0 +1,190 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.geode.mirror + +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient +import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl +import com.vitorpamplona.quartz.nip01Core.relay.server.NostrServer +import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket +import com.vitorpamplona.quartz.nip01Core.store.IEventStore +import com.vitorpamplona.quartz.utils.Log +import com.vitorpamplona.quartz.utils.TimeUtils +import kotlinx.coroutines.CancellationException +import kotlinx.coroutines.CoroutineScope +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.SupervisorJob +import kotlinx.coroutines.cancel +import kotlinx.coroutines.channels.Channel +import kotlinx.coroutines.launch +import okhttp3.OkHttpClient +import java.util.concurrent.atomic.AtomicLong + +/** + * One upstream relay this relay mirrors, from the `[[mirror]]` config. + * + * [trusted] is the relay-to-relay trust switch: events streamed from this + * upstream skip Schnorr signature verification on ingest. The trusted + * identity is [url] — the address *this* relay dialed (TLS-authenticated + * for `wss://`), never anything the peer claims — so the skip can't be + * hijacked by an inbound client. + */ +class MirrorUpstream( + val url: NormalizedRelayUrl, + val trusted: Boolean, + /** How far back the initial REQ reaches. 0 = live-only from connect. */ + val backfillSeconds: Long = 0L, +) + +/** + * Streams events from configured upstream relays into the local relay — + * geode's equivalent of `strfry router` in the "down" direction. + * + * One [NostrClient] holds every upstream connection; the client owns + * reconnects, exponential backoff, and re-sending the REQ after a drop. + * Each upstream gets its own subscription over an open filter + * (`since = now - backfill`), and every EVENT that arrives is handed to + * [NostrServer.ingest] — the same group-commit writer and live fanout a + * client publish takes — with `skipVerify` set for [MirrorUpstream.trusted] + * upstreams (the upstream already verified its ingest; re-verifying here + * only burns CPU — Schnorr verify profiles at ~8% of busy ingest CPU). + * + * After a reconnect the upstream replays everything since the boot-time + * `since`; replayed duplicates are rejected by the store's unique id + * constraint and only show up in [rejected]. + * + * Listener callbacks can't suspend, so events funnel through an unbounded + * [inbound] channel into one consumer coroutine whose [NostrServer.ingest] + * call suspends on the ingest queue's backpressure. The buffer is unbounded + * for the same reason the client's receive channels are (see + * `BasicOkHttpWebSocket`): blocking the socket reader parks the backlog on + * infrastructure that isn't ours. + */ +class MirrorWorker( + private val upstreams: List, + private val server: NostrServer, + /** + * Transport override for tests (e.g. `InProcessRelays`). Defaults to + * a real OkHttp WebSocket per upstream. + */ + websocketBuilder: WebsocketBuilder? = null, +) : AutoCloseable { + private val scope = CoroutineScope(Dispatchers.IO + SupervisorJob()) + + private val okhttp: OkHttpClient? = if (websocketBuilder == null) OkHttpClient.Builder().build() else null + + private val client = + NostrClient( + websocketBuilder = websocketBuilder ?: BasicOkHttpWebSocket.Builder { okhttp!! }, + parentScope = scope, + ) + + private class Inbound( + val event: Event, + val skipVerify: Boolean, + ) + + private val inbound = Channel(Channel.UNLIMITED) + + /** Events accepted into the local store (excludes duplicates). */ + val accepted = AtomicLong(0) + + /** Events the store rejected — mostly duplicate replays after a reconnect. */ + val rejected = AtomicLong(0) + + /** Dials every upstream and starts streaming. Call once. */ + fun start() { + scope.launch { + for (msg in inbound) { + try { + server.ingest(msg.event, msg.skipVerify) { outcome -> + when (outcome) { + IEventStore.InsertOutcome.Accepted -> accepted.incrementAndGet() + is IEventStore.InsertOutcome.Rejected -> { + rejected.incrementAndGet() + Log.d("MirrorWorker") { "rejected ${msg.event.id}: ${outcome.reason}" } + } + } + } + } catch (e: CancellationException) { + throw e + } catch (e: Throwable) { + // A server shutting down closes its ingest queue while + // events may still be buffered here; an uncaught throw + // would leak into the scope's default handler (and + // poison unrelated runTest tests on CI). Stop pulling — + // the relay beneath us is going away. + Log.w("MirrorWorker") { "ingest failed, stopping mirror consumer: ${e.message}" } + break + } + } + } + + val since = TimeUtils.now() + upstreams.forEachIndexed { i, up -> + val listener = + object : SubscriptionListener { + override fun onEvent( + event: Event, + isLive: Boolean, + relay: NormalizedRelayUrl, + forFilters: List?, + ) { + inbound.trySend(Inbound(event, up.trusted)) + } + + override fun onCannotConnect( + relay: NormalizedRelayUrl, + message: String, + forFilters: List?, + ) { + Log.w("MirrorWorker") { "cannot reach upstream ${relay.url}: $message" } + } + } + client.subscribe( + subId = "geode-mirror-$i", + filters = mapOf(up.url to listOf(Filter(since = since - up.backfillSeconds))), + listener = listener, + ) + } + client.connect() + } + + /** + * Stops pulling from every upstream. In-flight ingest submissions + * drain through the server's queue; events still buffered in + * [inbound] are dropped — the next boot's `since` overlaps only if + * the operator configured a backfill window, which is the documented + * trade-off of a live mirror. + */ + override fun close() { + // Close the client first so no listener callback races the + // channel close below. + runCatching { client.close() } + inbound.close() + scope.cancel() + okhttp?.dispatcher?.executorService?.shutdown() + okhttp?.connectionPool?.evictAll() + } +} diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt index e21469938e..2d478e52ef 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt @@ -46,6 +46,36 @@ class StaticConfigTest { assertEquals(false, c.options.verify_signatures) } + @Test + fun mirrorSectionDefaultsToEmpty() { + assertTrue(StaticConfig.fromToml("").mirror.isEmpty()) + } + + @Test + fun parsesMirrorUpstreams() { + val toml = + """ + [[mirror]] + url = "wss://trusted.upstream.example/" + trusted = true + backfill_seconds = 3600 + + [[mirror]] + url = "wss://public.upstream.example/" + """.trimIndent() + + val c = StaticConfig.fromToml(toml) + + assertEquals(2, c.mirror.size) + assertEquals("wss://trusted.upstream.example/", c.mirror[0].url) + assertEquals(true, c.mirror[0].trusted) + assertEquals(3600L, c.mirror[0].backfill_seconds) + // Trust is opt-in per upstream: the default is mirror-but-verify. + assertEquals("wss://public.upstream.example/", c.mirror[1].url) + assertEquals(false, c.mirror[1].trusted) + assertEquals(0L, c.mirror[1].backfill_seconds) + } + @Test fun parsesAllSectionsTogether() { val toml = diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt new file mode 100644 index 0000000000..440e40abf7 --- /dev/null +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt @@ -0,0 +1,165 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.geode.mirror + +import com.vitorpamplona.geode.InProcessRelays +import com.vitorpamplona.geode.RelayEngine +import com.vitorpamplona.geode.testing.preload +import com.vitorpamplona.geode.testing.publish +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.core.toHexKey +import com.vitorpamplona.quartz.nip01Core.crypto.EventAssembler +import com.vitorpamplona.quartz.nip01Core.crypto.KeyPair +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.TimeUtils +import kotlinx.coroutines.delay +import kotlinx.coroutines.runBlocking +import kotlinx.coroutines.withTimeout +import org.junit.After +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * End-to-end `[[mirror]]` behavior over the in-process transport: a + * downstream relay (verification ON via the parallel-verify queue) dials + * an upstream and streams its events. + * + * - `trusted = true` — the relay-to-relay trust switch — must land + * events the downstream could never verify itself. + * - `trusted = false` must keep verify-everything semantics: forged + * events from the upstream are dropped, valid ones land. + */ +class MirrorWorkerTest { + private val upstreamUrl = RelayUrlNormalizer.normalize("ws://upstream.relay/") + private val downstreamUrl = RelayUrlNormalizer.normalize("ws://downstream.relay/") + + /** Upstream side: no verification (EmptyPolicy), so forged events store fine. */ + private val hub = InProcessRelays() + + /** Downstream side: signature verification on, in the IngestQueue. */ + private val downstreamStore = EventStore(null) + private val downstream = + RelayEngine( + url = downstreamUrl, + store = downstreamStore, + parallelVerify = true, + ) + + private var worker: MirrorWorker? = null + + private val signer = KeyPair() + + @After + fun tearDown() { + worker?.close() + downstream.close() + hub.close() + } + + private fun startMirror(trusted: Boolean): MirrorWorker = + MirrorWorker( + upstreams = listOf(MirrorUpstream(upstreamUrl, trusted = trusted, backfillSeconds = 3600)), + server = downstream.server, + websocketBuilder = hub, + ).also { + worker = it + it.start() + } + + private fun forgedEvent(idSeed: Int): Event = + Event( + id = idSeed.toString().padStart(64, '0'), + pubKey = "1".repeat(64), + createdAt = TimeUtils.now() - idSeed, + kind = 1, + tags = emptyArray(), + content = "forged $idSeed", + sig = "f".repeat(128), + ) + + private fun signedEvent(content: String): Event = + EventAssembler.hashAndSign( + pubKey = signer.pubKey.toHexKey(), + createdAt = TimeUtils.now() - 5, + kind = 1, + tags = emptyArray(), + content = content, + privKey = signer.privKey!!, + ) + + private suspend fun awaitDownstreamCount(expected: Int) = + withTimeout(15_000) { + while (downstreamStore.count(Filter()) < expected) delay(25) + } + + private suspend fun await(condition: suspend () -> Boolean) = + withTimeout(15_000) { + while (!condition()) delay(25) + } + + // NOTE: assertions are on the downstream STORE, not on exact counter + // values — the client may legitimately re-send its REQ while settling + // (connect + filter sync), so an upstream can replay an event twice and + // the duplicate shows up as one extra rejection. That's the documented + // mirror behavior, not a failure. + + @Test + fun trustedUpstreamLandsEventsTheDownstreamCannotVerify() = + runBlocking { + val stored = forgedEvent(1) + val live = forgedEvent(2) + + // Stored replay: exists on the upstream before the mirror dials. + hub.getOrCreate(upstreamUrl).preload(stored) + + startMirror(trusted = true) + awaitDownstreamCount(1) + + // Live tail: published upstream after the mirror subscribed. + hub.getOrCreate(upstreamUrl).publish(live) + awaitDownstreamCount(2) + + val ids = downstreamStore.query(Filter()).map { it.id }.toSet() + assertEquals(setOf(stored.id, live.id), ids) + } + + @Test + fun untrustedUpstreamStillVerifiesEverything() = + runBlocking { + val forged = forgedEvent(1) + val valid = signedEvent("the real one") + hub.getOrCreate(upstreamUrl).preload(forged, valid) + + val mirror = startMirror(trusted = false) + + // The valid event lands; the forged one is verified and dropped + // (a forged event can never land untrusted, so once both have + // been processed the store can only hold the valid one). + await { mirror.rejected.get() >= 1 && downstreamStore.count(Filter()) == 1 } + + val stored = downstreamStore.query(Filter()).single() + assertEquals(valid.id, stored.id) + assertTrue(downstreamStore.query(Filter(ids = listOf(forged.id))).isEmpty()) + } +} From a2b38cd1a3ef5c4ac1f186641288be4959071fb7 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 21:54:49 +0000 Subject: [PATCH 03/21] feat(geode): per-upstream [[mirror]] filters, strfry-router parity MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Each [[mirror]] entry takes an optional NIP-01 filter as a JSON object string, like strfry-router's per-stream filter. It is applied twice, matching strfry's design (cmd_router.cpp): - it shapes the REQ sent upstream (the mirror still owns since via backfill_seconds and strips limit — the subscription is unbounded); - every delivered event is re-checked against it before ingest, so an upstream answering outside its REQ — including a trusted one whose events skip signature verification — can only inject events inside the operator-declared scope. Out-of-scope deliveries surface on a 'filtered' counter. Malformed filter JSON fails the boot, not the first delivery. Several disjoint scopes for one upstream = repeat [[mirror]] with the same url. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- geode/config.example.toml | 10 +++++ .../kotlin/com/vitorpamplona/geode/Main.kt | 28 ++++++++++++-- .../geode/config/StaticConfig.kt | 12 ++++++ .../geode/mirror/MirrorWorker.kt | 37 ++++++++++++++++++- .../geode/config/StaticConfigTest.kt | 18 ++++++++- .../geode/mirror/MirrorWorkerTest.kt | 37 +++++++++++++++++-- 6 files changed, 132 insertions(+), 10 deletions(-) diff --git a/geode/config.example.toml b/geode/config.example.toml index 98ad0fc2d4..6b18b478d2 100644 --- a/geode/config.example.toml +++ b/geode/config.example.toml @@ -99,10 +99,20 @@ require_auth = false # mirror-but-verify. Only trust relays you operate or whose ingest # discipline you'd stake your own db on. # +# `filter` (optional) scopes an upstream, as a NIP-01 filter JSON +# object — same idea as strfry-router's per-stream filter. It shapes +# the REQ sent upstream AND every delivered event is re-checked +# against it before ingest, so even a trusted upstream can only +# inject events inside the declared scope. `since`/`limit` inside it +# are ignored (backfill_seconds owns the time window). Omit to mirror +# everything; for several disjoint scopes, repeat [[mirror]] with the +# same url. +# # [[mirror]] # url = "wss://upstream.example.com/" # trusted = true # backfill_seconds = 3600 +# filter = '{"kinds":[0,1,3,7],"#t":["nostr"]}' [admin] # NIP-86 relay management API. When `pubkeys` is non-empty, the relay diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt index 89dc1c989f..d096b34f8f 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt @@ -26,6 +26,8 @@ import com.vitorpamplona.geode.config.RuntimeConfigData import com.vitorpamplona.geode.config.StaticConfig import com.vitorpamplona.geode.mirror.MirrorUpstream import com.vitorpamplona.geode.mirror.MirrorWorker +import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy @@ -181,11 +183,29 @@ fun main(args: Array) { // anyway. Never mirror ourselves: a self-URL would echo every local // publish back forever. val upstreams = - config.mirror.map { + config.mirror.map { m -> + // The optional scope filter is parsed eagerly so a malformed + // JSON object fails the boot, not the first delivery. Its + // since/limit are stripped: the mirror owns the time window + // (backfill_seconds) and never bounds the subscription. + val scope = + m.filter?.let { json -> + val parsed = + try { + OptimizedJsonMapper.fromJsonTo(json) + } catch (e: Exception) { + throw IllegalArgumentException( + "[[mirror]] filter for ${m.url} is not a valid NIP-01 filter object: $json", + e, + ) + } + parsed.copy(since = null, limit = null) + } MirrorUpstream( - url = it.url.normalizeRelayUrl(), - trusted = it.trusted, - backfillSeconds = it.backfill_seconds, + url = m.url.normalizeRelayUrl(), + trusted = m.trusted, + backfillSeconds = m.backfill_seconds, + filter = scope, ) } require(upstreams.none { it.url == advertisedUrl }) { diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt index ba7291330a..7191da08f3 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt @@ -181,6 +181,18 @@ data class StaticConfig( val trusted: Boolean = false, /** How far back the initial subscription reaches. 0 = live-only. */ val backfill_seconds: Long = 0L, + /** + * Optional NIP-01 filter as a JSON object string (strfry-router + * parity), e.g. `'{"kinds":[0,1,3],"#t":["nostr"]}'`. Scopes what + * this upstream is asked for AND what it is allowed to deliver — + * every received event is re-checked against it before ingest, so + * even a trusted upstream can't push events outside the declared + * scope. `since` is managed by the mirror (see [backfill_seconds]) + * and `limit` is transport-level, so both are ignored if present. + * Omitted = mirror everything. For several disjoint filters, add + * several `[[mirror]]` entries with the same url. + */ + val filter: String? = null, ) /** diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt index f6f5b18468..2350bf72c5 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt @@ -55,6 +55,16 @@ class MirrorUpstream( val trusted: Boolean, /** How far back the initial REQ reaches. 0 = live-only from connect. */ val backfillSeconds: Long = 0L, + /** + * Optional scope for this upstream (strfry-router's per-stream + * `filter`). Used twice: it shapes the REQ sent upstream, and every + * delivered event is re-checked against it before ingest — so even a + * [trusted] upstream can only inject events inside the declared + * scope. Its `since`/`limit` are ignored ([backfillSeconds] owns the + * time window; the subscription is unbounded). `null` mirrors + * everything. + */ + val filter: Filter? = null, ) /** @@ -113,6 +123,13 @@ class MirrorWorker( /** Events the store rejected — mostly duplicate replays after a reconnect. */ val rejected = AtomicLong(0) + /** + * Deliveries dropped by the [MirrorUpstream.filter] re-check before + * ever reaching the store — an upstream sending these is answering + * outside the REQ it was given. + */ + val filtered = AtomicLong(0) + /** Dials every upstream and starts streaming. Call once. */ fun start() { scope.launch { @@ -151,6 +168,17 @@ class MirrorWorker( relay: NormalizedRelayUrl, forFilters: List?, ) { + // strfry-router parity: never take the upstream's + // word for what matched. Re-checking the configured + // scope here means even a trusted (skip-verify) + // upstream can only inject events the operator + // declared — the REQ shapes what we ask for, this + // shapes what we accept. + if (up.filter != null && !up.filter.match(event)) { + filtered.incrementAndGet() + Log.d("MirrorWorker") { "out-of-scope from ${relay.url}: ${event.id}" } + return + } inbound.trySend(Inbound(event, up.trusted)) } @@ -162,9 +190,16 @@ class MirrorWorker( Log.w("MirrorWorker") { "cannot reach upstream ${relay.url}: $message" } } } + // The operator's filter scopes the REQ; the mirror owns the + // time window (since) and never bounds the result (limit). + val reqFilter = + (up.filter ?: Filter()).copy( + since = since - up.backfillSeconds, + limit = null, + ) client.subscribe( subId = "geode-mirror-$i", - filters = mapOf(up.url to listOf(Filter(since = since - up.backfillSeconds))), + filters = mapOf(up.url to listOf(reqFilter)), listener = listener, ) } diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt index 2d478e52ef..765bc58881 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt @@ -20,6 +20,8 @@ */ package com.vitorpamplona.geode.config +import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import java.io.File import kotlin.test.Test import kotlin.test.assertEquals @@ -51,6 +53,16 @@ class StaticConfigTest { assertTrue(StaticConfig.fromToml("").mirror.isEmpty()) } + @Test + fun mirrorFilterJsonParsesToANip01Filter() { + // The exact parse Main.kt runs on [[mirror]].filter at boot. + val f = OptimizedJsonMapper.fromJsonTo("""{"kinds":[0,1],"#t":["nostr"],"since":123,"limit":5}""") + assertEquals(listOf(0, 1), f.kinds) + assertEquals(listOf("nostr"), f.tags?.get("t")) + assertEquals(123L, f.since) + assertEquals(5, f.limit) + } + @Test fun parsesMirrorUpstreams() { val toml = @@ -59,6 +71,7 @@ class StaticConfigTest { url = "wss://trusted.upstream.example/" trusted = true backfill_seconds = 3600 + filter = '{"kinds":[0,1,3],"#t":["nostr"]}' [[mirror]] url = "wss://public.upstream.example/" @@ -70,10 +83,13 @@ class StaticConfigTest { assertEquals("wss://trusted.upstream.example/", c.mirror[0].url) assertEquals(true, c.mirror[0].trusted) assertEquals(3600L, c.mirror[0].backfill_seconds) - // Trust is opt-in per upstream: the default is mirror-but-verify. + assertEquals("""{"kinds":[0,1,3],"#t":["nostr"]}""", c.mirror[0].filter) + // Trust and scoping are opt-in per upstream: the default is + // mirror-everything-but-verify. assertEquals("wss://public.upstream.example/", c.mirror[1].url) assertEquals(false, c.mirror[1].trusted) assertEquals(0L, c.mirror[1].backfill_seconds) + assertEquals(null, c.mirror[1].filter) } @Test diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt index 440e40abf7..e2f62e2531 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt @@ -77,9 +77,12 @@ class MirrorWorkerTest { hub.close() } - private fun startMirror(trusted: Boolean): MirrorWorker = + private fun startMirror( + trusted: Boolean, + filter: Filter? = null, + ): MirrorWorker = MirrorWorker( - upstreams = listOf(MirrorUpstream(upstreamUrl, trusted = trusted, backfillSeconds = 3600)), + upstreams = listOf(MirrorUpstream(upstreamUrl, trusted = trusted, backfillSeconds = 3600, filter = filter)), server = downstream.server, websocketBuilder = hub, ).also { @@ -87,12 +90,15 @@ class MirrorWorkerTest { it.start() } - private fun forgedEvent(idSeed: Int): Event = + private fun forgedEvent( + idSeed: Int, + kind: Int = 1, + ): Event = Event( id = idSeed.toString().padStart(64, '0'), pubKey = "1".repeat(64), createdAt = TimeUtils.now() - idSeed, - kind = 1, + kind = kind, tags = emptyArray(), content = "forged $idSeed", sig = "f".repeat(128), @@ -144,6 +150,29 @@ class MirrorWorkerTest { assertEquals(setOf(stored.id, live.id), ids) } + @Test + fun filterScopesTheMirrorToDeclaredKinds() = + runBlocking { + val wantedStored = forgedEvent(1, kind = 1) + val unwantedStored = forgedEvent(2, kind = 7) + hub.getOrCreate(upstreamUrl).preload(wantedStored, unwantedStored) + + startMirror(trusted = true, filter = Filter(kinds = listOf(1))) + awaitDownstreamCount(1) + + // Live tail: the out-of-scope kind is published FIRST, so by the + // time the in-scope one lands downstream (same connection, same + // ordered pipeline), the kind-7 has already had its chance. + hub.getOrCreate(upstreamUrl).publish(forgedEvent(3, kind = 7)) + val wantedLive = forgedEvent(4, kind = 1) + hub.getOrCreate(upstreamUrl).publish(wantedLive) + awaitDownstreamCount(2) + + val stored = downstreamStore.query(Filter()) + assertEquals(setOf(wantedStored.id, wantedLive.id), stored.map { it.id }.toSet()) + assertTrue(stored.all { it.kind == 1 }) + } + @Test fun untrustedUpstreamStillVerifiesEverything() = runBlocking { From c2cabf3c4710490d08358380dfa4b0bb96a78d74 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 22:26:15 +0000 Subject: [PATCH 04/21] =?UTF-8?q?feat(geode):=20mirror=20survives=20upstre?= =?UTF-8?q?am=20restarts=20=E2=80=94=20retry=20pump,=20ping=20keepalive,?= =?UTF-8?q?=20e2e=20proof?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Measured over the real OkHttp transport (upstream killed mid-mirror and restarted on the same port): NostrClient re-dials once on disconnect and then falls back to its 60s keep-alive, which showed up as a 61s mirror blackout when the immediate re-dial raced the port rebind. - Retry pump: the worker nudges the pool every 5s. Cheap and safe — reconnectIfNeedsTo skips connected relays and each relay's exponential backoff (1s doubling, 5min cap) still gates real dial attempts, so dead upstreams aren't hammered. Restart recovery drops from 61s to ~6s in the new reconnect test. - WebSocket pings (120s, same as the Android relay pool): without them a half-open connection (network drop, no FIN) never fires onDisconnected and the mirror would stall silently forever. - MirrorWorkerReconnectTest: end-to-end over a real Ktor port — ride through the drop, re-subscribe, drop the duplicate replay, pull the event that only exists on the new instance. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../geode/mirror/MirrorWorker.kt | 44 ++++- .../geode/mirror/MirrorWorkerReconnectTest.kt | 157 ++++++++++++++++++ 2 files changed, 200 insertions(+), 1 deletion(-) create mode 100644 geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt index 2350bf72c5..d12fffc496 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt @@ -37,8 +37,10 @@ import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.SupervisorJob import kotlinx.coroutines.cancel import kotlinx.coroutines.channels.Channel +import kotlinx.coroutines.delay import kotlinx.coroutines.launch import okhttp3.OkHttpClient +import java.time.Duration import java.util.concurrent.atomic.AtomicLong /** @@ -102,7 +104,24 @@ class MirrorWorker( ) : AutoCloseable { private val scope = CoroutineScope(Dispatchers.IO + SupervisorJob()) - private val okhttp: OkHttpClient? = if (websocketBuilder == null) OkHttpClient.Builder().build() else null + /** + * The ping interval matters on a long-running daemon: a half-open + * upstream connection (network drop with no FIN) never fires + * onDisconnected on its own, so without pings the client would + * believe it is connected forever and stop mirroring silently. A + * missed pong fails the socket, which routes into the normal + * disconnect → backoff → re-dial path. Same value the Android app + * uses for its relay pool. + */ + private val okhttp: OkHttpClient? = + if (websocketBuilder == null) { + OkHttpClient + .Builder() + .pingInterval(Duration.ofSeconds(PING_INTERVAL_SECS)) + .build() + } else { + null + } private val client = NostrClient( @@ -204,6 +223,21 @@ class MirrorWorker( ) } client.connect() + + // Retry pump. NostrClient re-dials once on disconnect and then + // relies on its 60s keep-alive — measured as a 61s mirror blackout + // when an upstream restarts and the immediate re-dial races the + // port rebind. Poking more often costs nothing: reconnectIfNeedsTo + // skips connected relays, and each relay's exponential backoff + // (1s doubling, 5min cap) still gates actual dial attempts, so a + // long-dead upstream is not hammered — only a briefly-restarting + // one is picked back up in seconds instead of a minute. + scope.launch { + while (true) { + delay(RECONNECT_POKE_MS) + client.reconnect(onlyIfChanged = true, ignoreRetryDelays = false) + } + } } /** @@ -222,4 +256,12 @@ class MirrorWorker( okhttp?.dispatcher?.executorService?.shutdown() okhttp?.connectionPool?.evictAll() } + + private companion object { + /** Matches the Android app's relay-pool WebSocket ping interval. */ + const val PING_INTERVAL_SECS = 120L + + /** How often the retry pump nudges disconnected upstreams. */ + const val RECONNECT_POKE_MS = 5_000L + } } diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt new file mode 100644 index 0000000000..9268ca0384 --- /dev/null +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt @@ -0,0 +1,157 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.geode.mirror + +import com.vitorpamplona.geode.KtorRelay +import com.vitorpamplona.geode.RelayEngine +import com.vitorpamplona.geode.testing.preload +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.TimeUtils +import kotlinx.coroutines.delay +import kotlinx.coroutines.runBlocking +import kotlinx.coroutines.withTimeout +import org.junit.After +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * The mirror is a long-running daemon, so surviving its upstream is not + * optional. This drives the REAL production transport (OkHttp WebSocket + * against a real Ktor port — no in-process shortcut): the upstream is + * stopped mid-mirror and a new instance is brought up on the same port. + * The worker must ride through the disconnect (NostrClient's backoff + + * keep-alive own the re-dial), re-send its REQ, drop the replayed + * duplicate, and pull the event that only exists on the new instance. + * + * Timing note: after a stable connection drops, the client's backoff is + * reset to 1s and re-dial attempts double from there. The worker's 5s + * retry pump nudges the pool well ahead of NostrClient's own 60s + * keep-alive (without it, this test measured a 61s blackout when the + * immediate re-dial raced the port rebind; with it, ~6s). The generous + * timeout only covers a worst-case scheduling stall on a loaded CI + * runner. + */ +class MirrorWorkerReconnectTest { + private val downstreamStore = EventStore(null) + private val downstream = + RelayEngine( + url = "ws://127.0.0.1:7797/".normalizeRelayUrl(), + store = downstreamStore, + parallelVerify = true, + ) + + private var worker: MirrorWorker? = null + private var upstreamServer: KtorRelay? = null + private var upstreamEngine: RelayEngine? = null + + @After + fun tearDown() { + worker?.close() + upstreamServer?.stop(gracePeriodMillis = 0, timeoutMillis = 1_000) + upstreamEngine?.close() + downstream.close() + } + + private fun startUpstream(port: Int): KtorRelay { + val engine = RelayEngine(url = "ws://127.0.0.1:7796/".normalizeRelayUrl()) + val server = KtorRelay(engine, host = "127.0.0.1", port = port).start() + upstreamEngine = engine + upstreamServer = server + return server + } + + private fun forgedEvent(idSeed: Int): Event = + Event( + id = idSeed.toString().padStart(64, '0'), + pubKey = "1".repeat(64), + createdAt = TimeUtils.now() - idSeed, + kind = 1, + tags = emptyArray(), + content = "forged $idSeed", + sig = "f".repeat(128), + ) + + private suspend fun awaitDownstreamCount(expected: Int) = + withTimeout(120_000) { + while (downstreamStore.count(Filter()) < expected) delay(50) + } + + @Test + fun mirrorSurvivesUpstreamRestart() = + runBlocking { + val beforeRestart = forgedEvent(1) + val afterRestart = forgedEvent(2) + + // First upstream instance on an OS-assigned port. + val first = startUpstream(port = 0) + val port = + first.url + .normalizeRelayUrl() + .url + .substringAfterLast(':') + .trimEnd('/') + .toInt() + upstreamEngine!!.preload(beforeRestart) + + // Real OkHttp transport: websocketBuilder deliberately omitted. + val mirror = + MirrorWorker( + upstreams = + listOf( + MirrorUpstream( + url = first.url.normalizeRelayUrl(), + trusted = true, + backfillSeconds = 3600, + ), + ), + server = downstream.server, + ).also { worker = it } + mirror.start() + awaitDownstreamCount(1) + + // Kill the upstream: every socket drops, the port closes. + first.stop(gracePeriodMillis = 0, timeoutMillis = 1_000) + upstreamEngine!!.close() + + // Bring up a NEW instance on the SAME port. Its store has the + // old event (so the re-sent REQ replays a duplicate) plus one + // that only exists post-restart. + val second = startUpstream(port = port) + upstreamEngine!!.preload(beforeRestart, afterRestart) + + // The worker must reconnect on its own — no external poke — + // re-subscribe, and pull the new event. + awaitDownstreamCount(2) + + assertEquals( + setOf(beforeRestart.id, afterRestart.id), + downstreamStore.query(Filter()).map { it.id }.toSet(), + ) + // The duplicate replay of the pre-restart event was dropped by + // the store, not double-inserted. + assertEquals(2, downstreamStore.count(Filter())) + + second.stop(gracePeriodMillis = 0, timeoutMillis = 1_000) + } +} From 175cac3e467a77d083c5d337a4fac5ad9a448e75 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 22:36:06 +0000 Subject: [PATCH 05/21] =?UTF-8?q?docs(quartz):=20plan=20=E2=80=94=20increm?= =?UTF-8?q?ental=20live=20negentropy=20storage?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Design for backlog item 3 of the relay performance campaign: an always-current (created_at, id) index maintained from the store's write path so cold NEG-OPENs stop paying the full scan + O(n log n) seal (~340 ms at 50k events vs strfry's ~21 ms off its live tree). Covers the snapshot/COW model, the removal-correctness split (RETURNING deltas for replaceable overwrites, wholesale invalidation for rare delete paths), IndexingStrategy gating so app-side stores are untouched, and the micro + relayBench A/B measurement plan. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- PLANS.md | 3 +- ...26-07-03-incremental-negentropy-storage.md | 89 +++++++++++++++++++ quartz/plans/README.md | 3 +- 3 files changed, 93 insertions(+), 2 deletions(-) create mode 100644 quartz/plans/2026-07-03-incremental-negentropy-storage.md diff --git a/PLANS.md b/PLANS.md index e65a25495d..e3474caec3 100644 --- a/PLANS.md +++ b/PLANS.md @@ -33,7 +33,7 @@ listing (including archived plans), open that folder's `README.md`. | amethyst | 21 | 19 | 1 | 1 | 0 | [amethyst/plans](amethyst/plans/README.md) | | nestsClient | 26 | 23 | 1 | 2 | 0 | [nestsClient/plans](nestsClient/plans/README.md) | | desktopApp | 13 | 10 | 2 | 1 | 0 | [desktopApp/plans](desktopApp/plans/README.md) | -| quartz | 9 | 7 | 0 | 2 | 0 | [quartz/plans](quartz/plans/README.md) | +| quartz | 10 | 7 | 0 | 3 | 0 | [quartz/plans](quartz/plans/README.md) | | commons | 6 | 2 | 2 | 2 | 0 | [commons/plans](commons/plans/README.md) | | cli | 6 | 5 | 1 | 0 | 0 | [cli/plans](cli/plans/README.md) | | quic | 4 | 3 | 0 | 0 | 1 | [quic/plans](quic/plans/README.md) | @@ -67,6 +67,7 @@ them under each folder's `archive/` via the per-module index above. | amethyst | [napplet-inter-applet](amethyst/plans/2026-06-20-napplet-inter-applet.md) | NAP-INC / NAP-INTENT inter-applet messaging; prerequisites (multi-applet hosting, archetype registry, `MESSAGING` capability) not built. | | quartz | [local-headers-explorer](quartz/plans/2026-05-08-local-headers-explorer.md) | Headers-only Bitcoin P2P client to verify NIP-03 OTS attestations without a trusted block explorer. | | quartz | [giftwrap-deletion-requests](quartz/plans/2026-06-12-giftwrap-deletion-requests.md) | Let a recipient-authored kind-5 delete/block a gift wrap (kind 1059) addressed to them. | +| quartz | [incremental-negentropy-storage](quartz/plans/2026-07-03-incremental-negentropy-storage.md) | Always-current (created_at, id) index so cold NEG-OPENs stop paying a full scan + seal. | | commons | [event-renderer](commons/plans/2026-04-21-event-renderer.md) | Cross-platform UI-agnostic `RenderedEvent` subsystem shared by Amy, Desktop, Android; not started. | | commons | [amethyst-to-commons-migration](commons/plans/2026-05-30-amethyst-to-commons-migration.md) | Roadmap to move shared `amethyst` Android code into `commons`; keystone `Account`/`LocalCache` extraction not begun. | | desktopApp | [embedded-wallet-phase2-research](desktopApp/plans/2026-05-21-embedded-wallet-phase2-research.md) | Research for an embedded self-custodial Lightning wallet (Breez/ldk-node/lightning-kmp); parked, no code. | diff --git a/quartz/plans/2026-07-03-incremental-negentropy-storage.md b/quartz/plans/2026-07-03-incremental-negentropy-storage.md new file mode 100644 index 0000000000..6121a3c7c6 --- /dev/null +++ b/quartz/plans/2026-07-03-incremental-negentropy-storage.md @@ -0,0 +1,89 @@ +# Incremental live negentropy storage + +**Status: planned** — backlog item 3 of the relay performance campaign +(PR #3466 follow-up). + +## Problem + +A cold NEG-OPEN pays a full scan + O(n log n) seal: `snapshotIdsForNegentropy` +(SQL scan of every matching row) → `NegentropyServerSession.sealVector` +(sort + seal). relayBench measured ~340 ms per reconcile at 50k events before +the single-slot snapshot cache; strfry answers ~21 ms off its always-current +in-memory view. + +The cache added in #3466 (keyed on filter JSON + write generation + 30 s TTL) +only helps the *identical-filter, zero-writes-in-between* repeat. On a relay +ingesting continuously the generation moves constantly, so mirror heartbeats +are effectively always cold; any new filter is cold by definition. + +## Design + +### 1. `LiveNegentropyIndex` (quartz, server-only, opt-in) + +An always-current sorted set of `(createdAt, id₃₂)` maintained from the +store's write path — strfry's `MemoryView` equivalent, ~40 B/entry +(1M events ≈ 40 MB; capped by `negentropy.max_sync_events`). + +- **Structure**: single sorted array with binary-search insert. Nostr inserts + are near-tail (created_at ≈ now), so the memmove is tiny in the common + case; measure the out-of-order (backfill) worst case and only move to a + chunked layout if it shows. +- **Snapshot**: NEG-OPEN copies the array into a sealed `IStorage` — O(n) + arraycopy of already-sorted data (~2 MB at 50k, sub-ms), reusing the + existing single-slot cache so back-to-back opens share one snapshot. + Reconcile only reads (`size/getItem/iterate/indexAtOrBeforeBound`), so one + sealed snapshot serves any number of concurrent sessions. +- **Serves index-total filters only**: no `ids/authors/kinds/tags/search` + constraints. `since/until` ARE served — the structure is time-sorted, so a + time window is an index sub-range. Constrained filters keep the scan+seal + path (+ cache). The mirror-heartbeat pattern this optimizes is a broad + time-window filter, so this covers the case that matters. + +### 2. Removal correctness (the interesting part) + +The index must never advertise ids the store no longer has, or peers fetch +dead ids. Removal paths differ in frequency and get different treatment: + +- **Replaceable/addressable overwrite** (frequent — every kind 0/3/1xxxx/ + 3xxxx update): `ReplaceableModule`/`AddressableModule` run + `DELETE FROM event_headers WHERE …` inside the insert transaction. Add + `RETURNING created_at, id` (bundled SQLite ≥ 3.35) and report displaced + rows to the index alongside the insert. +- **Wholesale/rare paths** (kind-5 NIP-09, expiration sweep, right-to-vanish, + NIP-86 `delete(filter)`, FTS reindex/clear): invalidate the whole index; + the next NEG-OPEN rebuilds it lazily from one scan and incremental + maintenance resumes. Deletes are rare enough that occasional rebuilds beat + threading deltas through every module. + +### 3. Gating + +`IndexingStrategy.maintainLiveNegentropyIndex`, default **false** — library +defaults unchanged for app-side stores (campaign ground rule). geode's +`RelayIndexingStrategy` turns it on; config kill-switch under +`[negentropy]`. + +### 4. Concurrency + +Mutations happen only on the writer path (single-writer mutex — same +discipline as the FTS worker). Snapshots swap in via copy-on-write so a +reconcile never observes a mid-insert array. Shutdown: the index is memory +only, rebuilt on boot from the first NEG-OPEN's scan; no lifecycle beyond +the store's own close (ground rule 4: no worker, no uncaught exceptions). + +## Measurement plan + +1. Micro: quartz jvmTest benchmark, cold NEG-OPEN time at 50k events, + before/after (expect ~340 ms → single-digit ms). +2. Headline: relayBench pairwise sync, alternating A/B runs on the 50k + corpus (container noise ±10–30%; never trust a single run). Keep only + if it wins end-to-end; document either way. + +## Milestones + +1. `LiveNegentropyIndex` + unit tests (ordering, tail/backfill inserts, + removal, snapshot immutability under concurrent insert, cap behavior). +2. Displaced-row `RETURNING` plumbing in Replaceable/Addressable modules + + wholesale-invalidation hooks on the rare paths. +3. `LiveEventStore.sealedNegentropyStorage` wiring: index-total filters + from the index; everything else keeps scan+seal+cache. +4. Benchmarks (micro then relayBench A/B); revert if not a real win. diff --git a/quartz/plans/README.md b/quartz/plans/README.md index 5c0c538031..577d380717 100644 --- a/quartz/plans/README.md +++ b/quartz/plans/README.md @@ -1,12 +1,13 @@ # quartz plans -_Audited 2026-06-30. 9 plans: 7 shipped (archived), 0 in-progress, 2 queued, 0 abandoned._ +_Audited 2026-06-30. 10 plans: 7 shipped (archived), 0 in-progress, 3 queued, 0 abandoned._ ## Queued | Plan | Summary | | ---- | ------- | | [2026-05-08-local-headers-explorer.md](2026-05-08-local-headers-explorer.md) | Headers-only Bitcoin P2P client to verify NIP-03 OTS attestations without a trusted block explorer. | | [2026-06-12-giftwrap-deletion-requests.md](2026-06-12-giftwrap-deletion-requests.md) | Let a recipient-authored kind-5 delete/block a gift wrap (kind 1059) addressed to them. | +| [2026-07-03-incremental-negentropy-storage.md](2026-07-03-incremental-negentropy-storage.md) | Always-current (created_at, id) index so cold NEG-OPENs stop paying a full scan + seal (~340 ms at 50k vs strfry's ~21 ms). | ## Archived (shipped) | Plan | Summary | From 7ad7dee5fa3fa130c833102d09d2a8458f0e8f3a Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 22:41:35 +0000 Subject: [PATCH 06/21] =?UTF-8?q?feat(quartz):=20LiveNegentropyIndex=20?= =?UTF-8?q?=E2=80=94=20always-current=20(created=5Fat,=20id)=20set=20for?= =?UTF-8?q?=20NIP-77?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Milestone 1 of quartz/plans/2026-07-03-incremental-negentropy-storage.md. Sorted array with binary-search insert (near-tail in the common case), itemized remove for displaced rows, wholesale invalidate for delete paths that can't itemize, and sealed snapshots memoized per mutation generation — reconcile only reads, so one snapshot backs any number of concurrent sessions and stays immutable under later writes. Over-cap answers null so the caller keeps the strfry-parity NEG-ERR. Not wired into any store yet; next milestones plumb displaced-row deltas from the SQLite modules and serve index-total filters from it. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../nip77Negentropy/LiveNegentropyIndex.kt | 215 ++++++++++++++++++ .../LiveNegentropyIndexTest.kt | 174 ++++++++++++++ 2 files changed, 389 insertions(+) create mode 100644 quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndex.kt create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndexTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndex.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndex.kt new file mode 100644 index 0000000000..dd32d50d67 --- /dev/null +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndex.kt @@ -0,0 +1,215 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip77Negentropy + +import com.vitorpamplona.negentropy.storage.IStorage +import com.vitorpamplona.quartz.nip01Core.store.IdAndTime +import kotlin.concurrent.atomics.AtomicBoolean +import kotlin.concurrent.atomics.ExperimentalAtomicApi + +/** + * Always-current `(created_at, id)` index for NIP-77 negentropy — the + * strfry-parity answer to cold NEG-OPEN cost. Instead of paying a full + * table scan + O(n log n) seal on every open (relayBench: ~340 ms at + * 50k events; strfry serves ~21 ms off its always-current tree), the + * relay maintains this sorted set incrementally from the write path + * and a NEG-OPEN only pays one O(n) copy into a sealed snapshot — and + * even that is cached until the next mutation. + * + * Ordering matches negentropy's item order (`StorageUnit.compareTo`): + * `created_at` ascending, then id bytes ascending — for lowercase hex + * ids, byte order and string order agree, so entries compare by + * ([IdAndTime.createdAt], [IdAndTime.id]). + * + * Lifecycle contract (see + * `quartz/plans/2026-07-03-incremental-negentropy-storage.md`): + * + * - **Populate** with [rebuild] from one scan, then keep current with + * [insert] / [remove] as the store mutates. Inserts are near-tail in + * the common case (`created_at` ≈ now), so the memmove is tiny. + * - **Removals the caller can't itemize** (kind-5 deletes by filter, + * expiration sweeps, admin purges) call [invalidate] instead; the + * index answers nothing until the next [rebuild]. Correctness rule: + * the index must never advertise an id the store no longer has — + * peers would fetch dead ids — so when in doubt, invalidate. + * - **Snapshot** with [sealedSnapshot]; reconciliation only reads the + * sealed storage, so one snapshot backs any number of concurrent + * sessions and stays valid even if the live index mutates after. + * + * Thread-safety: all operations take a short spin lock (same pattern as + * `LiveEventStore`'s replay dedup). Mutations arrive from the store's + * single writer; snapshots from any REQ coroutine. + */ +@OptIn(ExperimentalAtomicApi::class) +class LiveNegentropyIndex { + private val lock = AtomicBoolean(false) + + /** Sorted by (createdAt, id). Only touched under [locked]. */ + private var entries = ArrayList() + + /** False until the first [rebuild], and after any [invalidate]. */ + private var populated = false + + /** Bumped on every mutation; keys the memoized snapshot. */ + private var generation = 0L + + private var cachedSnapshot: IStorage? = null + private var cachedGeneration = -1L + + private inline fun locked(block: () -> R): R { + while (lock.exchange(true)) { + while (lock.load()) { } + } + try { + return block() + } finally { + lock.store(false) + } + } + + private fun compare( + a: IdAndTime, + b: IdAndTime, + ): Int { + val byTime = a.createdAt.compareTo(b.createdAt) + if (byTime != 0) return byTime + return a.id.compareTo(b.id) + } + + /** + * Binary search for [entry]'s position. Returns the index when + * present, or `-(insertionPoint) - 1` when absent — same contract + * as `java.util.Collections.binarySearch`. + */ + private fun search(entry: IdAndTime): Int { + var low = 0 + var high = entries.size - 1 + while (low <= high) { + val mid = (low + high) ushr 1 + val cmp = compare(entries[mid], entry) + when { + cmp < 0 -> low = mid + 1 + cmp > 0 -> high = mid - 1 + else -> return mid + } + } + return -(low + 1) + } + + /** True once [rebuild] ran and no [invalidate] happened since. */ + fun isPopulated(): Boolean = locked { populated } + + fun size(): Int = locked { entries.size } + + /** + * Replaces the whole index with [snapshot] (any order; sorted here) + * and marks it populated. This is the recovery path after boot or + * [invalidate] — one store scan, then incremental maintenance + * resumes. + */ + fun rebuild(snapshot: List) { + val sorted = ArrayList(snapshot) + sorted.sortWith(::compare) + locked { + entries = sorted + populated = true + generation++ + cachedSnapshot = null + } + } + + /** + * Drops the index content and answers nothing until the next + * [rebuild]. Call from any store mutation whose displaced rows + * can't be itemized (delete-by-filter, expiration sweep, vanish). + */ + fun invalidate() { + locked { + entries = ArrayList() + populated = false + generation++ + cachedSnapshot = null + } + } + + /** Adds one entry. Duplicate `(createdAt, id)` pairs are ignored. */ + fun insert(entry: IdAndTime) { + locked { + if (!populated) return + val at = search(entry) + if (at >= 0) return + entries.add(-(at + 1), entry) + generation++ + cachedSnapshot = null + } + } + + /** Removes one entry; no-op when absent. */ + fun remove(entry: IdAndTime) { + locked { + if (!populated) return + val at = search(entry) + if (at < 0) return + entries.removeAt(at) + generation++ + cachedSnapshot = null + } + } + + /** + * A **sealed** negentropy storage of the current content, ready to + * back a [NegentropyServerSession]. Memoized per mutation + * generation: back-to-back NEG-OPENs with no writes in between (the + * mirror-heartbeat pattern) share one snapshot; after a write, only + * the first open pays the O(n) copy + seal of already-sorted data. + * + * Returns `null` when the index is not [isPopulated] (caller falls + * back to a scan — and should [rebuild] from it), or when the + * content exceeds [maxEntries] (caller answers NEG-ERR, the + * strfry-parity `"blocked: too many query results"`). + */ + fun sealedSnapshot(maxEntries: Int): IStorage? { + // Fast path reuses the memoized snapshot; the slow path copies + // the entries out under the lock and seals OUTSIDE it so a big + // seal doesn't stall the writer. + var toSeal: ArrayList? = null + var sealGeneration = 0L + locked { + if (!populated) return null + if (entries.size > maxEntries) return null + val cached = cachedSnapshot + if (cached != null && cachedGeneration == generation) return cached + toSeal = ArrayList(entries) + sealGeneration = generation + } + + val sealed = NegentropyServerSession.sealVector(toSeal!!) + locked { + // Last sealer wins; any competing snapshot of the same + // generation is equivalent content-wise. + if (populated && sealGeneration == generation) { + cachedSnapshot = sealed + cachedGeneration = sealGeneration + } + } + return sealed + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndexTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndexTest.kt new file mode 100644 index 0000000000..53135494cd --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip77Negentropy/LiveNegentropyIndexTest.kt @@ -0,0 +1,174 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip77Negentropy + +import com.vitorpamplona.negentropy.Negentropy +import com.vitorpamplona.negentropy.storage.IStorage +import com.vitorpamplona.quartz.nip01Core.store.IdAndTime +import com.vitorpamplona.quartz.utils.Hex +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNotNull +import kotlin.test.assertNotSame +import kotlin.test.assertNull +import kotlin.test.assertSame +import kotlin.test.assertTrue + +class LiveNegentropyIndexTest { + private fun id(seed: Int): String = seed.toString(16).padStart(64, '0') + + private fun entry( + time: Long, + seed: Int = time.toInt(), + ) = IdAndTime(time, id(seed)) + + /** The index content, read back through the sealed storage. */ + private fun IStorage.toEntries(): List = map { IdAndTime(it.timestamp, it.id.toHexString()) } + + private fun contentOf(index: LiveNegentropyIndex): List = assertNotNull(index.sealedSnapshot(Int.MAX_VALUE)).toEntries() + + @Test + fun answersNothingUntilRebuilt() { + val index = LiveNegentropyIndex() + assertNull(index.sealedSnapshot(1000)) + + // Mutations before the first rebuild are folded into the scan + // that populates it — they must not resurrect a dropped index. + index.insert(entry(1)) + assertNull(index.sealedSnapshot(1000)) + assertEquals(0, index.size()) + } + + @Test + fun rebuildSortsAndSnapshotMatches() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(5), entry(1), entry(3))) + + assertTrue(index.isPopulated()) + assertEquals(3, index.size()) + assertEquals(listOf(entry(1), entry(3), entry(5)), contentOf(index)) + } + + @Test + fun insertsKeepOrderIncludingBackfillAndTies() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(10), entry(30))) + + index.insert(entry(40)) // tail (the common, cheap case) + index.insert(entry(20)) // out-of-order backfill + index.insert(IdAndTime(10, id(99))) // same timestamp, distinct id + + assertEquals( + listOf(entry(10), IdAndTime(10, id(99)), entry(20), entry(30), entry(40)), + contentOf(index), + ) + } + + @Test + fun duplicateInsertIsIgnored() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(1))) + + index.insert(entry(1)) + + assertEquals(1, index.size()) + } + + @Test + fun removeDropsExactlyTheEntry() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(1), entry(2), entry(3))) + + index.remove(entry(2)) + index.remove(entry(7)) // absent: no-op + + assertEquals(listOf(entry(1), entry(3)), contentOf(index)) + } + + @Test + fun invalidateDropsEverythingUntilNextRebuild() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(1), entry(2))) + + index.invalidate() + + assertTrue(!index.isPopulated()) + assertNull(index.sealedSnapshot(1000)) + + index.rebuild(listOf(entry(9))) + assertEquals(listOf(entry(9)), contentOf(index)) + } + + @Test + fun snapshotIsMemoizedUntilTheNextMutation() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(1))) + + val first = index.sealedSnapshot(1000) + val second = index.sealedSnapshot(1000) + assertSame(first, second) + + index.insert(entry(2)) + val third = index.sealedSnapshot(1000) + assertNotSame(first, third) + } + + @Test + fun snapshotIsImmutableUnderLaterMutations() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(1), entry(2))) + + val snapshot = assertNotNull(index.sealedSnapshot(1000)) + index.insert(entry(3)) + index.remove(entry(1)) + + // The reconcile a peer started before those writes still sees + // the state it opened against. + assertEquals(listOf(entry(1), entry(2)), snapshot.toEntries()) + assertEquals(listOf(entry(2), entry(3)), contentOf(index)) + } + + @Test + fun overCapAnswersNullSoTheCallerSendsNegErr() { + val index = LiveNegentropyIndex() + index.rebuild(listOf(entry(1), entry(2), entry(3))) + + assertNull(index.sealedSnapshot(2)) + assertNotNull(index.sealedSnapshot(3)) + } + + @Test + fun snapshotDrivesARealReconcileSession() { + // End-to-end sanity: the sealed snapshot must be a valid server + // storage for an actual NegentropyServerSession handshake. + val index = LiveNegentropyIndex() + index.rebuild((1..50).map { entry(it.toLong()) }) + + val storage = assertNotNull(index.sealedSnapshot(1000)) + val server = NegentropyServerSession("sub", storage) + + // A client with an empty set opens the session; the server's + // first response must be a parseable NEG-MSG (non-null). + val clientInitial = Negentropy(NegentropyServerSession.sealVector(emptyList())).initiate() + val response = server.processMessage(Hex.encode(clientInitial)) + assertNotNull(response) + } +} From 73ee5cce994bb28199d29d277194239eabf602f8 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 23:05:23 +0000 Subject: [PATCH 07/21] perf(quartz/geode): serve full-set NEG-OPENs from the live negentropy index MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Milestones 2-3 of quartz/plans/2026-07-03-incremental-negentropy-storage.md. SQLiteEventStore now maintains the LiveNegentropyIndex when the strategy opts in (geode's does by default; [negentropy].live_index = false turns it off; app-side stores are untouched): - Write paths collect a LiveIndexDelta and apply it after COMMIT while still holding the writer mutex — rolled-back savepoint rows never reach the index and updates land in exact commit order (also vs the rebuild, which runs under the same mutex). - Replaceable/addressable overwrites report the row their BEFORE-INSERT trigger displaces via one indexed pre-SELECT that mirrors the trigger predicate (including the NIP-01 lowest-id tie-break and the d_tag-NULL case). - Paths that can't itemize (kind-5, vanish, delete-by-filter/id, expiration sweeps, clearDB) invalidate; the next NEG-OPEN rebuilds from one scan on the writer connection. - Until that first NEG-OPEN populates the index, ingest pays zero bookkeeping — the populated check happens under the writer mutex so it can't race the rebuild. LiveEventStore serves a single unconstrained filter (the relay-relay sync default; relayBench sends exactly this) from the index; everything else keeps the scan+seal path and its single-slot cache. Micro-benchmark at 50k events (LiveNegentropyBenchmark, in-container): scan+seal cold path 80-100 ms; index post-write open 9-16 ms (~5-10x). The relayBench A/B is the acceptance gate and comes next. Correctness: LiveNegentropyIndexStoreTest asserts index content == snapshotIdsForNegentropy scan after every mutation pattern (overwrites, losers, kind-5 rebuilds, filter deletes, mixed-outcome batches, transactions); the full geode suite (NIP-77 + interop sync tests) runs with the index on. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../kotlin/com/vitorpamplona/geode/Main.kt | 7 +- .../geode/RelayIndexingStrategy.kt | 37 +-- .../geode/config/StaticConfig.kt | 6 + .../relay/server/backend/LiveEventStore.kt | 9 + .../quartz/nip01Core/store/IEventStore.kt | 13 ++ .../nip01Core/store/ObservableEventStore.kt | 2 + .../nip01Core/store/sqlite/EventStore.kt | 2 + .../store/sqlite/IndexingStrategy.kt | 12 + .../store/sqlite/SQLiteEventStore.kt | 218 +++++++++++++++-- .../sqlite/LiveNegentropyIndexStoreTest.kt | 219 ++++++++++++++++++ .../prodbench/LiveNegentropyBenchmark.kt | 125 ++++++++++ 11 files changed, 620 insertions(+), 30 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt create mode 100644 quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/LiveNegentropyBenchmark.kt diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt index d096b34f8f..19e470853b 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt @@ -126,7 +126,12 @@ fun main(args: Array) { cliInfoFile?.let { RelayInfo.fromFile(it) } ?: config.resolveInfo(fullTextSearch) - val store: IEventStore = EventStore(dbName = dbFile, relay = advertisedUrl, indexStrategy = relayIndexingStrategy(fullTextSearch)) + val store: IEventStore = + EventStore( + dbName = dbFile, + relay = advertisedUrl, + indexStrategy = relayIndexingStrategy(fullTextSearch, config.negentropy.live_index), + ) val policyBuilder: () -> IRelayPolicy = { composePolicy(config, advertisedUrl, requireAuth, optionalAuth, verifySigs, parallelVerify) diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt index 81528e96aa..9b3a116f3f 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt @@ -40,21 +40,30 @@ import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy * per-event tokenization on ingest (relayBench measured it at roughly a * quarter of write cost) at the price of `search` filters matching * nothing — pair it with a NIP-11 doc that doesn't advertise 50. + * @param liveNegentropyIndex keep the always-current `(created_at, id)` + * set that serves full-corpus NIP-77 NEG-OPENs without a scan + seal + * (strfry answers those off its live tree; the scan+seal path measured + * ~340 ms per cold open at 50k events). Costs ~40 B/event of heap and + * one indexed pre-SELECT per replaceable insert. + * `[negentropy].live_index = false` turns it off. */ -fun relayIndexingStrategy(fullTextSearch: Boolean = true) = - DefaultIndexingStrategy( - indexEventsByCreatedAtAlone = true, - // Authors-only filters (no kinds) are relay-common — archives, - // migration tools. strfry maintains the same (pubkey, created_at) - // index unconditionally; without it the filter walks the whole - // time index. - indexEventsByPubkeyAlone = true, - indexFullTextSearch = fullTextSearch, - // Tokenize off the commit path; NostrServer drives the catch-up - // worker and search queries drain it first, so NIP-50 stays - // exactly as fresh while publishes stop paying for it. - deferFullTextSearchIndexing = fullTextSearch, - ) +fun relayIndexingStrategy( + fullTextSearch: Boolean = true, + liveNegentropyIndex: Boolean = true, +) = DefaultIndexingStrategy( + indexEventsByCreatedAtAlone = true, + // Authors-only filters (no kinds) are relay-common — archives, + // migration tools. strfry maintains the same (pubkey, created_at) + // index unconditionally; without it the filter walks the whole + // time index. + indexEventsByPubkeyAlone = true, + indexFullTextSearch = fullTextSearch, + // Tokenize off the commit path; NostrServer drives the catch-up + // worker and search queries drain it first, so NIP-50 stays + // exactly as fresh while publishes stop paying for it. + deferFullTextSearchIndexing = fullTextSearch, + maintainLiveNegentropyIndex = liveNegentropyIndex, +) /** Stock relay strategy — everything on, matching geode's defaults. */ val RelayIndexingStrategy = relayIndexingStrategy() diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt index 7191da08f3..7079a3fa28 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt @@ -153,6 +153,12 @@ data class StaticConfig( val frame_size_limit: Long = 500_000L, val max_sync_events: Int = 1_000_000, val max_sessions_per_connection: Int = 200, + /** + * Keep an always-current in-memory `(created_at, id)` set so + * full-corpus NEG-OPENs skip the table scan + seal (strfry + * parity). ~40 B per stored event of heap; on by default. + */ + val live_index: Boolean = true, ) data class AuthorizationSection( diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt index 6bed5eae8d..1d35362134 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt @@ -381,6 +381,15 @@ class LiveEventStore( filters: List, maxEntries: Int, ): IStorage? { + // Full-set NEG-OPENs — a single unconstrained filter, the shape + // relay-relay sync sends by default — are served from the store's + // always-current index when it maintains one: no scan, no + // O(n log n) seal, cold or not. `null` falls through to the scan + // path, which also owns the over-cap NEG-ERR detection. + if (filters.size == 1 && filters[0].isEmpty()) { + store.liveNegentropySnapshot(maxEntries)?.let { return it } + } + val generation = writeGeneration.load() val key = filters.joinToString("") { it.toJson() } + "cap=$maxEntries" val now = TimeUtils.now() diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/IEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/IEventStore.kt index da3dee4326..28122a586d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/IEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/IEventStore.kt @@ -20,6 +20,7 @@ */ package com.vitorpamplona.quartz.nip01Core.store +import com.vitorpamplona.negentropy.storage.IStorage import com.vitorpamplona.quartz.nip01Core.core.Event import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl @@ -152,6 +153,18 @@ interface IEventStore : AutoCloseable { } } + /** + * Sealed negentropy storage of the store's FULL set from an + * always-current in-memory index — the fast path for NEG-OPENs whose + * filter matches everything, skipping [snapshotIdsForNegentropy]'s + * scan and the O(n log n) seal. Returns `null` when the store keeps + * no such index (the default; see + * [com.vitorpamplona.quartz.nip01Core.store.sqlite.IndexingStrategy.maintainLiveNegentropyIndex]) + * or the set exceeds [maxEntries] — callers fall back to the scan + * path either way. + */ + suspend fun liveNegentropySnapshot(maxEntries: Int): IStorage? = null + /** * True when NIP-50 tokenization is deferred and something must drive * [ftsCatchUp] for search to see new events. The relay server checks diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/ObservableEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/ObservableEventStore.kt index 35e0427f25..ca0d372d6d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/ObservableEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/ObservableEventStore.kt @@ -194,6 +194,8 @@ class ObservableEventStore( maxEntries: Int?, ): List = inner.snapshotIdsForNegentropy(filters, maxEntries) + override suspend fun liveNegentropySnapshot(maxEntries: Int) = inner.liveNegentropySnapshot(maxEntries) + override suspend fun delete(filter: Filter) { inner.delete(filter) _changes.emit(StoreChange.DeleteByFilter(listOf(filter))) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt index cfad881a67..82d1df1c0b 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt @@ -76,6 +76,8 @@ class EventStore( maxEntries: Int?, ): List = store.snapshotIdsForNegentropy(filters, maxEntries) + override suspend fun liveNegentropySnapshot(maxEntries: Int) = store.liveNegentropySnapshot(maxEntries) + override val needsFtsCatchUp: Boolean get() = store.needsFtsCatchUp override suspend fun ftsCatchUp(batchSize: Int) = store.ftsCatchUp(batchSize) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt index fbcbda829e..aa7d38ca25 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt @@ -119,6 +119,17 @@ interface IndexingStrategy { */ val indexFullTextSearch: Boolean + /** + * Maintain an always-current in-memory `(created_at, id)` set (a + * [com.vitorpamplona.quartz.nip77Negentropy.LiveNegentropyIndex]) so + * NIP-77 NEG-OPENs over the full corpus skip the scan + O(n log n) + * seal — strfry answers those off its live tree. Costs ~40 B/event + * of heap plus one indexed pre-SELECT per replaceable insert, which + * only makes sense on a *relay*; client-side stores don't serve + * NEG-OPENs, so the default is **off**. + */ + val maintainLiveNegentropyIndex: Boolean get() = false + fun shouldIndex( kind: Int, tag: Tag, @@ -136,6 +147,7 @@ class DefaultIndexingStrategy( override val useAndIndexIdOnOrderBy: Boolean = false, override val indexFullTextSearch: Boolean = true, override val deferFullTextSearchIndexing: Boolean = false, + override val maintainLiveNegentropyIndex: Boolean = false, ) : IndexingStrategy { override fun shouldIndex( kind: Int, diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt index 1eaf481024..2a802b4c23 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt @@ -24,16 +24,23 @@ import androidx.sqlite.SQLiteConnection import androidx.sqlite.SQLiteDriver import androidx.sqlite.SQLiteException import androidx.sqlite.driver.bundled.BundledSQLiteDriver +import com.vitorpamplona.negentropy.storage.IStorage +import com.vitorpamplona.quartz.nip01Core.core.AddressableEvent import com.vitorpamplona.quartz.nip01Core.core.Event import com.vitorpamplona.quartz.nip01Core.core.HexKey +import com.vitorpamplona.quartz.nip01Core.core.isAddressable import com.vitorpamplona.quartz.nip01Core.core.isEphemeral +import com.vitorpamplona.quartz.nip01Core.core.isReplaceable import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.store.FtsReindexProgress import com.vitorpamplona.quartz.nip01Core.store.IEventStore import com.vitorpamplona.quartz.nip01Core.store.IdAndTime import com.vitorpamplona.quartz.nip01Core.store.RawEvent +import com.vitorpamplona.quartz.nip09Deletions.DeletionEvent import com.vitorpamplona.quartz.nip40Expiration.isExpired +import com.vitorpamplona.quartz.nip62RequestToVanish.RequestToVanishEvent +import com.vitorpamplona.quartz.nip77Negentropy.LiveNegentropyIndex class SQLiteEventStore( val driver: SQLiteDriver = BundledSQLiteDriver(), @@ -74,6 +81,16 @@ class SQLiteEventStore( indexStrategy, ) + /** + * Always-current `(created_at, id)` set for NIP-77 (see + * [IndexingStrategy.maintainLiveNegentropyIndex]). Populated lazily + * by the first [liveNegentropySnapshot]; kept current by the write + * paths through a [LiveIndexDelta] applied only after each + * transaction commits, so a rolled-back row never leaks into it. + */ + val liveNegentropyIndex: LiveNegentropyIndex? = + if (indexStrategy.maintainLiveNegentropyIndex) LiveNegentropyIndex() else null + val modules = listOf( seedModule, @@ -182,10 +199,12 @@ class SQLiteEventStore( } } - suspend fun clearDB() = + suspend fun clearDB() { pool.useWriter { db -> modules.reversed().forEach { it.deleteAll(db) } } + liveNegentropyIndex?.invalidate() + } suspend fun vacuum() = pool.useWriter { db -> @@ -225,15 +244,127 @@ class SQLiteEventStore( db.execSQL("ANALYZE") } + /** + * Live-index mutations gathered during one write transaction and + * applied only after its COMMIT — a rolled-back row (savepoint or + * whole-transaction failure) must never reach the index, and the + * index must never advertise an id before it is durable. `null` + * when [liveNegentropyIndex] is off: zero cost on the write path. + */ + internal class LiveIndexDelta { + val added = ArrayList() + val removed = ArrayList() + + /** + * Set when a row's side effects delete OTHER rows in ways this + * delta can't itemize (kind-5 targets, vanish-by-pubkey). The + * whole index is dropped and lazily rebuilt from one scan. + */ + var invalidateAll = false + } + + /** + * Starts delta tracking for a write transaction, or `null` when there + * is nothing to track — index off, or not (yet) populated. An + * unpopulated index is covered by its next [LiveNegentropyIndex.rebuild] + * scan instead, so ingest pays zero bookkeeping until the first + * NEG-OPEN actually builds the index. + * + * MUST be called while already holding the writer connection: the + * populated check races [liveNegentropySnapshot]'s rebuild otherwise + * (rebuild flips `populated` under the writer mutex — deciding out + * here could skip tracking for a transaction that commits after the + * rebuild's scan, silently losing its rows). + */ + private fun newDeltaOrNull(): LiveIndexDelta? = if (liveNegentropyIndex?.isPopulated() == true) LiveIndexDelta() else null + + /** Records one successfully inserted row into the pending delta. */ + private fun LiveIndexDelta.recordAccepted( + event: Event, + displaced: IdAndTime?, + ) { + if (event is DeletionEvent || (event is RequestToVanishEvent && event.shouldVanishFrom(relay))) { + // Their own row lands too, but the rebuild scan picks it up. + invalidateAll = true + return + } + displaced?.let { removed += it } + added += IdAndTime(event.createdAt, event.id) + } + + private fun applyAfterCommit(delta: LiveIndexDelta?) { + val index = liveNegentropyIndex ?: return + if (delta == null) return + if (delta.invalidateAll) { + index.invalidate() + return + } + delta.removed.forEach(index::remove) + delta.added.forEach(index::insert) + } + + /** + * The row the replaceable/addressable BEFORE-INSERT trigger is about + * to delete for [event], or `null` when nothing will be displaced. + * Must run *before* the insert (the trigger fires during it) and + * mirrors the trigger predicates exactly — including the NIP-01 + * lowest-id-wins tie-break. At most one row can match thanks to the + * partial unique indexes, and the same indexes make this lookup + * O(log n). Only called when the live index is on. + */ + private fun displacedBy( + event: Event, + db: SQLiteConnection, + ): IdAndTime? { + // The addressable branch requires the parsed class to carry a + // d-tag, mirroring the header insert: events without one store + // d_tag NULL, and the trigger's `d_tag = NEW.d_tag` never + // matches NULL — so nothing gets displaced. + val addressable = event.kind.isAddressable() && event is AddressableEvent + val sql = + when { + event.kind.isReplaceable() -> + """ + SELECT created_at, id FROM event_headers + WHERE kind = ? AND pubkey = ? + AND (created_at < ? OR (created_at = ? AND id > ?)) + """.trimIndent() + + addressable -> + """ + SELECT created_at, id FROM event_headers + WHERE kind = ? AND pubkey = ? AND d_tag = ? + AND kind >= 30000 AND kind < 40000 + AND (created_at < ? OR (created_at = ? AND id > ?)) + """.trimIndent() + + else -> return null + } + db.prepare(sql).use { stmt -> + var i = 1 + stmt.bindLong(i++, event.kind.toLong()) + stmt.bindText(i++, event.pubKey) + if (addressable) stmt.bindText(i++, (event as AddressableEvent).dTag()) + stmt.bindLong(i++, event.createdAt) + stmt.bindLong(i++, event.createdAt) + stmt.bindText(i, event.id) + if (!stmt.step()) return null + return IdAndTime(stmt.getLong(0), stmt.getText(1)) + } + } + private fun innerInsertEvent( event: Event, db: SQLiteConnection, + delta: LiveIndexDelta? = null, ) { + val displaced = if (delta != null) displacedBy(event, db) else null val headerId = eventIndexModule.insert(event, db) deletionModule.insert(event, db) expirationModule.insert(event, headerId, db) fullTextSearchModule.insert(event, headerId, db) rightToVanishModule.insert(event, relay, headerId, db) + delta?.recordAccepted(event, displaced) } suspend fun insertEvent(event: Event) { @@ -241,9 +372,15 @@ class SQLiteEventStore( if (event.kind.isEphemeral()) return pool.useWriter { db -> + val delta = newDeltaOrNull() db.transaction { - innerInsertEvent(event, this) + innerInsertEvent(event, this, delta) } + // Still holding the writer mutex: the transaction above has + // committed, and applying here keeps index updates in exact + // commit order across every writer (and the rebuild, which + // also runs under this mutex). + applyAfterCommit(delta) } } @@ -268,11 +405,14 @@ class SQLiteEventStore( if (events.isEmpty()) return emptyList() val outcomes = ArrayList(events.size) pool.useWriter { db -> + val delta = newDeltaOrNull() db.transaction { events.forEachIndexed { i, event -> - outcomes.add(insertWithSavepoint(event, i, this)) + outcomes.add(insertWithSavepoint(event, i, this, delta)) } } + // Post-commit, pre-mutex-release: see insertEvent. + applyAfterCommit(delta) } return outcomes } @@ -281,6 +421,7 @@ class SQLiteEventStore( event: Event, index: Int, db: SQLiteConnection, + delta: LiveIndexDelta?, ): IEventStore.InsertOutcome { if (event.isExpired()) { return IEventStore.InsertOutcome.Rejected("blocked: Cannot insert an expired event") @@ -290,7 +431,10 @@ class SQLiteEventStore( val sp = "ev$index" db.execSQL("SAVEPOINT $sp") return try { - innerInsertEvent(event, db) + // The delta records this row only after every module insert + // succeeded (last step of innerInsertEvent), so a savepoint + // rollback below leaves the pending delta untouched. + innerInsertEvent(event, db, delta) db.execSQL("RELEASE SAVEPOINT $sp") IEventStore.InsertOutcome.Accepted } catch (e: Throwable) { @@ -304,24 +448,28 @@ class SQLiteEventStore( } } - inner class Transaction( + inner class Transaction internal constructor( val db: SQLiteConnection, + private val delta: LiveIndexDelta?, ) : IEventStore.ITransaction { override fun insert(event: Event) { if (event.isExpired()) throw SQLiteException("blocked: Cannot insert an expired event") if (event.kind.isEphemeral()) return - innerInsertEvent(event, db) + innerInsertEvent(event, db, delta) } } suspend fun transaction(body: Transaction.() -> Unit) { pool.useWriter { db -> + val delta = newDeltaOrNull() db.transaction { - with(Transaction(this)) { + with(Transaction(this, delta)) { body() } } + // Post-commit, pre-mutex-release: see insertEvent. + applyAfterCommit(delta) } } @@ -366,17 +514,57 @@ class SQLiteEventStore( maxEntries: Int? = null, ): List = pool.useReader { queryBuilder.snapshotIdsForNegentropy(filters, it, maxEntries) } - suspend fun delete(filter: Filter) = pool.useWriter { queryBuilder.delete(filter, it) } + suspend fun delete(filter: Filter) { + pool.useWriter { queryBuilder.delete(filter, it) } + // Delete-by-filter can't itemize what it removed; drop the live + // index and let the next NEG-OPEN rebuild from one scan. + liveNegentropyIndex?.invalidate() + } - suspend fun delete(filters: List) = pool.useWriter { queryBuilder.delete(filters, it) } + suspend fun delete(filters: List) { + pool.useWriter { queryBuilder.delete(filters, it) } + liveNegentropyIndex?.invalidate() + } - suspend fun delete(id: HexKey): Int = - pool.useWriter { db -> - db.execSQL("DELETE FROM event_headers WHERE id = ?", arrayOf(id)) - db.changes() + suspend fun delete(id: HexKey): Int { + val changes = + pool.useWriter { db -> + db.execSQL("DELETE FROM event_headers WHERE id = ?", arrayOf(id)) + db.changes() + } + if (changes > 0) liveNegentropyIndex?.invalidate() + return changes + } + + suspend fun deleteExpiredEvents() { + val swept = + pool.useWriter { db -> + expirationModule.deleteExpiredEvents(db) + db.changes() + } + if (swept > 0) liveNegentropyIndex?.invalidate() + } + + /** + * Sealed live-index snapshot of the FULL stored set for NIP-77, or + * `null` when the live index is off (strategy default) or the set + * exceeds [maxEntries] — callers fall back to the scan path either + * way. The first call after boot or an invalidation rebuilds the + * index from one scan **on the writer connection**, so no insert can + * commit between the scan and the rebuild — after that, the write + * paths keep it current and this is a cached-snapshot lookup. + */ + suspend fun liveNegentropySnapshot(maxEntries: Int): IStorage? { + val index = liveNegentropyIndex ?: return null + if (!index.isPopulated()) { + pool.useWriter { db -> + if (!index.isPopulated()) { + index.rebuild(queryBuilder.snapshotIdsForNegentropy(listOf(Filter()), db, null)) + } + } } - - suspend fun deleteExpiredEvents() = pool.useWriter { expirationModule.deleteExpiredEvents(it) } + return index.sealedSnapshot(maxEntries) + } /** * Wipe and rebuild the NIP-50 full-text search index for every diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt new file mode 100644 index 0000000000..3102f162de --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt @@ -0,0 +1,219 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.store.sqlite + +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.store.IEventStore +import com.vitorpamplona.quartz.nip01Core.store.IdAndTime +import com.vitorpamplona.quartz.utils.EventFactory +import kotlinx.coroutines.test.runTest +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNotNull +import kotlin.test.assertNull +import kotlin.test.assertTrue + +/** + * The live negentropy index's one correctness rule: at every point, a + * snapshot must advertise **exactly** the `(created_at, id)` set a fresh + * scan of the store would — never a deleted/displaced id (peers would + * fetch dead events), never missing a committed one. Every mutation + * pattern the write path knows is run against a store with the index on + * and compared to `snapshotIdsForNegentropy`. + */ +class LiveNegentropyIndexStoreTest { + private val strategy = DefaultIndexingStrategy(maintainLiveNegentropyIndex = true) + + private fun hexId(seed: Int): String = seed.toString().padStart(64, '0') + + private fun pubkey(seed: Int): String = seed.toString().padStart(64, 'a') + + private val sig = "0".repeat(128) + + private fun event( + idSeed: Int, + kind: Int = 1, + createdAt: Long = idSeed.toLong(), + pubKey: String = pubkey(1), + tags: Array> = emptyArray(), + content: String = "", + ): Event = EventFactory.create(hexId(idSeed), pubKey, createdAt, kind, tags, content, sig) + + private fun newStore() = EventStore(dbName = null, indexStrategy = strategy) + + /** Index content and scan content must be identical, entry for entry. */ + private suspend fun assertIndexMatchesScan(store: EventStore) { + val snapshot = assertNotNull(store.liveNegentropySnapshot(Int.MAX_VALUE), "index should serve a snapshot") + val fromIndex = snapshot.map { IdAndTime(it.timestamp, it.id.toHexString()) } + val fromScan = + store + .snapshotIdsForNegentropy(listOf(Filter()), null) + .sortedWith(compareBy({ it.createdAt }, { it.id })) + assertEquals(fromScan, fromIndex) + } + + @Test + fun populatesLazilyAndTracksPlainInserts() = + runTest { + val store = newStore() + store.insert(event(1)) + store.insert(event(2)) + + // First snapshot rebuilds from a scan… + assertIndexMatchesScan(store) + + // …and later inserts are tracked incrementally. + store.insert(event(3)) + store.batchInsert(listOf(event(4), event(5))) + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun replaceableOverwriteDropsTheDisplacedId() = + runTest { + val store = newStore() + store.insert(event(1, kind = 0, createdAt = 100, pubKey = pubkey(7))) + assertIndexMatchesScan(store) + + // Newer kind-0 from the same author displaces the old row. + store.insert(event(2, kind = 0, createdAt = 200, pubKey = pubkey(7))) + assertIndexMatchesScan(store) + assertEquals(1, store.count(Filter(kinds = listOf(0)))) + + // An older one loses to the stored row and must not disturb + // the index (the store rejects it). + runCatching { store.insert(event(3, kind = 0, createdAt = 50, pubKey = pubkey(7))) } + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun addressableOverwriteDropsOnlyTheSameDTag() = + runTest { + val store = newStore() + val author = pubkey(9) + store.insert(event(1, kind = 30023, createdAt = 100, pubKey = author, tags = arrayOf(arrayOf("d", "post-a")))) + store.insert(event(2, kind = 30023, createdAt = 100, pubKey = author, tags = arrayOf(arrayOf("d", "post-b")))) + assertIndexMatchesScan(store) + + // Overwrites post-a; post-b must survive. + store.insert(event(3, kind = 30023, createdAt = 200, pubKey = author, tags = arrayOf(arrayOf("d", "post-a")))) + assertIndexMatchesScan(store) + assertEquals(2, store.count(Filter(kinds = listOf(30023)))) + + store.close() + } + + @Test + fun deletionEventInvalidatesAndRebuildMatches() = + runTest { + val store = newStore() + val author = pubkey(3) + val target = event(1, createdAt = 100, pubKey = author) + store.insert(target) + store.insert(event(2, createdAt = 110, pubKey = author)) + assertIndexMatchesScan(store) + + // Kind-5 removes the target AND stores itself; the index + // can't itemize that, so it must rebuild — and match. + store.insert( + event(4, kind = 5, createdAt = 120, pubKey = author, tags = arrayOf(arrayOf("e", target.id))), + ) + assertIndexMatchesScan(store) + assertTrue(store.query(Filter(ids = listOf(target.id))).isEmpty()) + + store.close() + } + + @Test + fun deleteByFilterInvalidatesAndRebuildMatches() = + runTest { + val store = newStore() + store.insert(event(1, kind = 1)) + store.insert(event(2, kind = 7)) + assertIndexMatchesScan(store) + + store.delete(Filter(kinds = listOf(7))) + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun rejectedRowsInABatchNeverEnterTheIndex() = + runTest { + val store = newStore() + store.insert(event(1)) + assertIndexMatchesScan(store) + + // Duplicate id (rejected) mixed with a fresh row (accepted). + val outcomes = store.batchInsert(listOf(event(1), event(2))) + assertTrue(outcomes[0] is IEventStore.InsertOutcome.Rejected) + assertEquals(IEventStore.InsertOutcome.Accepted, outcomes[1]) + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun transactionInsertsLandAfterCommit() = + runTest { + val store = newStore() + store.insert(event(1)) + assertIndexMatchesScan(store) + + store.transaction { + insert(event(2)) + insert(event(3)) + } + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun overCapFallsBackToNull() = + runTest { + val store = newStore() + store.insert(event(1)) + store.insert(event(2)) + store.insert(event(3)) + assertIndexMatchesScan(store) + + assertNull(store.liveNegentropySnapshot(2)) + assertNotNull(store.liveNegentropySnapshot(3)) + + store.close() + } + + @Test + fun defaultStrategyKeepsNoIndex() = + runTest { + val store = EventStore(dbName = null) + store.insert(event(1)) + assertNull(store.liveNegentropySnapshot(Int.MAX_VALUE)) + store.close() + } +} diff --git a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/LiveNegentropyBenchmark.kt b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/LiveNegentropyBenchmark.kt new file mode 100644 index 0000000000..5abfc1672b --- /dev/null +++ b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/LiveNegentropyBenchmark.kt @@ -0,0 +1,125 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.prodbench + +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.nip77Negentropy.NegentropyServerSession +import com.vitorpamplona.quartz.utils.EventFactory +import kotlinx.coroutines.runBlocking +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNotNull + +/** + * Measures what the live negentropy index buys on a NEG-OPEN against the + * scan + O(n log n) seal it replaces, at the relayBench corpus size where + * the cold path measured ~340 ms (50k events; strfry serves ~21 ms). + * + * Three numbers, printed for the record: + * 1. scan+seal — the pre-index cold path (`snapshotIdsForNegentropy` + + * `sealVector`), what every open pays when the single-slot cache + * misses. + * 2. index first-open — includes the one-time lazy rebuild scan. + * 3. index post-write open — the steady state on a busy relay: a write + * lands (invalidating any memoized snapshot), then a NEG-OPEN + * arrives. This is the number that was previously ~scan+seal and is + * the point of the feature. + * + * No speed assertion — container noise makes ratios flaky; correctness + * (identical content between both paths) is asserted instead. Timings + * are informational and the real A/B happens in relayBench. + */ +class LiveNegentropyBenchmark { + companion object { + const val EVENTS = 50_000 + const val POST_WRITE_OPENS = 50 + } + + private fun hexId(seed: Int): String = seed.toString(16).padStart(64, '0') + + private fun pubkey(seed: Int): String = (seed % 512).toString(16).padStart(64, 'a') + + private val sig = "0".repeat(128) + + private fun event(seed: Int): Event = + EventFactory.create( + id = hexId(seed), + pubKey = pubkey(seed), + createdAt = 1_600_000_000L + (seed * 7919) % 1_000_000, // scattered, not pre-sorted + kind = 1, + tags = emptyArray(), + content = "live negentropy benchmark $seed", + sig = sig, + ) + + @Test + fun coldOpenVsLiveIndexAt50k() = + runBlocking { + val store = EventStore(dbName = null, indexStrategy = DefaultIndexingStrategy(maintainLiveNegentropyIndex = true)) + + (1..EVENTS).chunked(2000).forEach { chunk -> + store.batchInsert(chunk.map { event(it) }) + } + + // 1. The pre-index cold path. + var scanSealNanos = 0L + var scanned = 0 + run { + val t0 = System.nanoTime() + val entries = store.snapshotIdsForNegentropy(listOf(Filter()), null) + val sealed = NegentropyServerSession.sealVector(entries) + scanSealNanos = System.nanoTime() - t0 + scanned = sealed.size() + } + + // 2. First index open: pays the lazy rebuild once. + val t1 = System.nanoTime() + val first = assertNotNull(store.liveNegentropySnapshot(Int.MAX_VALUE)) + val firstOpenNanos = System.nanoTime() - t1 + assertEquals(scanned, first.size()) + + // 3. Steady state: one write per open, so every open pays the + // index's real per-open cost (copy + seal of sorted data), + // never the memoized-snapshot freebie. + var postWriteNanos = 0L + for (i in 0 until POST_WRITE_OPENS) { + store.insert(event(EVENTS + 1 + i)) + val t2 = System.nanoTime() + assertNotNull(store.liveNegentropySnapshot(Int.MAX_VALUE)) + postWriteNanos += System.nanoTime() - t2 + } + val postWriteAvgMs = postWriteNanos / POST_WRITE_OPENS / 1e6 + + // Correctness cross-check after all the churn. + val finalSnapshot = assertNotNull(store.liveNegentropySnapshot(Int.MAX_VALUE)) + assertEquals(EVENTS + POST_WRITE_OPENS, finalSnapshot.size()) + + println("LiveNegentropyBenchmark @ ${EVENTS / 1000}k events") + println(" scan+seal cold path: ${"%8.2f".format(scanSealNanos / 1e6)} ms") + println(" index first open: ${"%8.2f".format(firstOpenNanos / 1e6)} ms (includes one-time rebuild)") + println(" index post-write open: ${"%8.2f".format(postWriteAvgMs)} ms (avg of $POST_WRITE_OPENS)") + + store.close() + } +} From de388907c2d80d3ea16ac8e1b7c0c8a3d1b1bcdf Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 23:09:56 +0000 Subject: [PATCH 08/21] =?UTF-8?q?docs(quartz):=20record=20live-negentropy-?= =?UTF-8?q?index=20A/B=20results=20=E2=80=94=20shipped?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two relayBench runs (50k corpus, both geode variants side by side, then order-reversed): identical-set reconcile 56/57 ms with the index vs 110/130 ms without in both orders (~2.2x, the post-write NEG-OPEN the old cache always missed); ingest and first-ever reconcile deltas flip with relay order, i.e. run-order noise — no regression. Also records the run-order-bias protocol note for future A/Bs. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- ...26-07-03-incremental-negentropy-storage.md | 25 +++++++++++++++++-- 1 file changed, 23 insertions(+), 2 deletions(-) diff --git a/quartz/plans/2026-07-03-incremental-negentropy-storage.md b/quartz/plans/2026-07-03-incremental-negentropy-storage.md index 6121a3c7c6..9e4a2ef643 100644 --- a/quartz/plans/2026-07-03-incremental-negentropy-storage.md +++ b/quartz/plans/2026-07-03-incremental-negentropy-storage.md @@ -1,7 +1,28 @@ # Incremental live negentropy storage -**Status: planned** — backlog item 3 of the relay performance campaign -(PR #3466 follow-up). +**Status: shipped** — backlog item 3 of the relay performance campaign +(PR #3466 follow-up). All four milestones landed on this branch. + +**Measured results.** Micro (LiveNegentropyBenchmark, 50k events, +in-container): scan+seal cold path 80–100 ms per open; index post-write +open 9–16 ms (~5–10×); first open ~70 ms (pays the one-time rebuild). +relayBench A/B (both variants in ONE run — geode with index vs geode +with `[negentropy].live_index = false` — 50k corpus, then repeated with +the relay order reversed to control for run-order bias): + + - **identical-set reconcile: 56/57 ms with the index vs 110/130 ms + without, in both orders — the real ~2.2× win.** This phase runs + after the delta transfer wrote events, so the old single-slot cache + always misses — the exact gap the index closes. + - initial (first-ever) reconcile and ingest throughput: whichever + relay ran first won/lost respectively in BOTH runs — pure run-order + effect, i.e. no ingest regression and (as designed) no win on the + very first open, which pays the lazy rebuild. + +Measurement note for future A/Bs: putting both variants in one +relayBench run controls container noise better than alternating runs, +but per-phase run-order bias is real (~100 ms on initial reconcile, +~10% on ingest) — always repeat with the order reversed. ## Problem From 0d6a9b1e4d4d8da13561ac0993dc6cebcacd3903 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 23:15:09 +0000 Subject: [PATCH 09/21] docs(quartz): record strfry head-to-head for the live negentropy index MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Same container, 50k corpus, geode --no-search, strfry built from source: identical-set reconcile 41.4 ms (geode) vs 30.1 ms (strfry) — down to ~1.4x from the campaign-opening full-scan-per-open; cold reconcile 177 vs 112 ms (geode's first open pays the lazy rebuild); ingest at parity in this container; storage 72.8 vs 106 MiB. Notes the zero-copy IStorage follow-up that would close the remaining ~11 ms. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- ...026-07-03-incremental-negentropy-storage.md | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/quartz/plans/2026-07-03-incremental-negentropy-storage.md b/quartz/plans/2026-07-03-incremental-negentropy-storage.md index 9e4a2ef643..ca10f6b091 100644 --- a/quartz/plans/2026-07-03-incremental-negentropy-storage.md +++ b/quartz/plans/2026-07-03-incremental-negentropy-storage.md @@ -24,6 +24,24 @@ relayBench run controls container noise better than alternating runs, but per-phase run-order bias is real (~100 ms on initial reconcile, ~10% on ingest) — always repeat with the order reversed. +**strfry head-to-head** (same container, 50k corpus, `--geode-no-search`, +strfry built from source, single run): identical-set reconcile geode +41.4 ms vs strfry 30.1 ms — the always-current-tree gap is down to +~1.4× from the campaign-opening "full scan + seal per open". Initial +(cold) reconcile 177 vs 112 ms: strfry keeps its tree across the whole +run while geode's first open pays the lazy rebuild. Ingest measured at +parity in this container (7,972 vs 7,860 ev/s — treat as ±noise, the +earlier 1.59× gap was measured on different hardware); storage 72.8 vs +106.0 MiB. + +**Possible follow-up** if the residual matters: the per-open cost is now +the O(n) copy + re-seal into a `StorageVector` (~16 ms at 50k). A custom +immutable `IStorage` view over the live entries (copy-on-write chunks, +no re-seal) would make an open O(1), matching strfry's zero-copy read of +its tree — that is the "interesting part" the original backlog item +anticipated, deferred until a benchmark says the remaining ~11 ms is +worth it. + ## Problem A cold NEG-OPEN pays a full scan + O(n log n) seal: `snapshotIdsForNegentropy` From fb29d655fb3338582b0808a3671984b6964cc90c Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 23:54:19 +0000 Subject: [PATCH 10/21] perf(quartz): inline fast path for bounded small REQs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Backlog item 2 (small-REQ dispatch floor), first cut. relayBench at 50k shows geode ~2.5x slower than strfry when a REQ returns ~20 events (author-archive 1.18 vs 0.48 ms EOSE p50) while WINNING the 500-event scenarios — with tiny results the fixed per-REQ cost dominates. A new stage benchmark (SmallReqFloorBenchmark) decomposes the floor: at ~21 rows, raw SQL is 0.18 ms and the remaining ~0.42 ms is live-subscription machinery plus per-REQ coroutine dispatch. A REQ whose filters all carry a limit summing to <= 256 now runs its stored replay inline on the receive coroutine and only retains a live-tail handle (SessionBackend.queryRawInline; LiveEventStore's queryRaw is re-expressed on the same core) — no per-REQ launch, no Job, no dispatcher handoffs, no awaitCancellation scaffolding. Unbounded REQs keep the launched path so CLOSE can always interrupt a giant replay. In-process time-to-EOSE for ~21-row REQs drops 0.60 -> 0.50 ms; the bigger effect expected under concurrency (no per-REQ Job+dispatch churn) is relayBench's to judge. InlineReqFastPathTest pins wire-behavior parity: stored-then-EOSE ordering, live tail delivery and CLOSE detachment, same-subId replacement, launched fallback for unbounded and over-cap REQs. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../nip01Core/relay/server/RelaySession.kt | 121 +++++++++--- .../relay/server/backend/LiveEventStore.kt | 32 +++- .../relay/server/backend/SessionBackend.kt | 36 ++++ .../relay/server/InlineReqFastPathTest.kt | 169 +++++++++++++++++ .../relay/prodbench/SmallReqFloorBenchmark.kt | 176 ++++++++++++++++++ 5 files changed, 508 insertions(+), 26 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt create mode 100644 quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/SmallReqFloorBenchmark.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt index 05cec0fad3..0ac21a81b5 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt @@ -35,6 +35,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.CloseCmd import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.CountCmd import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.EventCmd import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.ReqCmd +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.server.backend.RequestContext import com.vitorpamplona.quartz.nip01Core.relay.server.backend.SessionBackend import com.vitorpamplona.quartz.nip01Core.relay.server.policies.IRelayPolicy @@ -49,7 +50,6 @@ import com.vitorpamplona.quartz.utils.Log import com.vitorpamplona.quartz.utils.cache.LargeCache import kotlinx.coroutines.CancellationException import kotlinx.coroutines.CoroutineScope -import kotlinx.coroutines.Job import kotlinx.coroutines.channels.ClosedSendChannelException import kotlinx.coroutines.launch import kotlin.concurrent.atomics.AtomicLong @@ -75,7 +75,16 @@ class RelaySession( */ val id: Long = nextConnectionId(), ) : AutoCloseable { - private val subscriptions = LargeCache() + /** + * One active REQ. Launched subscriptions cancel their coroutine; + * inline subscriptions (the bounded small-REQ fast path, which never + * had a coroutine) close their live-tail registration directly. + */ + private fun interface ActiveSubscription { + fun cancel() + } + + private val subscriptions = LargeCache() /** * The authenticated-identity store for this connection. The engine is the @@ -103,8 +112,8 @@ class RelaySession( private fun addSubscription( subId: String, - job: Job, - ) = subscriptions.put(subId, job) + sub: ActiveSubscription, + ) = subscriptions.put(subId, sub) private fun cancelSubscription(subId: String): Boolean = subscriptions.remove(subId)?.let { @@ -113,7 +122,7 @@ class RelaySession( } ?: false fun cancelAllSubscriptions() { - subscriptions.forEach { _, job -> job.cancel() } + subscriptions.forEach { _, sub -> sub.cancel() } subscriptions.clear() negentropy.clear() } @@ -263,7 +272,28 @@ class RelaySession( } // -- NIP-01: REQ ---------------------------------------------------------- - private fun handleReq(cmd: ReqCmd) { + + /** + * A REQ qualifies for the inline fast path when its replay is + * provably small: every filter carries a `limit` and the limits sum + * to at most [INLINE_REPLAY_MAX_ROWS]. Inline replays run on this + * connection's receive coroutine — no per-REQ launch, no dispatcher + * handoffs (profiled at ~2/3 of a ~20-row REQ's time-to-EOSE) — at + * the price of delaying the connection's NEXT command until EOSE, + * which a bounded replay keeps in the tens-of-rows range. Unbounded + * REQs keep the launched path so a CLOSE can always interrupt them. + */ + private fun isInlineEligible(filters: List): Boolean { + var total = 0 + for (f in filters) { + val limit = f.limit ?: return false + total += limit + if (total > INLINE_REPLAY_MAX_ROWS) return false + } + return true + } + + private suspend fun handleReq(cmd: ReqCmd) { // Ask the policy whether a *new* subscription may open (e.g. a // max_subscriptions cap). A re-REQ on an existing id replaces it // 1-for-1 and doesn't grow the count, so it's exempt. Checked before @@ -288,6 +318,54 @@ class RelaySession( // Policy may rewrite filters to match the user's access level. val filters = (result as PolicyResult.Accepted).cmd.filters + // The `["EVENT","",` prefix is built once per + // subscription, not per row (zero-decode paths only). + fun framePrefix() = + buildString { + append("[\"EVENT\",") + RawEvent.appendJsonQuoted(this, cmd.subId) + append(',') + } + + fun sendStored( + framePrefix: String, + raw: RawEvent, + ) = sendRaw( + buildString(framePrefix.length + raw.jsonTags.length + raw.content.length + 256) { + append(framePrefix) + raw.appendJsonObjectTo(this) + append(']') + }, + ) + + // Small-REQ fast path: a provably bounded replay runs inline on + // this receive coroutine — no launch, no dispatcher handoffs — + // and only the live-tail handle is retained. See + // [SessionBackend.queryRawInline] for the contract. + if (!policy.filtersOutgoingEvents && isInlineEligible(filters)) { + val handle = + try { + val prefix = framePrefix() + store.queryRawInline( + ctx = requestContext, + filters = filters, + onEachStored = { raw -> sendStored(prefix, raw) }, + onEachLive = { event -> send(EventMessage(cmd.subId, event)) }, + onEose = { send(EoseMessage(cmd.subId)) }, + ) + } catch (e: CancellationException) { + throw e + } catch (e: Exception) { + send(ClosedMessage.of(cmd.subId, MachineReadablePrefix.ERROR, e.message ?: "query failed")) + return + } + if (handle != null) { + addSubscription(cmd.subId) { handle.close() } + return + } + // Backend without an inline path: fall through to launch. + } + val job = scope.launch { try { @@ -308,26 +386,11 @@ class RelaySession( // Zero-decode path: the stored replay splices raw // storage strings straight into wire frames — no tags // parse, no Event materialization, no re-serialize. - // The `["EVENT","",` prefix is built once per - // subscription, not per row. - val framePrefix = - buildString { - append("[\"EVENT\",") - RawEvent.appendJsonQuoted(this, cmd.subId) - append(',') - } + val prefix = framePrefix() store.queryRaw( ctx = requestContext, filters = filters, - onEachStored = { raw -> - sendRaw( - buildString(framePrefix.length + raw.jsonTags.length + raw.content.length + 256) { - append(framePrefix) - raw.appendJsonObjectTo(this) - append(']') - }, - ) - }, + onEachStored = { raw -> sendStored(prefix, raw) }, onEachLive = { event -> send(EventMessage(cmd.subId, event)) }, onEose = { send(EoseMessage(cmd.subId)) }, ) @@ -343,7 +406,7 @@ class RelaySession( } } - addSubscription(cmd.subId, job) + addSubscription(cmd.subId) { job.cancel() } } // -- NIP-01: CLOSE -------------------------------------------------------- @@ -363,5 +426,15 @@ class RelaySession( /** Allocates the next process-unique connection id. */ fun nextConnectionId(): Long = connectionIdSeq.fetchAndAdd(1L) + + /** + * Ceiling on the summed filter limits for the inline REQ fast + * path. Sized so the worst-case inline replay stays around a + * millisecond (~20-row REQs profile at ~0.2 ms of storage work) + * — small enough that delaying the connection's next command is + * unnoticeable, large enough to cover the profile/thread/ + * notifications-style lookups clients hammer relays with. + */ + const val INLINE_REPLAY_MAX_ROWS = 256 } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt index 1d35362134..a397156143 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt @@ -258,6 +258,31 @@ class LiveEventStore( onEachLive: (Event) -> Unit, onEose: () -> Unit, ) { + val handle = queryRawInline(ctx, filters, onEachStored, onEachLive, onEose) + try { + // Suspend until the caller's coroutine is cancelled (NIP-01 + // CLOSE or connection drop); the live tail runs meanwhile. + awaitCancellation() + } finally { + handle.close() + } + } + + /** + * The registration + replay + EOSE core of [queryRaw], run entirely + * on the calling coroutine. Small bounded REQs take this directly + * (see [SessionBackend.queryRawInline]) and skip the per-REQ + * coroutine; [queryRaw] wraps it with `awaitCancellation` for the + * launched path. Never returns `null` — this backend always + * supports the inline path. + */ + override suspend fun queryRawInline( + ctx: RequestContext, + filters: List, + onEachStored: (RawEvent) -> Unit, + onEachLive: (Event) -> Unit, + onEose: () -> Unit, + ): SessionBackend.LiveSubscriptionHandle { drainFtsIfSearching(filters) val seenLock = AtomicBoolean(false) var seenIds: HashSet? = HashSet(1024) @@ -290,11 +315,14 @@ class LiveEventStore( onEachStored(raw) } onEose() + // Drop the dedupe set so the live path stops paying for it. seenLocked { seenIds = null } - awaitCancellation() - } finally { + } catch (e: Throwable) { + // A failed replay must not leak the registration. index.unregister(sub) + throw e } + return SessionBackend.LiveSubscriptionHandle { index.unregister(sub) } } override suspend fun count( diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt index 1119d9e9be..eb3ab83ce5 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt @@ -82,6 +82,42 @@ interface SessionBackend { onEose: () -> Unit, ): Unit = query(ctx, filters, onEachLive, onEose) + /** + * Keeps a live registration alive after [queryRawInline] returned. + * [close] detaches it — idempotent, callable from any thread. + */ + fun interface LiveSubscriptionHandle { + fun close() + } + + /** + * Inline variant of [queryRaw] for BOUNDED replays: runs the stored + * replay and [onEose] on the *calling* coroutine — no per-REQ + * `launch`, no dispatcher handoffs, no job to keep alive — and + * returns a [LiveSubscriptionHandle] whose `close()` ends the live + * tail (live events keep arriving via [onEachLive] from the ingest + * path until then). This is the small-REQ fast path: profiling put + * the per-REQ coroutine machinery at ~2/3 of a ~20-row REQ's + * time-to-EOSE. + * + * Callers MUST only use it when the replay is bounded (e.g. every + * filter carries a small `limit`): the replay occupies the session's + * receive coroutine, so a giant replay here would delay that + * connection's subsequent commands — including the CLOSE that could + * have cancelled it. Unbounded REQs belong on [queryRaw] in a + * separate coroutine. + * + * Returns `null` when the backend has no inline path (the default) — + * callers fall back to [queryRaw]. + */ + suspend fun queryRawInline( + ctx: RequestContext, + filters: List, + onEachStored: (RawEvent) -> Unit, + onEachLive: (Event) -> Unit, + onEose: () -> Unit, + ): LiveSubscriptionHandle? = null + /** Answers a NIP-45 COUNT with an exact cardinality for the caller in [ctx]. */ suspend fun count( ctx: RequestContext, diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt new file mode 100644 index 0000000000..d6e33b8a5e --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt @@ -0,0 +1,169 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.server + +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import kotlinx.coroutines.ExperimentalCoroutinesApi +import kotlinx.coroutines.test.UnconfinedTestDispatcher +import kotlinx.coroutines.test.runTest +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * NIP-01 semantics of the inline small-REQ fast path (a REQ whose + * filters all carry a small `limit` runs its replay on the receive + * coroutine — see [RelaySession.INLINE_REPLAY_MAX_ROWS]). The wire + * behavior must be indistinguishable from the launched path: stored + * replay, EOSE, live tail, CLOSE handling, and same-subId replacement. + */ +@OptIn(ExperimentalCoroutinesApi::class) +class InlineReqFastPathTest { + private fun hexId(n: Int): String = n.toString().padStart(64, '0') + + private val pubkey = "46fcbe3065eaf1ae7811465924e48923363ff3f526bd6f73d7c184b16bd8ce4d" + private val sig = "0".repeat(128) + + private fun testEvent( + idSeed: Int, + kind: Int = 1, + createdAt: Long = idSeed.toLong(), + ) = Event(hexId(idSeed), pubkey, createdAt, kind, emptyArray(), "note $idSeed", sig) + + private suspend fun serverWith( + dispatcher: kotlinx.coroutines.CoroutineDispatcher, + vararg events: Event, + ): NostrServer { + val store = EventStore(null) + events.forEach { store.insert(it) } + return NostrServer(store = store, policyBuilder = { EmptyPolicy }, parentContext = dispatcher) + } + + @Test + fun boundedReqRepliesStoredThenEoseThenLiveTail() = + runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + val server = serverWith(dispatcher, testEvent(1), testEvent(2, kind = 7)) + + val frames = mutableListOf() + val session = server.connect { frames.add(it) } + + // limit=10 → inline path. + session.receive("""["REQ","small",{"kinds":[1],"limit":10}]""") + + assertEquals(1, frames.count { it.startsWith("[\"EVENT\",\"small\"") }) + assertTrue(frames.first { it.startsWith("[\"EVENT\"") }.contains(hexId(1))) + assertTrue(frames.last().startsWith("[\"EOSE\",\"small\"")) + + // Live tail: a matching publish after EOSE must reach the sub. + session.receive("""["EVENT",${testEvent(3).toJson()}]""") + assertTrue(frames.any { it.startsWith("[\"EVENT\",\"small\"") && it.contains(hexId(3)) }) + // Non-matching kind stays out. + session.receive("""["EVENT",${testEvent(4, kind = 7).toJson()}]""") + assertTrue(frames.none { it.startsWith("[\"EVENT\",\"small\"") && it.contains(hexId(4)) }) + + server.close() + } + + @Test + fun closeDetachesTheInlineLiveTail() = + runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + val server = serverWith(dispatcher, testEvent(1)) + + val frames = mutableListOf() + val session = server.connect { frames.add(it) } + + session.receive("""["REQ","small",{"kinds":[1],"limit":10}]""") + session.receive("""["CLOSE","small"]""") + + session.receive("""["EVENT",${testEvent(2).toJson()}]""") + assertTrue(frames.none { it.startsWith("[\"EVENT\",\"small\"") && it.contains(hexId(2)) }) + + server.close() + } + + @Test + fun sameSubIdReplacesInlineSubscription() = + runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + val server = serverWith(dispatcher, testEvent(1), testEvent(2, kind = 7)) + + val frames = mutableListOf() + val session = server.connect { frames.add(it) } + + session.receive("""["REQ","sub",{"kinds":[1],"limit":10}]""") + // Replace with a kind-7 filter (still inline). The old live + // tail must be gone: a new kind-1 publish stays silent, a + // kind-7 one is delivered. + session.receive("""["REQ","sub",{"kinds":[7],"limit":10}]""") + + session.receive("""["EVENT",${testEvent(3, kind = 1).toJson()}]""") + assertTrue(frames.none { it.startsWith("[\"EVENT\",\"sub\"") && it.contains(hexId(3)) }) + session.receive("""["EVENT",${testEvent(4, kind = 7).toJson()}]""") + assertTrue(frames.any { it.startsWith("[\"EVENT\",\"sub\"") && it.contains(hexId(4)) }) + + server.close() + } + + @Test + fun unboundedReqStillWorksViaLaunchedPath() = + runTest { + val dispatcher = UnconfinedTestDispatcher(testScheduler) + val server = serverWith(dispatcher, testEvent(1)) + + val frames = mutableListOf() + val session = server.connect { frames.add(it) } + + // No limit → not inline-eligible; must behave identically. + session.receive("""["REQ","big",{"kinds":[1]}]""") + + assertTrue(frames.any { it.startsWith("[\"EVENT\",\"big\"") && it.contains(hexId(1)) }) + assertTrue(frames.any { it.startsWith("[\"EOSE\",\"big\"") }) + + session.receive("""["EVENT",${testEvent(2).toJson()}]""") + assertTrue(frames.any { it.startsWith("[\"EVENT\",\"big\"") && it.contains(hexId(2)) }) + + server.close() + } + + @Test + fun oversizedLimitSumIsNotInlineEligible() = + runTest { + // Behavior parity either way — this documents the boundary: + // summed limits over the cap route to the launched path and + // still answer correctly. + val dispatcher = UnconfinedTestDispatcher(testScheduler) + val server = serverWith(dispatcher, testEvent(1)) + + val frames = mutableListOf() + val session = server.connect { frames.add(it) } + + session.receive("""["REQ","big",{"kinds":[1],"limit":${RelaySession.INLINE_REPLAY_MAX_ROWS + 1}}]""") + + assertTrue(frames.any { it.startsWith("[\"EVENT\",\"big\"") && it.contains(hexId(1)) }) + assertTrue(frames.any { it.startsWith("[\"EOSE\",\"big\"") }) + + server.close() + } +} diff --git a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/SmallReqFloorBenchmark.kt b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/SmallReqFloorBenchmark.kt new file mode 100644 index 0000000000..9fe3a6ec33 --- /dev/null +++ b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/prodbench/SmallReqFloorBenchmark.kt @@ -0,0 +1,176 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.prodbench + +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.server.NostrServer +import com.vitorpamplona.quartz.nip01Core.relay.server.backend.IngestQueue +import com.vitorpamplona.quartz.nip01Core.relay.server.backend.LiveEventStore +import com.vitorpamplona.quartz.nip01Core.relay.server.backend.RequestContext +import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy +import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.EventFactory +import kotlinx.coroutines.CompletableDeferred +import kotlinx.coroutines.CoroutineScope +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.SupervisorJob +import kotlinx.coroutines.cancel +import kotlinx.coroutines.launch +import kotlinx.coroutines.runBlocking +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * Decomposes the fixed per-REQ cost on SMALL results — relayBench shows + * geode ~2.5× slower than strfry when a REQ returns ~20 events + * (author-archive 1.18 vs 0.48 ms EOSE p50 at 50k), while geode WINS the + * 500-event scenarios. With tiny result sets the per-REQ floor dominates, + * so this isolates its stages: + * + * A. raw store query (SQL + row decode) — the unavoidable part + * B. LiveEventStore.queryRaw to EOSE — adds live-subscription + * registration (FilterIndex), the per-REQ dedupe HashSet, and the + * search-drain check + * C. full session dispatch (RelaySession.receive of the REQ json) — + * adds command parse, policy, coroutine launch, frame assembly, + * and the send callback + * + * B−A is the live machinery; C−B is dispatch+serialization. Printed + * per-stage medians tell which one owns the floor. No speed assertions — + * numbers are informational; changes get A/B'd in relayBench. + */ +class SmallReqFloorBenchmark { + companion object { + const val EVENTS = 50_000 + const val AUTHORS = 2_500 // ~20 events per author, matching author-archive + const val ROUNDS = 400 + const val WARMUP = 100 + } + + private fun hexId(seed: Int): String = seed.toString(16).padStart(64, '0') + + private fun pubkey(seed: Int): String = (seed % AUTHORS).toString(16).padStart(64, 'a') + + private val sig = "0".repeat(128) + + private fun event(seed: Int): Event = + EventFactory.create( + id = hexId(seed), + pubKey = pubkey(seed), + createdAt = 1_600_000_000L + (seed * 7919) % 1_000_000, + kind = 1, + tags = emptyArray(), + content = "small req floor benchmark $seed", + sig = sig, + ) + + private fun median(samples: LongArray): Double { + samples.sort() + return samples[samples.size / 2] / 1e6 + } + + @Test + fun perReqFloorAt50k() = + runBlocking { + val store = EventStore(dbName = null, indexStrategy = DefaultIndexingStrategy(indexEventsByPubkeyAlone = true)) + (1..EVENTS).chunked(2000).forEach { chunk -> store.batchInsert(chunk.map { event(it) }) } + + val scope = CoroutineScope(Dispatchers.Default + SupervisorJob()) + val live = LiveEventStore(store, IngestQueue(store, scope.coroutineContext)) + val ctx = + object : RequestContext { + override val connectionId = 0L + override val policy = EmptyPolicy + override val authenticatedUsers = emptySet() + } + val server = NostrServer(store = store, policyBuilder = { EmptyPolicy }, parentContext = scope.coroutineContext) + + fun filterFor(round: Int) = Filter(authors = listOf(pubkey(round)), kinds = listOf(1), limit = 50) + + fun filterJson(round: Int) = """["REQ","floor$round",{"authors":["${pubkey(round)}"],"kinds":[1],"limit":50}]""" + + // --- A: raw store query --- + val a = LongArray(ROUNDS) + var rowsA = 0 + repeat(WARMUP) { store.rawQuery(listOf(filterFor(it))) {} } + repeat(ROUNDS) { round -> + val t0 = System.nanoTime() + var n = 0 + store.rawQuery(listOf(filterFor(round))) { n++ } + a[round] = System.nanoTime() - t0 + rowsA += n + } + + // --- B: backend queryRaw to EOSE (live machinery included) --- + suspend fun timeBackend(round: Int): Long { + val eose = CompletableDeferred() + val t0 = System.nanoTime() + val job = + scope.launch { + live.queryRaw( + ctx = ctx, + filters = listOf(filterFor(round)), + onEachStored = {}, + onEachLive = {}, + onEose = { eose.complete(System.nanoTime() - t0) }, + ) + } + val nanos = eose.await() + job.cancel() + return nanos + } + repeat(WARMUP) { timeBackend(it) } + val b = LongArray(ROUNDS) + repeat(ROUNDS) { b[it] = timeBackend(it) } + + // --- C: full session dispatch, REQ json in → EOSE frame out --- + suspend fun timeSession(round: Int): Long { + val eose = CompletableDeferred() + val t0 = System.nanoTime() + val session = + server.connect { frame -> + if (frame.startsWith("[\"EOSE\"")) eose.complete(System.nanoTime() - t0) + } + session.receive(filterJson(round)) + val nanos = eose.await() + session.close() + return nanos + } + repeat(WARMUP) { timeSession(it) } + val c = LongArray(ROUNDS) + repeat(ROUNDS) { c[it] = timeSession(it) } + + assertEquals(true, rowsA > 0, "author filters must return rows") + + val mA = median(a) + val mB = median(b) + val mC = median(c) + println("SmallReqFloorBenchmark @ ${EVENTS / 1000}k events, ~${rowsA / ROUNDS} rows/req, medians of $ROUNDS") + println(" A raw store query: ${"%6.3f".format(mA)} ms") + println(" B backend queryRaw→EOSE: ${"%6.3f".format(mB)} ms (live machinery +${"%6.3f".format(mB - mA)})") + println(" C session REQ→EOSE: ${"%6.3f".format(mC)} ms (dispatch+frames +${"%6.3f".format(mC - mB)})") + + server.close() + scope.cancel() + } +} From 1b786f31b48af6d516cd6be0347401fda842c610 Mon Sep 17 00:00:00 2001 From: Claude Date: Fri, 3 Jul 2026 23:58:48 +0000 Subject: [PATCH 11/21] =?UTF-8?q?perf(quartz):=20broaden=20inline=20REQ=20?= =?UTF-8?q?eligibility=20=E2=80=94=20ids-bounded=20filters,=20cap=20512?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first cut's rule (every filter needs limit, sum <= 256) missed the shapes relays actually receive: relayBench's author-archive carries limit=500 and by-ids carries only an ids list, so the fast path never engaged in the acceptance run. Bounds now come from limit OR the ids count (ids are unique keys, the result cannot exceed the list), summed against a 512 cap — a 500-row inline replay measures single-digit ms, nothing a CLOSE could meaningfully preempt. Provably unbounded filters (no limit, no ids — e.g. a bare tag filter) keep the launched path. Pending: relayBench verdict decides whether the fast path stays at all, per the keep-only-winners rule. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../nip01Core/relay/server/RelaySession.kt | 35 ++++++++++--------- 1 file changed, 19 insertions(+), 16 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt index 0ac21a81b5..5f5c290af8 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt @@ -275,19 +275,21 @@ class RelaySession( /** * A REQ qualifies for the inline fast path when its replay is - * provably small: every filter carries a `limit` and the limits sum - * to at most [INLINE_REPLAY_MAX_ROWS]. Inline replays run on this - * connection's receive coroutine — no per-REQ launch, no dispatcher - * handoffs (profiled at ~2/3 of a ~20-row REQ's time-to-EOSE) — at - * the price of delaying the connection's NEXT command until EOSE, - * which a bounded replay keeps in the tens-of-rows range. Unbounded - * REQs keep the launched path so a CLOSE can always interrupt them. + * provably bounded: every filter carries a `limit`, or an `ids` list + * (ids are unique keys, so the result can't exceed the list), and + * the bounds sum to at most [INLINE_REPLAY_MAX_ROWS]. Inline replays + * run on this connection's receive coroutine — no per-REQ launch, no + * dispatcher handoffs (profiled at ~2/3 of a ~20-row REQ's + * time-to-EOSE) — at the price of delaying the connection's NEXT + * command until EOSE, which the row cap keeps in the + * single-digit-milliseconds range. Unbounded REQs keep the launched + * path so a CLOSE can always interrupt a genuinely giant replay. */ private fun isInlineEligible(filters: List): Boolean { var total = 0 for (f in filters) { - val limit = f.limit ?: return false - total += limit + val bound = f.limit ?: f.ids?.size ?: return false + total += bound if (total > INLINE_REPLAY_MAX_ROWS) return false } return true @@ -428,13 +430,14 @@ class RelaySession( fun nextConnectionId(): Long = connectionIdSeq.fetchAndAdd(1L) /** - * Ceiling on the summed filter limits for the inline REQ fast - * path. Sized so the worst-case inline replay stays around a - * millisecond (~20-row REQs profile at ~0.2 ms of storage work) - * — small enough that delaying the connection's next command is - * unnoticeable, large enough to cover the profile/thread/ - * notifications-style lookups clients hammer relays with. + * Ceiling on the summed per-filter bounds (limit or ids count) + * for the inline REQ fast path. Sized so the worst-case inline + * replay stays in single-digit milliseconds (500-row replays + * measure ~5–8 ms of storage+frame work) — small enough that + * delaying the connection's next command is unnoticeable, large + * enough to cover the limit≤500 feed/notifications/archive + * shapes real clients hammer relays with. */ - const val INLINE_REPLAY_MAX_ROWS = 256 + const val INLINE_REPLAY_MAX_ROWS = 512 } } From f22657c870e523a78f60afeec002b0d9684ead0f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 00:03:34 +0000 Subject: [PATCH 12/21] =?UTF-8?q?revert(quartz):=20inline=20small-REQ=20fa?= =?UTF-8?q?st=20path=20=E2=80=94=20no=20wire-level=20win,=20floor=20is=20t?= =?UTF-8?q?ransport-side?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Reverts the queryRawInline fast path (fb29d655, 1b786f31) per the keep-only-winners rule. Three relayBench runs at 50k (baseline, cap-256 where the path never engaged, cap-512 where author-archive/by-ids/ 500-limit feeds genuinely took it) showed no movement outside the container drift band — strfry's own numbers drifted ±30% between runs and inline-eligible scenarios moved the same as ineligible ones. The in-process win was real but small (~17%, 0.60 -> 0.50 ms per ~21-row REQ); the wire-level p50 is 1.2-1.7 ms, so the missing ~1 ms per REQ sits in the transport (Ktor frame send path + client round trip) — backlog item 6 territory, not dispatch. Findings, numbers, and the do-not-retry note live in quartz/plans/2026-07-04-small-req-floor.md; SmallReqFloorBenchmark stays as the measurement tool. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- quartz/plans/2026-07-04-small-req-floor.md | 58 ++++++ quartz/plans/README.md | 3 +- .../nip01Core/relay/server/RelaySession.kt | 124 +++---------- .../relay/server/backend/LiveEventStore.kt | 32 +--- .../relay/server/backend/SessionBackend.kt | 36 ---- .../relay/server/InlineReqFastPathTest.kt | 169 ------------------ 6 files changed, 86 insertions(+), 336 deletions(-) create mode 100644 quartz/plans/2026-07-04-small-req-floor.md delete mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt diff --git a/quartz/plans/2026-07-04-small-req-floor.md b/quartz/plans/2026-07-04-small-req-floor.md new file mode 100644 index 0000000000..7723eba0c5 --- /dev/null +++ b/quartz/plans/2026-07-04-small-req-floor.md @@ -0,0 +1,58 @@ +# Small-REQ dispatch floor — investigated, inline fast path reverted + +**Status: closed (negative result recorded).** Backlog item 2 of the +relay performance campaign. + +## The gap + +relayBench at 50k events: geode WINS most 500-event query scenarios +(hashtag 5.4 vs 8.4 ms, recent-window 4.1 vs 8.5) but loses ~2.5× on +small results — author-archive (19 events) 1.2–1.7 ms vs strfry's +~0.5–0.6, thread (23) likewise — and the @8conn throughput inverts +(strfry 2–3× geode). With ~20-row responses, throughput ≈ 1/latency: +there is a fixed per-REQ floor. + +## Decomposition (SmallReqFloorBenchmark, kept in jvmTest) + +In-process at 50k events, ~21 rows/REQ, medians of 400: + +| stage | ms | +|---|---:| +| A raw store query (SQL + row decode) | 0.18 | +| B + live machinery (FilterIndex reg/unreg, dedupe set) | 0.36 | +| C + session dispatch (parse, launch, frames) | 0.60 | + +Note: in-memory DBs have no reader pool (`useReader` falls back to the +writer mutex), so absolute numbers are conservative vs the file-DB +bench setup. + +## What was tried and why it was reverted + +An inline fast path (`SessionBackend.queryRawInline`): REQs with +provably bounded replays (limit or ids-count summing ≤ 512) ran their +stored replay on the receive coroutine and kept only a live-tail +handle — no per-REQ `launch`, no Job, no dispatcher handoffs. It cut +in-process time-to-EOSE ~17% (0.60 → 0.50 ms) with full wire-behavior +parity (stored→EOSE order, live tail, CLOSE, same-subId replacement). + +Three relayBench runs (baseline, cap-256 [path not engaged — bench +filters carry limit=500 or no limit], cap-512 [engaged for +author-archive/by-ids/500-limit feeds]) showed **no movement outside +the container drift band** — strfry's own numbers drifted ±30% run to +run, and inline-eligible scenarios moved the same as ineligible ones. +Reverted per the keep-only-winners rule. + +## Where the floor actually is + +In-process REQ→EOSE is ~0.5–0.6 ms, but the wire-level p50 is +1.2–1.7 ms: the missing ~1 ms per REQ is transport-side — the Ktor CIO +frame write path, per-frame sends with no batching, and the client +round trip — which is backlog item 6 (websocket send path, 13–25% of +ingest CPU in the JFR profile) territory. strfry completes the whole +round trip in under 0.6 ms on a single event loop with uWebSockets. + +**Do not retry** coroutine-dispatch shaving for this gap without first +measuring the transport side: instrument the time between +`RelaySession`'s `onSend` invocation and the frame actually leaving +the socket, and compare permessage-deflate/frame-batching settings +against strfry's uWS configuration. diff --git a/quartz/plans/README.md b/quartz/plans/README.md index 577d380717..d8ce0a8cab 100644 --- a/quartz/plans/README.md +++ b/quartz/plans/README.md @@ -1,6 +1,6 @@ # quartz plans -_Audited 2026-06-30. 10 plans: 7 shipped (archived), 0 in-progress, 3 queued, 0 abandoned._ +_Audited 2026-06-30. 11 plans: 7 shipped (archived), 0 in-progress, 3 queued, 1 closed (negative result)._ ## Queued | Plan | Summary | @@ -8,6 +8,7 @@ _Audited 2026-06-30. 10 plans: 7 shipped (archived), 0 in-progress, 3 queued, 0 | [2026-05-08-local-headers-explorer.md](2026-05-08-local-headers-explorer.md) | Headers-only Bitcoin P2P client to verify NIP-03 OTS attestations without a trusted block explorer. | | [2026-06-12-giftwrap-deletion-requests.md](2026-06-12-giftwrap-deletion-requests.md) | Let a recipient-authored kind-5 delete/block a gift wrap (kind 1059) addressed to them. | | [2026-07-03-incremental-negentropy-storage.md](2026-07-03-incremental-negentropy-storage.md) | Always-current (created_at, id) index so cold NEG-OPENs stop paying a full scan + seal (~340 ms at 50k vs strfry's ~21 ms). | +| [2026-07-04-small-req-floor.md](2026-07-04-small-req-floor.md) | Small-REQ dispatch floor: decomposed, inline fast path tried and reverted (no wire-level win); floor is transport-side. | ## Archived (shipped) | Plan | Summary | diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt index 5f5c290af8..05cec0fad3 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/RelaySession.kt @@ -35,7 +35,6 @@ import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.CloseCmd import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.CountCmd import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.EventCmd import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.ReqCmd -import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.server.backend.RequestContext import com.vitorpamplona.quartz.nip01Core.relay.server.backend.SessionBackend import com.vitorpamplona.quartz.nip01Core.relay.server.policies.IRelayPolicy @@ -50,6 +49,7 @@ import com.vitorpamplona.quartz.utils.Log import com.vitorpamplona.quartz.utils.cache.LargeCache import kotlinx.coroutines.CancellationException import kotlinx.coroutines.CoroutineScope +import kotlinx.coroutines.Job import kotlinx.coroutines.channels.ClosedSendChannelException import kotlinx.coroutines.launch import kotlin.concurrent.atomics.AtomicLong @@ -75,16 +75,7 @@ class RelaySession( */ val id: Long = nextConnectionId(), ) : AutoCloseable { - /** - * One active REQ. Launched subscriptions cancel their coroutine; - * inline subscriptions (the bounded small-REQ fast path, which never - * had a coroutine) close their live-tail registration directly. - */ - private fun interface ActiveSubscription { - fun cancel() - } - - private val subscriptions = LargeCache() + private val subscriptions = LargeCache() /** * The authenticated-identity store for this connection. The engine is the @@ -112,8 +103,8 @@ class RelaySession( private fun addSubscription( subId: String, - sub: ActiveSubscription, - ) = subscriptions.put(subId, sub) + job: Job, + ) = subscriptions.put(subId, job) private fun cancelSubscription(subId: String): Boolean = subscriptions.remove(subId)?.let { @@ -122,7 +113,7 @@ class RelaySession( } ?: false fun cancelAllSubscriptions() { - subscriptions.forEach { _, sub -> sub.cancel() } + subscriptions.forEach { _, job -> job.cancel() } subscriptions.clear() negentropy.clear() } @@ -272,30 +263,7 @@ class RelaySession( } // -- NIP-01: REQ ---------------------------------------------------------- - - /** - * A REQ qualifies for the inline fast path when its replay is - * provably bounded: every filter carries a `limit`, or an `ids` list - * (ids are unique keys, so the result can't exceed the list), and - * the bounds sum to at most [INLINE_REPLAY_MAX_ROWS]. Inline replays - * run on this connection's receive coroutine — no per-REQ launch, no - * dispatcher handoffs (profiled at ~2/3 of a ~20-row REQ's - * time-to-EOSE) — at the price of delaying the connection's NEXT - * command until EOSE, which the row cap keeps in the - * single-digit-milliseconds range. Unbounded REQs keep the launched - * path so a CLOSE can always interrupt a genuinely giant replay. - */ - private fun isInlineEligible(filters: List): Boolean { - var total = 0 - for (f in filters) { - val bound = f.limit ?: f.ids?.size ?: return false - total += bound - if (total > INLINE_REPLAY_MAX_ROWS) return false - } - return true - } - - private suspend fun handleReq(cmd: ReqCmd) { + private fun handleReq(cmd: ReqCmd) { // Ask the policy whether a *new* subscription may open (e.g. a // max_subscriptions cap). A re-REQ on an existing id replaces it // 1-for-1 and doesn't grow the count, so it's exempt. Checked before @@ -320,54 +288,6 @@ class RelaySession( // Policy may rewrite filters to match the user's access level. val filters = (result as PolicyResult.Accepted).cmd.filters - // The `["EVENT","",` prefix is built once per - // subscription, not per row (zero-decode paths only). - fun framePrefix() = - buildString { - append("[\"EVENT\",") - RawEvent.appendJsonQuoted(this, cmd.subId) - append(',') - } - - fun sendStored( - framePrefix: String, - raw: RawEvent, - ) = sendRaw( - buildString(framePrefix.length + raw.jsonTags.length + raw.content.length + 256) { - append(framePrefix) - raw.appendJsonObjectTo(this) - append(']') - }, - ) - - // Small-REQ fast path: a provably bounded replay runs inline on - // this receive coroutine — no launch, no dispatcher handoffs — - // and only the live-tail handle is retained. See - // [SessionBackend.queryRawInline] for the contract. - if (!policy.filtersOutgoingEvents && isInlineEligible(filters)) { - val handle = - try { - val prefix = framePrefix() - store.queryRawInline( - ctx = requestContext, - filters = filters, - onEachStored = { raw -> sendStored(prefix, raw) }, - onEachLive = { event -> send(EventMessage(cmd.subId, event)) }, - onEose = { send(EoseMessage(cmd.subId)) }, - ) - } catch (e: CancellationException) { - throw e - } catch (e: Exception) { - send(ClosedMessage.of(cmd.subId, MachineReadablePrefix.ERROR, e.message ?: "query failed")) - return - } - if (handle != null) { - addSubscription(cmd.subId) { handle.close() } - return - } - // Backend without an inline path: fall through to launch. - } - val job = scope.launch { try { @@ -388,11 +308,26 @@ class RelaySession( // Zero-decode path: the stored replay splices raw // storage strings straight into wire frames — no tags // parse, no Event materialization, no re-serialize. - val prefix = framePrefix() + // The `["EVENT","",` prefix is built once per + // subscription, not per row. + val framePrefix = + buildString { + append("[\"EVENT\",") + RawEvent.appendJsonQuoted(this, cmd.subId) + append(',') + } store.queryRaw( ctx = requestContext, filters = filters, - onEachStored = { raw -> sendStored(prefix, raw) }, + onEachStored = { raw -> + sendRaw( + buildString(framePrefix.length + raw.jsonTags.length + raw.content.length + 256) { + append(framePrefix) + raw.appendJsonObjectTo(this) + append(']') + }, + ) + }, onEachLive = { event -> send(EventMessage(cmd.subId, event)) }, onEose = { send(EoseMessage(cmd.subId)) }, ) @@ -408,7 +343,7 @@ class RelaySession( } } - addSubscription(cmd.subId) { job.cancel() } + addSubscription(cmd.subId, job) } // -- NIP-01: CLOSE -------------------------------------------------------- @@ -428,16 +363,5 @@ class RelaySession( /** Allocates the next process-unique connection id. */ fun nextConnectionId(): Long = connectionIdSeq.fetchAndAdd(1L) - - /** - * Ceiling on the summed per-filter bounds (limit or ids count) - * for the inline REQ fast path. Sized so the worst-case inline - * replay stays in single-digit milliseconds (500-row replays - * measure ~5–8 ms of storage+frame work) — small enough that - * delaying the connection's next command is unnoticeable, large - * enough to cover the limit≤500 feed/notifications/archive - * shapes real clients hammer relays with. - */ - const val INLINE_REPLAY_MAX_ROWS = 512 } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt index a397156143..1d35362134 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/LiveEventStore.kt @@ -258,31 +258,6 @@ class LiveEventStore( onEachLive: (Event) -> Unit, onEose: () -> Unit, ) { - val handle = queryRawInline(ctx, filters, onEachStored, onEachLive, onEose) - try { - // Suspend until the caller's coroutine is cancelled (NIP-01 - // CLOSE or connection drop); the live tail runs meanwhile. - awaitCancellation() - } finally { - handle.close() - } - } - - /** - * The registration + replay + EOSE core of [queryRaw], run entirely - * on the calling coroutine. Small bounded REQs take this directly - * (see [SessionBackend.queryRawInline]) and skip the per-REQ - * coroutine; [queryRaw] wraps it with `awaitCancellation` for the - * launched path. Never returns `null` — this backend always - * supports the inline path. - */ - override suspend fun queryRawInline( - ctx: RequestContext, - filters: List, - onEachStored: (RawEvent) -> Unit, - onEachLive: (Event) -> Unit, - onEose: () -> Unit, - ): SessionBackend.LiveSubscriptionHandle { drainFtsIfSearching(filters) val seenLock = AtomicBoolean(false) var seenIds: HashSet? = HashSet(1024) @@ -315,14 +290,11 @@ class LiveEventStore( onEachStored(raw) } onEose() - // Drop the dedupe set so the live path stops paying for it. seenLocked { seenIds = null } - } catch (e: Throwable) { - // A failed replay must not leak the registration. + awaitCancellation() + } finally { index.unregister(sub) - throw e } - return SessionBackend.LiveSubscriptionHandle { index.unregister(sub) } } override suspend fun count( diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt index eb3ab83ce5..1119d9e9be 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/backend/SessionBackend.kt @@ -82,42 +82,6 @@ interface SessionBackend { onEose: () -> Unit, ): Unit = query(ctx, filters, onEachLive, onEose) - /** - * Keeps a live registration alive after [queryRawInline] returned. - * [close] detaches it — idempotent, callable from any thread. - */ - fun interface LiveSubscriptionHandle { - fun close() - } - - /** - * Inline variant of [queryRaw] for BOUNDED replays: runs the stored - * replay and [onEose] on the *calling* coroutine — no per-REQ - * `launch`, no dispatcher handoffs, no job to keep alive — and - * returns a [LiveSubscriptionHandle] whose `close()` ends the live - * tail (live events keep arriving via [onEachLive] from the ingest - * path until then). This is the small-REQ fast path: profiling put - * the per-REQ coroutine machinery at ~2/3 of a ~20-row REQ's - * time-to-EOSE. - * - * Callers MUST only use it when the replay is bounded (e.g. every - * filter carries a small `limit`): the replay occupies the session's - * receive coroutine, so a giant replay here would delay that - * connection's subsequent commands — including the CLOSE that could - * have cancelled it. Unbounded REQs belong on [queryRaw] in a - * separate coroutine. - * - * Returns `null` when the backend has no inline path (the default) — - * callers fall back to [queryRaw]. - */ - suspend fun queryRawInline( - ctx: RequestContext, - filters: List, - onEachStored: (RawEvent) -> Unit, - onEachLive: (Event) -> Unit, - onEose: () -> Unit, - ): LiveSubscriptionHandle? = null - /** Answers a NIP-45 COUNT with an exact cardinality for the caller in [ctx]. */ suspend fun count( ctx: RequestContext, diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt deleted file mode 100644 index d6e33b8a5e..0000000000 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/InlineReqFastPathTest.kt +++ /dev/null @@ -1,169 +0,0 @@ -/* - * Copyright (c) 2025 Vitor Pamplona - * - * Permission is hereby granted, free of charge, to any person obtaining a copy of - * this software and associated documentation files (the "Software"), to deal in - * the Software without restriction, including without limitation the rights to use, - * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the - * Software, and to permit persons to whom the Software is furnished to do so, - * subject to the following conditions: - * - * The above copyright notice and this permission notice shall be included in all - * copies or substantial portions of the Software. - * - * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR - * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS - * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR - * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN - * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION - * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. - */ -package com.vitorpamplona.quartz.nip01Core.relay.server - -import com.vitorpamplona.quartz.nip01Core.core.Event -import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy -import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore -import kotlinx.coroutines.ExperimentalCoroutinesApi -import kotlinx.coroutines.test.UnconfinedTestDispatcher -import kotlinx.coroutines.test.runTest -import kotlin.test.Test -import kotlin.test.assertEquals -import kotlin.test.assertTrue - -/** - * NIP-01 semantics of the inline small-REQ fast path (a REQ whose - * filters all carry a small `limit` runs its replay on the receive - * coroutine — see [RelaySession.INLINE_REPLAY_MAX_ROWS]). The wire - * behavior must be indistinguishable from the launched path: stored - * replay, EOSE, live tail, CLOSE handling, and same-subId replacement. - */ -@OptIn(ExperimentalCoroutinesApi::class) -class InlineReqFastPathTest { - private fun hexId(n: Int): String = n.toString().padStart(64, '0') - - private val pubkey = "46fcbe3065eaf1ae7811465924e48923363ff3f526bd6f73d7c184b16bd8ce4d" - private val sig = "0".repeat(128) - - private fun testEvent( - idSeed: Int, - kind: Int = 1, - createdAt: Long = idSeed.toLong(), - ) = Event(hexId(idSeed), pubkey, createdAt, kind, emptyArray(), "note $idSeed", sig) - - private suspend fun serverWith( - dispatcher: kotlinx.coroutines.CoroutineDispatcher, - vararg events: Event, - ): NostrServer { - val store = EventStore(null) - events.forEach { store.insert(it) } - return NostrServer(store = store, policyBuilder = { EmptyPolicy }, parentContext = dispatcher) - } - - @Test - fun boundedReqRepliesStoredThenEoseThenLiveTail() = - runTest { - val dispatcher = UnconfinedTestDispatcher(testScheduler) - val server = serverWith(dispatcher, testEvent(1), testEvent(2, kind = 7)) - - val frames = mutableListOf() - val session = server.connect { frames.add(it) } - - // limit=10 → inline path. - session.receive("""["REQ","small",{"kinds":[1],"limit":10}]""") - - assertEquals(1, frames.count { it.startsWith("[\"EVENT\",\"small\"") }) - assertTrue(frames.first { it.startsWith("[\"EVENT\"") }.contains(hexId(1))) - assertTrue(frames.last().startsWith("[\"EOSE\",\"small\"")) - - // Live tail: a matching publish after EOSE must reach the sub. - session.receive("""["EVENT",${testEvent(3).toJson()}]""") - assertTrue(frames.any { it.startsWith("[\"EVENT\",\"small\"") && it.contains(hexId(3)) }) - // Non-matching kind stays out. - session.receive("""["EVENT",${testEvent(4, kind = 7).toJson()}]""") - assertTrue(frames.none { it.startsWith("[\"EVENT\",\"small\"") && it.contains(hexId(4)) }) - - server.close() - } - - @Test - fun closeDetachesTheInlineLiveTail() = - runTest { - val dispatcher = UnconfinedTestDispatcher(testScheduler) - val server = serverWith(dispatcher, testEvent(1)) - - val frames = mutableListOf() - val session = server.connect { frames.add(it) } - - session.receive("""["REQ","small",{"kinds":[1],"limit":10}]""") - session.receive("""["CLOSE","small"]""") - - session.receive("""["EVENT",${testEvent(2).toJson()}]""") - assertTrue(frames.none { it.startsWith("[\"EVENT\",\"small\"") && it.contains(hexId(2)) }) - - server.close() - } - - @Test - fun sameSubIdReplacesInlineSubscription() = - runTest { - val dispatcher = UnconfinedTestDispatcher(testScheduler) - val server = serverWith(dispatcher, testEvent(1), testEvent(2, kind = 7)) - - val frames = mutableListOf() - val session = server.connect { frames.add(it) } - - session.receive("""["REQ","sub",{"kinds":[1],"limit":10}]""") - // Replace with a kind-7 filter (still inline). The old live - // tail must be gone: a new kind-1 publish stays silent, a - // kind-7 one is delivered. - session.receive("""["REQ","sub",{"kinds":[7],"limit":10}]""") - - session.receive("""["EVENT",${testEvent(3, kind = 1).toJson()}]""") - assertTrue(frames.none { it.startsWith("[\"EVENT\",\"sub\"") && it.contains(hexId(3)) }) - session.receive("""["EVENT",${testEvent(4, kind = 7).toJson()}]""") - assertTrue(frames.any { it.startsWith("[\"EVENT\",\"sub\"") && it.contains(hexId(4)) }) - - server.close() - } - - @Test - fun unboundedReqStillWorksViaLaunchedPath() = - runTest { - val dispatcher = UnconfinedTestDispatcher(testScheduler) - val server = serverWith(dispatcher, testEvent(1)) - - val frames = mutableListOf() - val session = server.connect { frames.add(it) } - - // No limit → not inline-eligible; must behave identically. - session.receive("""["REQ","big",{"kinds":[1]}]""") - - assertTrue(frames.any { it.startsWith("[\"EVENT\",\"big\"") && it.contains(hexId(1)) }) - assertTrue(frames.any { it.startsWith("[\"EOSE\",\"big\"") }) - - session.receive("""["EVENT",${testEvent(2).toJson()}]""") - assertTrue(frames.any { it.startsWith("[\"EVENT\",\"big\"") && it.contains(hexId(2)) }) - - server.close() - } - - @Test - fun oversizedLimitSumIsNotInlineEligible() = - runTest { - // Behavior parity either way — this documents the boundary: - // summed limits over the cap route to the launched path and - // still answer correctly. - val dispatcher = UnconfinedTestDispatcher(testScheduler) - val server = serverWith(dispatcher, testEvent(1)) - - val frames = mutableListOf() - val session = server.connect { frames.add(it) } - - session.receive("""["REQ","big",{"kinds":[1],"limit":${RelaySession.INLINE_REPLAY_MAX_ROWS + 1}}]""") - - assertTrue(frames.any { it.startsWith("[\"EVENT\",\"big\"") && it.contains(hexId(1)) }) - assertTrue(frames.any { it.startsWith("[\"EOSE\",\"big\"") }) - - server.close() - } -} From fd6662ca30fb5e8c9c9c263267e5640c5577a41c Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 00:38:09 +0000 Subject: [PATCH 13/21] =?UTF-8?q?perf(quartz):=20TCP=5FNODELAY=20for=20eve?= =?UTF-8?q?ry=20relay=20websocket=20client=20=E2=80=94=20kills=20CLOSE?= =?UTF-8?q?=E2=86=92REQ=20Nagle=20stalls?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Found while attributing the small-REQ wire floor (backlog item 6, latency half): geode's new WireReqFloorBenchmark measured a flat 43.7 ms per REQ round trip that survived every server-side change — store configs, dispatchers, the pump — and then vanished when the round's preceding CLOSE was dropped. Root cause is client-side: OkHttp does not set TCP_NODELAY, relays never answer a CLOSE (NIP-01), so its bytes sit unACKed for the peer's ~40 ms delayed-ACK window and Nagle holds the next REQ behind them. CLOSE-then-REQ is a Nostr client's hottest pattern — every feed/filter switch. relayBench's harness client already shipped a no-delay socket factory (which is why benchmark numbers never showed the stall) but the production clients did not. New TcpNoDelaySocketFactory (quartz jvmAndroid, next to BasicOkHttpWebSocket) is now used by the Android relay pool factory, the Desktop relay client, amy's relay connections, and geode's mirror worker. Direct connections only — SOCKS/Tor paths are untouched. With the factory, the benchmark puts geode's ~21-row REQ at ~1.25 ms on the wire (matching relayBench): ~0.6 ms Ktor CIO+OkHttp loopback floor, ~0.5 ms per-REQ server work (already investigated). Per-frame burst cost measured negligible and the pump adds ~nothing, so the send-path latency angle of backlog item 6 is closed as not-a-problem; its ingest-CPU share remains a separate throughput question. Findings recorded in quartz/plans/2026-07-04-small-req-floor.md. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../okhttp/OkHttpClientFactoryForRelays.kt | 7 + .../com/vitorpamplona/amethyst/cli/Context.kt | 3 +- .../amethyst/cli/commands/NipCommand.kt | 3 +- .../amethyst/cli/commands/NostrConnect.kt | 3 +- .../desktop/network/DesktopHttpClient.kt | 4 + .../geode/mirror/MirrorWorker.kt | 2 + .../geode/perf/WireReqFloorBenchmark.kt | 324 ++++++++++++++++++ quartz/plans/2026-07-04-small-req-floor.md | 43 ++- .../sockets/okhttp/TcpNoDelaySocketFactory.kt | 83 +++++ 9 files changed, 457 insertions(+), 15 deletions(-) create mode 100644 geode/src/test/kotlin/com/vitorpamplona/geode/perf/WireReqFloorBenchmark.kt create mode 100644 quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt diff --git a/amethyst/src/main/java/com/vitorpamplona/amethyst/service/okhttp/OkHttpClientFactoryForRelays.kt b/amethyst/src/main/java/com/vitorpamplona/amethyst/service/okhttp/OkHttpClientFactoryForRelays.kt index 7704771de1..8b27d5eee3 100644 --- a/amethyst/src/main/java/com/vitorpamplona/amethyst/service/okhttp/OkHttpClientFactoryForRelays.kt +++ b/amethyst/src/main/java/com/vitorpamplona/amethyst/service/okhttp/OkHttpClientFactoryForRelays.kt @@ -20,6 +20,7 @@ */ package com.vitorpamplona.amethyst.service.okhttp +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory import com.vitorpamplona.quartz.utils.Log import okhttp3.Dispatcher import okhttp3.OkHttpClient @@ -57,6 +58,12 @@ class OkHttpClientFactoryForRelays( OkHttpClient .Builder() .dispatcher(myDispatcher) + // TCP_NODELAY: a CLOSE (never answered by relays) followed by a + // REQ — every feed switch — otherwise nagles the REQ behind the + // unACKed CLOSE for the peer's delayed-ACK window (~40 ms+). + // See quartz TcpNoDelaySocketFactory. Direct connections only; + // the Tor SOCKS path is unaffected. + .socketFactory(TcpNoDelaySocketFactory) .dns(dns) .eventListenerFactory(DnsInvalidatingEventListener.Factory(dns)) .followRedirects(true) diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt index a96fc68f69..4ffe1a77fc 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/Context.kt @@ -50,6 +50,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory import com.vitorpamplona.quartz.nip01Core.signers.NostrSigner import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerInternal import com.vitorpamplona.quartz.nip01Core.store.IEventStore @@ -111,7 +112,7 @@ class Context( val identity: Identity, val state: RunState, ) : AutoCloseable { - private val okhttp = OkHttpClient.Builder().build() + private val okhttp = OkHttpClient.Builder().socketFactory(TcpNoDelaySocketFactory).build() val client: NostrClient = NostrClient( diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NipCommand.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NipCommand.kt index 665c4a039c..446c53fd48 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NipCommand.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NipCommand.kt @@ -31,6 +31,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.client.single.newSubId import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory import kotlinx.coroutines.channels.Channel import kotlinx.coroutines.channels.Channel.Factory.UNLIMITED import kotlinx.coroutines.selects.select @@ -143,7 +144,7 @@ object NipCommand { timeoutMs: Long, ): List> { if (SEARCH_RELAYS.isEmpty()) return emptyList() - val okhttp = OkHttpClient.Builder().build() + val okhttp = OkHttpClient.Builder().socketFactory(TcpNoDelaySocketFactory).build() val client = NostrClient(websocketBuilder = BasicOkHttpWebSocket.Builder { okhttp }) val filter = Filter(kinds = listOf(NIPTEXT_KIND, WIKI_KIND, LONGFORM_KIND), search = "NIP-$slug", limit = 10) val subId = newSubId() diff --git a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NostrConnect.kt b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NostrConnect.kt index c26c166854..08ffe57337 100644 --- a/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NostrConnect.kt +++ b/cli/src/main/kotlin/com/vitorpamplona/amethyst/cli/commands/NostrConnect.kt @@ -35,6 +35,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory import com.vitorpamplona.quartz.nip01Core.signers.NostrSignerInternal import com.vitorpamplona.quartz.nip46RemoteSigner.BunkerResponse import com.vitorpamplona.quartz.nip46RemoteSigner.NostrConnectEvent @@ -128,7 +129,7 @@ object NostrConnect { System.err.println("[nostrconnect] paste this into your signer within ${timeoutMs / 1000}s:") System.err.println(offer) - val okhttp = OkHttpClient.Builder().build() + val okhttp = OkHttpClient.Builder().socketFactory(TcpNoDelaySocketFactory).build() val client = NostrClient(websocketBuilder = BasicOkHttpWebSocket.Builder { okhttp }) val incoming = Channel(UNLIMITED) val subId = newSubId() diff --git a/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt b/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt index a63091f7a9..6f4c290668 100644 --- a/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt +++ b/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt @@ -25,6 +25,7 @@ import com.vitorpamplona.amethyst.commons.tor.TorType import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.normalizer.isLocalHost import com.vitorpamplona.quartz.nip01Core.relay.normalizer.isOnion +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory import kotlinx.coroutines.CoroutineScope import kotlinx.coroutines.flow.SharingStarted import kotlinx.coroutines.flow.StateFlow @@ -66,6 +67,9 @@ class DesktopHttpClient( private val directClient: OkHttpClient by lazy { OkHttpClient .Builder() + // TCP_NODELAY: keeps CLOSE-then-REQ bursts (feed switches) from + // nagling behind the unACKed CLOSE — see TcpNoDelaySocketFactory. + .socketFactory(TcpNoDelaySocketFactory) .connectionPool(sharedConnectionPool) .connectTimeout(BASE_TIMEOUT_SECONDS, TimeUnit.SECONDS) .readTimeout(BASE_TIMEOUT_SECONDS, TimeUnit.SECONDS) diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt index d12fffc496..f87737d89c 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt @@ -28,6 +28,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.server.NostrServer import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.BasicOkHttpWebSocket +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory import com.vitorpamplona.quartz.nip01Core.store.IEventStore import com.vitorpamplona.quartz.utils.Log import com.vitorpamplona.quartz.utils.TimeUtils @@ -117,6 +118,7 @@ class MirrorWorker( if (websocketBuilder == null) { OkHttpClient .Builder() + .socketFactory(TcpNoDelaySocketFactory) .pingInterval(Duration.ofSeconds(PING_INTERVAL_SECS)) .build() } else { diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/perf/WireReqFloorBenchmark.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/perf/WireReqFloorBenchmark.kt new file mode 100644 index 0000000000..b0778a2d07 --- /dev/null +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/perf/WireReqFloorBenchmark.kt @@ -0,0 +1,324 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.geode.perf + +import com.vitorpamplona.geode.KtorRelay +import com.vitorpamplona.geode.RelayEngine +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl +import com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp.TcpNoDelaySocketFactory +import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.EventFactory +import io.ktor.server.application.install +import io.ktor.server.cio.CIO +import io.ktor.server.engine.embeddedServer +import io.ktor.server.routing.routing +import io.ktor.server.websocket.WebSockets +import io.ktor.server.websocket.webSocket +import io.ktor.websocket.Frame +import io.ktor.websocket.readText +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.SupervisorJob +import kotlinx.coroutines.delay +import kotlinx.coroutines.launch +import kotlinx.coroutines.runBlocking +import okhttp3.OkHttpClient +import okhttp3.Request +import okhttp3.Response +import okhttp3.WebSocket +import okhttp3.WebSocketListener +import java.util.concurrent.ArrayBlockingQueue +import java.util.concurrent.TimeUnit +import kotlin.test.Test +import kotlin.test.assertTrue + +/** + * Splits the wire-level small-REQ floor into transport vs relay work, + * over the production stack (Ktor CIO server, OkHttp client, loopback). + * + * - echo-1 / echo-22: a bare Ktor CIO websocket replying with 1 / 22 + * ~250 B frames — the transport stack's round-trip floor; the burst + * variants show per-frame cost is negligible. + * - echo external / ext+1ms: replies produced from a foreign coroutine, + * immediately or after the connection loop parks — proves cross-context + * sends into CIO are prompt. + * - geode REQ / NOTICE / empty REQ / inproc: the relay's own layers. + * + * Historical note — this benchmark found the TcpNoDelaySocketFactory + * issue: without TCP_NODELAY on the client, a CLOSE (which relays never + * answer) followed by a REQ nagles the REQ behind the unACKed CLOSE for + * the ~40 ms delayed-ACK window, measured here as a flat 43.7 ms per + * round. With the factory (now used by every production client), geode's + * ~21-row REQ costs ~1.2 ms on the wire — matching relayBench — of which + * ~0.6 ms is the transport floor and ~0.5 ms the per-REQ server work + * documented in quartz/plans/2026-07-04-small-req-floor.md. + */ +class WireReqFloorBenchmark { + companion object { + const val EVENTS = 50_000 + const val AUTHORS = 2_500 + const val ROUNDS = 300 + const val WARMUP = 50 + const val BURST = 22 + } + + private fun hexId(seed: Int): String = seed.toString(16).padStart(64, '0') + + private fun pubkey(seed: Int): String = (seed % AUTHORS).toString(16).padStart(64, 'a') + + private val sig = "0".repeat(128) + + private fun event(seed: Int): Event = + EventFactory.create( + id = hexId(seed), + pubKey = pubkey(seed), + createdAt = 1_600_000_000L + (seed * 7919) % 1_000_000, + kind = 1, + tags = emptyArray(), + content = "wire req floor benchmark $seed", + sig = sig, + ) + + private fun median(samples: LongArray): Double { + samples.sort() + return samples[samples.size / 2] / 1e6 + } + + /** Blocking one-connection driver: send [request], await [expectedFrames] replies. */ + private class Driver( + client: OkHttpClient, + url: String, + ) { + private val frames = ArrayBlockingQueue(4096) + val socket: WebSocket + + init { + val opened = ArrayBlockingQueue(1) + socket = + client.newWebSocket( + Request.Builder().url(url).build(), + object : WebSocketListener() { + override fun onOpen( + webSocket: WebSocket, + response: Response, + ) { + opened.put(true) + } + + override fun onMessage( + webSocket: WebSocket, + text: String, + ) { + frames.put(text) + } + }, + ) + check(opened.poll(10, TimeUnit.SECONDS) == true) { "websocket did not open" } + } + + var lastFirstFrameNanos: Long = 0 + private set + + fun roundTrip( + request: String, + expectedFrames: Int, + ): Long { + val t0 = System.nanoTime() + socket.send(request) + repeat(expectedFrames) { i -> + checkNotNull(frames.poll(10, TimeUnit.SECONDS)) { "timed out waiting for frame" } + if (i == 0) lastFirstFrameNanos = System.nanoTime() - t0 + } + return System.nanoTime() - t0 + } + } + + @Test + fun transportVsRelayFloor() = + runBlocking { + // TCP_NODELAY, via the same factory production clients use: + // without it, a CLOSE (never answered) followed by a REQ nagles + // the REQ behind the unACKed CLOSE for the ~40 ms delayed-ACK + // window — this benchmark measured a flat 43.7 ms per round + // before the factory existed, which is how the issue was found. + val client = OkHttpClient.Builder().socketFactory(TcpNoDelaySocketFactory).build() + + // --- bare Ktor CIO echo server: N frames of ~250 B per request --- + val payload = "x".repeat(250) + val echo = + embeddedServer(CIO, port = 0) { + install(WebSockets) + routing { + webSocket("/") { + for (frame in incoming) { + if (frame !is Frame.Text) continue + val text = frame.readText() + if (text.startsWith("x")) { + // Reply from an EXTERNAL coroutine — the + // cross-context path geode's launched REQ + // handler uses. + val n = text.drop(1).toIntOrNull() ?: 1 + launch(Dispatchers.Default) { + repeat(n) { outgoing.send(Frame.Text(payload)) } + } + } else if (text.startsWith("d")) { + // Same, but AFTER the connection loop has + // gone idle — does a parked CIO connection + // pick up a cross-thread send promptly? + val n = text.drop(1).toIntOrNull() ?: 1 + launch(Dispatchers.Default) { + delay(1) + repeat(n) { outgoing.send(Frame.Text(payload)) } + } + } else { + val n = text.toIntOrNull() ?: 1 + repeat(n) { outgoing.send(Frame.Text(payload)) } + } + } + } + } + } + echo.start(wait = false) + val echoPort = + echo.engine + .resolvedConnectors() + .first() + .port + + val echoDriver = Driver(client, "ws://127.0.0.1:$echoPort/") + repeat(WARMUP) { echoDriver.roundTrip("1", 1) } + val echo1 = LongArray(ROUNDS) { echoDriver.roundTrip("1", 1) } + repeat(WARMUP) { echoDriver.roundTrip("$BURST", BURST) } + val echoN = LongArray(ROUNDS) { echoDriver.roundTrip("$BURST", BURST) } + repeat(WARMUP) { echoDriver.roundTrip("x1", 1) } + val echoExternal = LongArray(ROUNDS) { echoDriver.roundTrip("x1", 1) } + repeat(WARMUP) { echoDriver.roundTrip("d1", 1) } + val echoDelayed = LongArray(ROUNDS) { echoDriver.roundTrip("d1", 1) } + + // --- geode: real REQ over the same stack --- + // Same store setup SmallReqFloorBenchmark proved answers this + // filter in ~0.18 ms in-process — this benchmark attributes + // the wire-vs-in-process delta, so the storage side must be + // the known-fast configuration. + val store = EventStore(dbName = null, indexStrategy = DefaultIndexingStrategy(indexEventsByPubkeyAlone = true)) + (1..EVENTS).chunked(2000).forEach { chunk -> store.batchInsert(chunk.map { event(it) }) } + val relay = RelayEngine(url = "ws://127.0.0.1:7795/".normalizeRelayUrl(), store = store, parentContext = Dispatchers.IO + SupervisorJob()) + val server = KtorRelay(relay, host = "127.0.0.1", port = 0).start() + + val geodeDriver = Driver(client, server.url) + + fun req(round: Int) = """["REQ","w$round",{"authors":["${pubkey(round)}"],"kinds":[1],"limit":50}]""" + + // Each REQ answers with (rows + 1) frames — the EVENTs then + // EOSE — and the row count is author-dependent, so resolve + // every round's expected count from the store up front. Subs + // are CLOSEd after each round (silently, per NIP-01) so the + // relay's live-subscription registry doesn't grow with the + // round count and skew later samples. + suspend fun rowsFor(round: Int): Int = store.query(Filter(authors = listOf(pubkey(round)), kinds = listOf(1), limit = 50)).size + + repeat(WARMUP) { round -> + val rows = rowsFor(round) + geodeDriver.roundTrip(req(round), rows + 1) + geodeDriver.socket.send("""["CLOSE","w$round"]""") + } + val geodeSamples = LongArray(ROUNDS) + val geodeFirst = LongArray(ROUNDS) + var totalRows = 0 + for (i in 0 until ROUNDS) { + val round = WARMUP + i + val rows = rowsFor(round) + totalRows += rows + geodeSamples[i] = geodeDriver.roundTrip(req(round), rows + 1) + geodeFirst[i] = geodeDriver.lastFirstFrameNanos + geodeDriver.socket.send("""["CLOSE","w$round"]""") + } + + // Probe A: malformed frame -> NOTICE, answered inline on the + // pump coroutine (no launch, no SQL). Probe B: REQ matching + // nothing -> lone EOSE (launch + SQL, no rows). + repeat(WARMUP) { geodeDriver.roundTrip("not json", 1) } + val notice = LongArray(ROUNDS) { geodeDriver.roundTrip("not json", 1) } + repeat(WARMUP) { i -> + geodeDriver.roundTrip("""["REQ","n$i",{"ids":["${"f".repeat(64)}"]}]""", 1) + geodeDriver.socket.send("""["CLOSE","n$i"]""") + } + val emptyReq = + LongArray(ROUNDS) { i -> + val t = geodeDriver.roundTrip("""["REQ","m$i",{"ids":["${"f".repeat(64)}"]}]""", 1) + geodeDriver.socket.send("""["CLOSE","m$i"]""") + t + } + // Same probe with NO CLOSE between rounds: if this is fast, the + // 44 ms is the client's CLOSE frame sitting unACKed (server + // sends nothing for a CLOSE) and Nagle holding the next REQ + // behind it until the ~40 ms delayed ACK — a client-side TCP + // artifact, not the relay. + repeat(WARMUP) { i -> geodeDriver.roundTrip("""["REQ","nc$i",{"ids":["${"a".repeat(64)}"]}]""", 1) } + val emptyNoClose = + LongArray(ROUNDS) { i -> + geodeDriver.roundTrip("""["REQ","ncm$i",{"ids":["${"a".repeat(64)}"]}]""", 1) + } + + // Probe C: same RelayEngine, no wire — an in-process session + // while Ktor keeps serving. Splits engine-state issues from + // transport-adjacent ones. + val inproc = LongArray(ROUNDS) + run { + val q = ArrayBlockingQueue(4096) + val session = relay.server.connect { q.put(it) } + repeat(WARMUP) { i -> + session.receive("""["REQ","p$i",{"ids":["${"e".repeat(64)}"]}]""") + checkNotNull(q.poll(10, TimeUnit.SECONDS)) + session.receive("""["CLOSE","p$i"]""") + } + for (i in 0 until ROUNDS) { + val t0 = System.nanoTime() + session.receive("""["REQ","q$i",{"ids":["${"e".repeat(64)}"]}]""") + checkNotNull(q.poll(10, TimeUnit.SECONDS)) + inproc[i] = System.nanoTime() - t0 + session.receive("""["CLOSE","q$i"]""") + } + session.close() + } + + assertTrue(totalRows > 0) + println("WireReqFloorBenchmark (loopback, OkHttp client) @ ${EVENTS / 1000}k events, medians of $ROUNDS") + println(" echo 1 frame: ${"%6.3f".format(median(echo1))} ms") + println(" echo $BURST frames: ${"%6.3f".format(median(echoN))} ms") + println(" echo 1 frame (external): ${"%6.3f".format(median(echoExternal))} ms") + println(" echo 1 frame (ext+1ms): ${"%6.3f".format(median(echoDelayed))} ms") + println(" geode REQ (~${totalRows / ROUNDS} rows+EOSE): ${"%6.3f".format(median(geodeSamples))} ms (first frame ${"%6.3f".format(median(geodeFirst))} ms)") + println(" geode NOTICE (inline): ${"%6.3f".format(median(notice))} ms") + println(" geode empty REQ (launch): ${"%6.3f".format(median(emptyReq))} ms") + println(" geode empty REQ no CLOSE: ${"%6.3f".format(median(emptyNoClose))} ms") + println(" geode empty REQ (inproc): ${"%6.3f".format(median(inproc))} ms") + + geodeDriver.socket.close(1000, null) + echoDriver.socket.close(1000, null) + server.stop(0, 1_000) + relay.close() + echo.stop(0, 500) + client.dispatcher.executorService.shutdown() + } +} diff --git a/quartz/plans/2026-07-04-small-req-floor.md b/quartz/plans/2026-07-04-small-req-floor.md index 7723eba0c5..77ea84fe55 100644 --- a/quartz/plans/2026-07-04-small-req-floor.md +++ b/quartz/plans/2026-07-04-small-req-floor.md @@ -42,17 +42,36 @@ the container drift band** — strfry's own numbers drifted ±30% run to run, and inline-eligible scenarios moved the same as ineligible ones. Reverted per the keep-only-winners rule. -## Where the floor actually is +## Where the floor actually is (corrected after WireReqFloorBenchmark) -In-process REQ→EOSE is ~0.5–0.6 ms, but the wire-level p50 is -1.2–1.7 ms: the missing ~1 ms per REQ is transport-side — the Ktor CIO -frame write path, per-frame sends with no batching, and the client -round trip — which is backlog item 6 (websocket send path, 13–25% of -ingest CPU in the JFR profile) territory. strfry completes the whole -round trip in under 0.6 ms on a single event loop with uWebSockets. +The follow-up wire benchmark (geode's `WireReqFloorBenchmark`, Ktor CIO ++ OkHttp on loopback) attributed the full path: -**Do not retry** coroutine-dispatch shaving for this gap without first -measuring the transport side: instrument the time between -`RelaySession`'s `onSend` invocation and the frame actually leaving -the socket, and compare permessage-deflate/frame-batching settings -against strfry's uWS configuration. +| leg | ms | +|---|---:| +| bare Ktor CIO echo round trip (1 or 22 frames — same) | ~0.9–1.0 | +| geode NOTICE (inline, full pump + Ktor send) | ~0.5 | +| geode empty REQ (launch + SQL, 0 rows) | ~0.6–0.8 | +| geode ~21-row REQ, wire | ~1.25 (= relayBench's number) | + +**geode's websocket send path has no latency problem** — per-frame burst +cost is negligible (echo-22 ≈ echo-1), the pump adds ~nothing (NOTICE ≈ +0.5 ms), and the residual vs strfry (~0.5 ms/REQ) is the per-REQ server +work already investigated above. Frame batching / permessage-deflate +would not move these numbers. Backlog item 6's remaining open angle is +the INGEST-side CPU share (13–25% in the JFR profile) — a throughput +question, not this latency one. + +**The real find was client-side.** The first wire measurements showed a +flat 43.7 ms per REQ — which turned out to be the benchmark's own OkHttp +client: OkHttp does not set TCP_NODELAY, and the CLOSE-then-REQ pattern +(every feed/filter switch!) nagles the REQ behind the unACKed CLOSE +(relays never answer CLOSE) for the ~40 ms delayed-ACK window. +relayBench's harness client already carried a no-delay socket factory — +which is why bench numbers never showed it — but the production clients +(Android relay pool, Desktop, amy, geode's mirror) did not. Fixed by +`TcpNoDelaySocketFactory` (quartz jvmAndroid), now used by all of them. + +**Do not retry** relay-side latency work for the small-REQ gap; the +addressable remainder is the ~0.4 ms of per-REQ dispatch machinery this +doc's revert already covers, and it does not show on the wire. diff --git a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt new file mode 100644 index 0000000000..2f175a3b3b --- /dev/null +++ b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt @@ -0,0 +1,83 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.relay.sockets.okhttp + +import java.net.InetAddress +import java.net.InetSocketAddress +import java.net.Socket +import javax.net.SocketFactory + +/** + * [SocketFactory] that enables `TCP_NODELAY` — pass to + * `OkHttpClient.Builder.socketFactory` for every client that holds Nostr + * relay WebSockets. + * + * OkHttp does not disable Nagle's algorithm, and a Nostr client's hottest + * wire pattern trips over it: CLOSE immediately followed by REQ — every + * feed or filter switch. Relays never reply to a CLOSE (NIP-01), so its + * bytes sit unACKed for the peer's delayed-ACK window, and Nagle holds + * the REQ back behind them until that ACK arrives. Measured on loopback + * as a flat ~44 ms added to every such REQ's round trip (geode's + * `WireReqFloorBenchmark`, which is how this was found); WAN delayed-ACK + * windows are the same order. relayBench's harness client ships the same + * fix, which is why benchmark numbers never showed it. + * + * OkHttp only uses this factory for direct connections — proxied (e.g. + * Tor SOCKS) connections take a different path and keep their transport's + * defaults. + */ +object TcpNoDelaySocketFactory : SocketFactory() { + private fun socket() = Socket().apply { tcpNoDelay = true } + + override fun createSocket(): Socket = socket() + + override fun createSocket( + host: String?, + port: Int, + ): Socket = socket().apply { connect(InetSocketAddress(host, port)) } + + override fun createSocket( + host: String?, + port: Int, + localHost: InetAddress?, + localPort: Int, + ): Socket = + socket().apply { + bind(InetSocketAddress(localHost, localPort)) + connect(InetSocketAddress(host, port)) + } + + override fun createSocket( + host: InetAddress?, + port: Int, + ): Socket = socket().apply { connect(InetSocketAddress(host, port)) } + + override fun createSocket( + address: InetAddress?, + port: Int, + localAddress: InetAddress?, + localPort: Int, + ): Socket = + socket().apply { + bind(InetSocketAddress(localAddress, localPort)) + connect(InetSocketAddress(address, port)) + } +} From 79373672dfc5a77ab7a85ecd16b05663007b0948 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 00:48:22 +0000 Subject: [PATCH 14/21] feat(geode/quartz): [database] tuning knobs + periodic PRAGMA optimize MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Backlog items 4-5 plumbing, config-gated and off by default so quartz library defaults stay untouched for the app-side stores: - quartz: SQLiteEventStore/EventStore accept extraPragmas (applied on every pooled connection AFTER the built-in configuration, so they can override it) and expose optimize() — an analysis_limit-bounded PRAGMA optimize for incremental planner-stats refresh. - geode: [database] readers / mmap_size / temp_store_memory map onto the store; optimize_interval_seconds drives a maintenance coroutine (cancelled first in the shutdown hook, before the store closes). The A/B verdict on whether the example config should RECOMMEND any of these on container-class hardware follows in the next commit — the knobs themselves are operator tools worth having either way, since mmap/temp-store value is hardware-dependent. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- geode/config.example.toml | 16 +++++ .../kotlin/com/vitorpamplona/geode/Main.kt | 36 ++++++++-- .../geode/config/StaticConfig.kt | 24 +++++++ .../geode/config/StaticConfigTest.kt | 28 ++++++++ .../nip01Core/store/sqlite/EventStore.kt | 8 ++- .../store/sqlite/SQLiteEventStore.kt | 26 +++++++ .../store/sqlite/ExtraPragmasTest.kt | 69 +++++++++++++++++++ 7 files changed, 202 insertions(+), 5 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/ExtraPragmasTest.kt diff --git a/geode/config.example.toml b/geode/config.example.toml index 6b18b478d2..6c50f51848 100644 --- a/geode/config.example.toml +++ b/geode/config.example.toml @@ -46,6 +46,22 @@ path = "/" in_memory = false file = "/var/lib/geode/events.db" +# Reader-connection pool size (file-backed stores only). Default 4. +# readers = 4 + +# PRAGMA mmap_size in bytes — maps the db file into memory so reads +# skip the pread syscall. Off by default (SQLite default). +# mmap_size = 268435456 + +# PRAGMA temp_store = MEMORY — RAM instead of temp files for large +# sorts. Off by default. +# temp_store_memory = true + +# Refresh query-planner statistics (PRAGMA optimize) every N seconds. +# Incremental and usually a no-op; keeps the planner from drifting +# onto the wrong index as the corpus grows. Off by default. +# optimize_interval_seconds = 3600 + [options] # Drop events whose Schnorr signature does not verify. Strongly # recommended for any relay accepting traffic from real clients. diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt index 19e470853b..4a7b1e6179 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt @@ -37,9 +37,14 @@ import com.vitorpamplona.quartz.nip01Core.relay.server.policies.OptionalAuthPoli import com.vitorpamplona.quartz.nip01Core.relay.server.policies.RejectFutureEventsPolicy import com.vitorpamplona.quartz.nip01Core.relay.server.policies.VerifyAuthOnlyPolicy import com.vitorpamplona.quartz.nip01Core.relay.server.policies.VerifyPolicy -import com.vitorpamplona.quartz.nip01Core.store.IEventStore import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore import com.vitorpamplona.quartz.nip77Negentropy.NegentropySettings +import kotlinx.coroutines.CoroutineScope +import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.SupervisorJob +import kotlinx.coroutines.cancel +import kotlinx.coroutines.delay +import kotlinx.coroutines.launch import java.io.File /** @@ -126,11 +131,20 @@ fun main(args: Array) { cliInfoFile?.let { RelayInfo.fromFile(it) } ?: config.resolveInfo(fullTextSearch) - val store: IEventStore = + // Deployment tuning from `[database]` — off by default; the quartz + // library defaults stay tuned for the app-side stores. + val extraPragmas = + buildList { + config.database.mmap_size?.let { add("PRAGMA mmap_size = $it;") } + if (config.database.temp_store_memory) add("PRAGMA temp_store = MEMORY;") + } + val store = EventStore( dbName = dbFile, relay = advertisedUrl, indexStrategy = relayIndexingStrategy(fullTextSearch, config.negentropy.live_index), + numReaders = config.database.readers ?: 4, + extraPragmas = extraPragmas, ) val policyBuilder: () -> IRelayPolicy = { @@ -223,12 +237,26 @@ fun main(args: Array) { MirrorWorker(upstreams, relay.server).also { it.start() } } + // Periodic query-planner statistics refresh (`PRAGMA optimize`). + // Incremental and usually a no-op; failures are swallowed — a missed + // refresh only means slightly staler planner stats until the next tick. + val maintenanceScope = CoroutineScope(Dispatchers.IO + SupervisorJob()) + config.database.optimize_interval_seconds?.let { secs -> + maintenanceScope.launch { + while (true) { + delay(secs * 1000) + runCatching { store.optimize() } + } + } + } + Runtime.getRuntime().addShutdownHook( Thread { // Each step wrapped so a throw in any stage doesn't skip // `relay.close()` (which closes the SQLite store). The mirror - // goes first: stop pulling new events before the queue and - // store beneath it shut down. + // and maintenance go first: stop touching the store before the + // queue and store beneath them shut down. + runCatching { maintenanceScope.cancel() } runCatching { mirror?.close() } runCatching { server.stop() } runCatching { relay.close() } diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt index 7079a3fa28..7145139f57 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt @@ -106,6 +106,30 @@ data class StaticConfig( /** True keeps an in-memory SQLite db (default — events vanish on restart). */ val in_memory: Boolean = true, val file: String? = null, + /** + * Reader-connection pool size. `null` keeps quartz's default (4). + * Only meaningful for file-backed stores; in-memory databases + * share the single writer connection regardless. + */ + val readers: Int? = null, + /** + * `PRAGMA mmap_size` in bytes, e.g. `268435456` for 256 MiB. + * Maps the database file into memory so reads skip the pread + * syscall + page-cache copy. `null` keeps SQLite's default (off). + */ + val mmap_size: Long? = null, + /** + * `PRAGMA temp_store = MEMORY` — keeps sort/temp b-trees for + * large queries in RAM instead of temp files. + */ + val temp_store_memory: Boolean = false, + /** + * Refresh query-planner statistics (`PRAGMA analysis_limit; + * PRAGMA optimize`) every this many seconds. Incremental and + * usually a no-op, but keeps the planner from drifting onto the + * wrong index as the corpus grows/changes shape. `null` = never. + */ + val optimize_interval_seconds: Long? = null, ) data class OptionsSection( diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt index 765bc58881..5974fb9294 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt @@ -48,6 +48,34 @@ class StaticConfigTest { assertEquals(false, c.options.verify_signatures) } + @Test + fun parsesDatabaseTuningKnobs() { + val toml = + """ + [database] + in_memory = false + file = "/tmp/x.db" + readers = 8 + mmap_size = 268435456 + temp_store_memory = true + optimize_interval_seconds = 3600 + """.trimIndent() + + val c = StaticConfig.fromToml(toml) + + assertEquals(8, c.database.readers) + assertEquals(268435456L, c.database.mmap_size) + assertEquals(true, c.database.temp_store_memory) + assertEquals(3600L, c.database.optimize_interval_seconds) + + // And all knobs default to off/null so plain configs are untouched. + val d = StaticConfig.fromToml("") + assertEquals(null, d.database.readers) + assertEquals(null, d.database.mmap_size) + assertEquals(false, d.database.temp_store_memory) + assertEquals(null, d.database.optimize_interval_seconds) + } + @Test fun mirrorSectionDefaultsToEmpty() { assertTrue(StaticConfig.fromToml("").mirror.isEmpty()) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt index 82d1df1c0b..c0eb526e11 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/EventStore.kt @@ -39,8 +39,14 @@ class EventStore( dbName: String? = "events.db", override val relay: NormalizedRelayUrl? = "wss://quartz.local/".normalizeRelayUrl(), val indexStrategy: IndexingStrategy = DefaultIndexingStrategy(), + numReaders: Int = 4, + /** See [SQLiteEventStore.extraPragmas] — deployment-specific tuning. */ + extraPragmas: List = emptyList(), ) : IEventStore { - val store = SQLiteEventStore(BundledSQLiteDriver(), dbName, relay, indexStrategy) + val store = SQLiteEventStore(BundledSQLiteDriver(), dbName, relay, indexStrategy, numReaders, extraPragmas) + + /** Incremental planner-stats refresh — see [SQLiteEventStore.optimize]. */ + suspend fun optimize() = store.optimize() override suspend fun insert(event: Event) = store.insertEvent(event) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt index 2a802b4c23..94179e9406 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt @@ -48,6 +48,15 @@ class SQLiteEventStore( val relay: NormalizedRelayUrl? = null, val indexStrategy: IndexingStrategy = DefaultIndexingStrategy(), val numReaders: Int = 4, + /** + * Extra `PRAGMA` statements run on every pooled connection after the + * built-in configuration (cache size, WAL, busy timeout, …), so they + * can override it. Deployment-specific tuning goes here — e.g. + * `PRAGMA mmap_size = 268435456;` or `PRAGMA temp_store = MEMORY;` + * on server hardware — while the library defaults stay tuned for the + * app-side stores. Statements must be self-contained SQL. + */ + val extraPragmas: List = emptyList(), ) { companion object { const val DATABASE_VERSION = 4 @@ -133,6 +142,9 @@ class SQLiteEventStore( // upgrading its snapshot. With it, SQLite retries internally // for up to N ms before giving up. Matches Room's default. db.execSQL("PRAGMA busy_timeout = 5000;") + + // Deployment overrides last, so they win over the defaults. + extraPragmas.forEach { db.execSQL(it) } }, onMigrate = { db -> val currentVersion = getUserVersion(db) @@ -244,6 +256,20 @@ class SQLiteEventStore( db.execSQL("ANALYZE") } + /** + * Incremental planner-statistics refresh for long-running relays. + * `PRAGMA optimize` re-analyzes only tables whose content changed + * enough to matter, and `analysis_limit` bounds each table's scan — + * so a periodic call stays cheap (typically no-op) while keeping the + * planner from drifting onto the wrong index as the corpus grows. + * Runs on the writer connection; call it off the hot path. + */ + suspend fun optimize() = + pool.useWriter { db -> + db.execSQL("PRAGMA analysis_limit = 400;") + db.execSQL("PRAGMA optimize;") + } + /** * Live-index mutations gathered during one write transaction and * applied only after its COMMIT — a rolled-back row (savepoint or diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/ExtraPragmasTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/ExtraPragmasTest.kt new file mode 100644 index 0000000000..bd1a8d289c --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/ExtraPragmasTest.kt @@ -0,0 +1,69 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip01Core.store.sqlite + +import kotlinx.coroutines.test.runTest +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * [SQLiteEventStore.extraPragmas] must actually reach every pooled + * connection (after the built-in defaults, so they can override), and + * [SQLiteEventStore.optimize] must run cleanly on a live store. + */ +class ExtraPragmasTest { + private suspend fun SQLiteEventStore.pragmaValue(name: String): Long = + pool.useReader { db -> + db.prepare("PRAGMA $name").use { stmt -> + stmt.step() + stmt.getLong(0) + } + } + + @Test + fun extraPragmasApplyToConnections() = + runTest { + // temp_store: 0=default, 2=MEMORY. + val store = + SQLiteEventStore( + dbName = null, + extraPragmas = listOf("PRAGMA temp_store = MEMORY;"), + ) + assertEquals(2L, store.pragmaValue("temp_store")) + store.close() + } + + @Test + fun defaultsHaveNoExtraPragmas() = + runTest { + val store = SQLiteEventStore(dbName = null) + assertEquals(0L, store.pragmaValue("temp_store")) + store.close() + } + + @Test + fun optimizeRunsCleanly() = + runTest { + val store = SQLiteEventStore(dbName = null) + store.optimize() + store.close() + } +} From cc73639ae5fc2daf4d1c53299737f2fd50a26a41 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 00:50:37 +0000 Subject: [PATCH 15/21] =?UTF-8?q?docs(geode):=20SQLite=20knobs=20A/B=20ver?= =?UTF-8?q?dict=20=E2=80=94=20no=20winner,=20knobs=20stay=20off=20by=20def?= =?UTF-8?q?ault?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Backlog items 4-5 close. Both-variants-in-one-run A/B at 50k, repeated with relay order reversed: every delta flipped with the order (the second-running relay won queries and ingest latency in BOTH runs), so readers=8 / mmap_size=256MiB / temp_store=MEMORY / periodic PRAGMA optimize are all noise-level on container-class hardware. The config plumbing stays (hardware-dependent, operators should measure their own box); the example config now says so explicitly. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- PLANS.md | 2 +- geode/config.example.toml | 5 +++ geode/plans/2026-07-04-sqlite-knobs-ab.md | 50 +++++++++++++++++++++++ geode/plans/README.md | 8 +++- 4 files changed, 63 insertions(+), 2 deletions(-) create mode 100644 geode/plans/2026-07-04-sqlite-knobs-ab.md diff --git a/PLANS.md b/PLANS.md index e3474caec3..1b8267abd5 100644 --- a/PLANS.md +++ b/PLANS.md @@ -38,7 +38,7 @@ listing (including archived plans), open that folder's `README.md`. | cli | 6 | 5 | 1 | 0 | 0 | [cli/plans](cli/plans/README.md) | | quic | 4 | 3 | 0 | 0 | 1 | [quic/plans](quic/plans/README.md) | | quic/interop | 1 | 1 | 0 | 0 | 0 | [quic/interop/plans](quic/interop/plans/README.md) | -| geode | 4 | 4 | 0 | 0 | 0 | [geode/plans](geode/plans/README.md) | +| geode | 5 | 4 | 0 | 0 | 0 | [geode/plans](geode/plans/README.md) | | docs (frozen) | 52 | 48 | 2 | 0 | 2 | [docs/plans](docs/plans/README.md) | ## Live work (not shipped) diff --git a/geode/config.example.toml b/geode/config.example.toml index 6c50f51848..17f15c8d31 100644 --- a/geode/config.example.toml +++ b/geode/config.example.toml @@ -46,6 +46,11 @@ path = "/" in_memory = false file = "/var/lib/geode/events.db" +# NOTE: the four knobs below measured as pure noise in relayBench A/B +# runs on 4-core container hardware at 50k events (see +# geode/plans/2026-07-04-sqlite-knobs-ab.md) — their value is +# hardware-dependent, so benchmark on YOUR box before enabling. +# # Reader-connection pool size (file-backed stores only). Default 4. # readers = 4 diff --git a/geode/plans/2026-07-04-sqlite-knobs-ab.md b/geode/plans/2026-07-04-sqlite-knobs-ab.md new file mode 100644 index 0000000000..b1d5146fae --- /dev/null +++ b/geode/plans/2026-07-04-sqlite-knobs-ab.md @@ -0,0 +1,50 @@ +# Server-side SQLite knobs — A/B verdict: no winner on container hardware + +**Status: closed (negative result recorded).** Backlog items 4–5 of the +relay performance campaign. + +## What was tested + +The knobs the campaign had previously reverted as noise-inconclusive, +re-tested under the A/B protocol — both variants in ONE relayBench run +(identical container conditions), then the run repeated with the relay +order reversed to expose order bias. 50k synthetic corpus, all four +knobs on the candidate at once: + +```toml +[database] +readers = 8 # reader pool (default 4) +mmap_size = 268435456 # 256 MiB +temp_store_memory = true +optimize_interval_seconds = 15 # PRAGMA optimize between ingest and queries +``` + +## Result + +Every observed delta flipped with run order. In BOTH runs the +second-running relay won most query scenarios and ingest latency — +regardless of which variant it was: + +- run 1 (plain → knobs): knobs won 7/10 query scenarios, better ingest + p99 (8.2 vs 16.0 ms). +- run 2 (knobs → plain): plain won 6/10 query scenarios, better ingest + p50/p99/throughput (8,552 vs 8,015 ev/s). + +Storage size identical (±1 MiB). The periodic `PRAGMA optimize` also +did not move the author-archive/planner-sensitive scenarios. + +## Verdict + +**All knobs stay off by default.** The `[database]` config plumbing +ships anyway (readers / mmap_size / temp_store_memory / +optimize_interval_seconds) because their value is hardware-dependent — +an operator on NVMe with a large page cache or a memory-constrained VPS +should measure on their own box — and `optimize_interval_seconds` +remains sensible operationally for long-running relays even without a +measurable win at 50k-fresh-database scale. + +**Do not re-benchmark these on container-class 4-core hardware** without +first fixing the run-order bias: the protocol note in +`quartz/plans/2026-07-03-incremental-negentropy-storage.md` applies — +same-run A/B plus order reversal is the minimum, and anything that +doesn't survive the order flip is noise. diff --git a/geode/plans/README.md b/geode/plans/README.md index 14807fb2a5..43c34bc5a1 100644 --- a/geode/plans/README.md +++ b/geode/plans/README.md @@ -1,6 +1,6 @@ # geode plans -_Audited 2026-06-30. 4 plans: 4 shipped (archived), 0 in-progress, 0 queued, 0 abandoned._ +_Audited 2026-06-30. 5 plans: 4 shipped (archived), 0 in-progress, 0 queued, 1 closed (negative result)._ Performance-focused design docs for future work. Each file is a self-contained sketch — problem statement, observed numbers, proposed @@ -26,3 +26,9 @@ regressions show up in the regular CI matrix once they're enabled. - [archive/2026-05-07-live-broadcast-fanout-index.md](archive/2026-05-07-live-broadcast-fanout-index.md) - [archive/2026-05-07-connection-scaling.md](archive/2026-05-07-connection-scaling.md) - [archive/2026-05-07-negentropy-large-corpus.md](archive/2026-05-07-negentropy-large-corpus.md) + +## Closed (negative result) + +| Plan | Verdict | +| ---- | ------- | +| [2026-07-04-sqlite-knobs-ab.md](2026-07-04-sqlite-knobs-ab.md) | readers/mmap/temp_store/PRAGMA-optimize knobs: no winner on container hardware; every delta flipped with run order. Knobs ship config-gated, off by default. | From 3d23f563b11826cde1fabd9c2a7a45c7cc2bf8ec Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 01:32:18 +0000 Subject: [PATCH 16/21] =?UTF-8?q?feat(geode):=20mirror=20directions=20?= =?UTF-8?q?=E2=80=94=20dir=20=3D=20"down"=20|=20"up"=20|=20"both"=20(strfr?= =?UTF-8?q?y-router=20parity)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Each [[mirror]] entry now takes strfry-router's dir: - down (default): pull — subscribe to the upstream and ingest, exactly as before. - up: push — an in-process session on the LOCAL relay subscribes with the same scoped filter (so backfill_seconds and filter behave identically in both directions, and the relay's own policy chain gates what leaves), and every matching event is handed to the client's outbox for the upstream, which owns delivery and re-sends across reconnects. - both: pull and push, with echo suppression: a per-upstream LRU of recently exchanged ids keeps an event pulled down from being pushed straight back (and vice versa when the upstream fans our own publish back). Eviction only costs a duplicate round trip — the stores' unique-id constraints stay the correctness backstop. trusted (verify skip) remains a down-only concept; the up direction never verifies since the upstream does its own gatekeeping. Tests: up pushes both the backfill window and the live tail; both converges two stores with disjoint content and holds exact counts after the echo settles (no ping-pong). Live-published test events are genuinely signed — the local publish path verifies, which is also what the debugging showed: the mechanism was fine, the first version of the tests was pushing forged events into a verifying relay. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- geode/config.example.toml | 6 + .../kotlin/com/vitorpamplona/geode/Main.kt | 7 + .../geode/config/StaticConfig.kt | 7 + .../geode/mirror/MirrorWorker.kt | 222 ++++++++++++++---- .../geode/mirror/MirrorWorkerTest.kt | 69 +++++- 5 files changed, 266 insertions(+), 45 deletions(-) diff --git a/geode/config.example.toml b/geode/config.example.toml index 17f15c8d31..1e5d9df79e 100644 --- a/geode/config.example.toml +++ b/geode/config.example.toml @@ -129,8 +129,14 @@ require_auth = false # everything; for several disjoint scopes, repeat [[mirror]] with the # same url. # +# `dir` (strfry-router parity) sets the flow direction: "down" pulls +# from the upstream (default), "up" pushes this relay's matching +# events to it, "both" does both with echo suppression so the two +# directions don't ping-pong the same event. +# # [[mirror]] # url = "wss://upstream.example.com/" +# dir = "both" # trusted = true # backfill_seconds = 3600 # filter = '{"kinds":[0,1,3,7],"#t":["nostr"]}' diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt index 4a7b1e6179..8bb7895249 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt @@ -24,6 +24,7 @@ import com.vitorpamplona.geode.config.BannedEntry import com.vitorpamplona.geode.config.RuntimeConfig import com.vitorpamplona.geode.config.RuntimeConfigData import com.vitorpamplona.geode.config.StaticConfig +import com.vitorpamplona.geode.mirror.MirrorDirection import com.vitorpamplona.geode.mirror.MirrorUpstream import com.vitorpamplona.geode.mirror.MirrorWorker import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper @@ -220,11 +221,17 @@ fun main(args: Array) { } parsed.copy(since = null, limit = null) } + val direction = + MirrorDirection.parse(m.dir) + ?: throw IllegalArgumentException( + "[[mirror]] dir for ${m.url} must be \"down\", \"up\" or \"both\" (got \"${m.dir}\")", + ) MirrorUpstream( url = m.url.normalizeRelayUrl(), trusted = m.trusted, backfillSeconds = m.backfill_seconds, filter = scope, + direction = direction, ) } require(upstreams.none { it.url == advertisedUrl }) { diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt index 7145139f57..c1559f6726 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt @@ -223,6 +223,13 @@ data class StaticConfig( * several `[[mirror]]` entries with the same url. */ val filter: String? = null, + /** + * Flow direction, strfry-router's `dir`: `"down"` (pull from the + * upstream — the default), `"up"` (push this relay's matching + * events to it), or `"both"`. Both-way mirrors suppress echoes + * (an event pulled down is not pushed straight back). + */ + val dir: String = "down", ) /** diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt index f87737d89c..60643a24a7 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt @@ -21,8 +21,11 @@ package com.vitorpamplona.geode.mirror import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper import com.vitorpamplona.quartz.nip01Core.relay.client.NostrClient import com.vitorpamplona.quartz.nip01Core.relay.client.reqs.SubscriptionListener +import com.vitorpamplona.quartz.nip01Core.relay.commands.toClient.EventMessage +import com.vitorpamplona.quartz.nip01Core.relay.commands.toRelay.ReqCmd import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.server.NostrServer @@ -44,6 +47,33 @@ import okhttp3.OkHttpClient import java.time.Duration import java.util.concurrent.atomic.AtomicLong +/** + * Which way events flow between this relay and one upstream — + * strfry-router's per-stream `dir`. + */ +enum class MirrorDirection { + /** Pull: subscribe to the upstream and ingest what it sends. */ + DOWN, + + /** Push: publish this relay's matching events to the upstream. */ + UP, + + /** Pull and push. Echo suppression keeps the two from ping-ponging. */ + BOTH, + ; + + companion object { + /** Parses strfry's `"down"` / `"up"` / `"both"`; null if unknown. */ + fun parse(value: String): MirrorDirection? = + when (value.lowercase()) { + "down" -> DOWN + "up" -> UP + "both" -> BOTH + else -> null + } + } +} + /** * One upstream relay this relay mirrors, from the `[[mirror]]` config. * @@ -51,23 +81,28 @@ import java.util.concurrent.atomic.AtomicLong * upstream skip Schnorr signature verification on ingest. The trusted * identity is [url] — the address *this* relay dialed (TLS-authenticated * for `wss://`), never anything the peer claims — so the skip can't be - * hijacked by an inbound client. + * hijacked by an inbound client. Only meaningful for [MirrorDirection.DOWN] + * / [MirrorDirection.BOTH]; the up direction never verifies (the upstream + * does its own gatekeeping). */ class MirrorUpstream( val url: NormalizedRelayUrl, val trusted: Boolean, - /** How far back the initial REQ reaches. 0 = live-only from connect. */ + /** How far back the initial replay reaches, in BOTH directions. 0 = live-only from connect. */ val backfillSeconds: Long = 0L, /** * Optional scope for this upstream (strfry-router's per-stream - * `filter`). Used twice: it shapes the REQ sent upstream, and every - * delivered event is re-checked against it before ingest — so even a - * [trusted] upstream can only inject events inside the declared - * scope. Its `since`/`limit` are ignored ([backfillSeconds] owns the - * time window; the subscription is unbounded). `null` mirrors + * `filter`). Applied symmetrically: down, it shapes the REQ sent + * upstream AND every delivered event is re-checked before ingest — + * so even a [trusted] upstream can only inject events inside the + * declared scope; up, it selects which local events are pushed. Its + * `since`/`limit` are ignored ([backfillSeconds] owns the time + * window; the subscriptions are unbounded). `null` mirrors * everything. */ val filter: Filter? = null, + /** Flow direction — strfry-router's `dir`. Defaults to pull-only. */ + val direction: MirrorDirection = MirrorDirection.DOWN, ) /** @@ -151,6 +186,38 @@ class MirrorWorker( */ val filtered = AtomicLong(0) + /** Local events handed to the client's outbox for an up-direction upstream. */ + val sentUp = AtomicLong(0) + + /** + * Recently exchanged event ids, one set per up-capable upstream — + * the echo suppressor for [MirrorDirection.BOTH]. An event pulled + * DOWN from an upstream must not be pushed straight back UP to it + * (and one we pushed up must not be re-ingested when the upstream + * fans it back on our down subscription). Bounded LRU: eviction only + * costs a wasted round trip that the stores' unique-id constraints + * absorb, so correctness never depends on it. + */ + private class RecentIds( + private val capacity: Int, + ) { + private val map = + object : LinkedHashMap(capacity, 0.75f, true) { + override fun removeEldestEntry(eldest: Map.Entry) = size > capacity + } + + @Synchronized + fun add(id: String) { + map[id] = true + } + + @Synchronized + fun contains(id: String): Boolean = map.containsKey(id) + } + + /** Open in-process sessions feeding the up direction; closed with the worker. */ + private val upSessions = mutableListOf() + /** Dials every upstream and starts streaming. Call once. */ fun start() { scope.launch { @@ -181,48 +248,25 @@ class MirrorWorker( val since = TimeUtils.now() upstreams.forEachIndexed { i, up -> - val listener = - object : SubscriptionListener { - override fun onEvent( - event: Event, - isLive: Boolean, - relay: NormalizedRelayUrl, - forFilters: List?, - ) { - // strfry-router parity: never take the upstream's - // word for what matched. Re-checking the configured - // scope here means even a trusted (skip-verify) - // upstream can only inject events the operator - // declared — the REQ shapes what we ask for, this - // shapes what we accept. - if (up.filter != null && !up.filter.match(event)) { - filtered.incrementAndGet() - Log.d("MirrorWorker") { "out-of-scope from ${relay.url}: ${event.id}" } - return - } - inbound.trySend(Inbound(event, up.trusted)) - } + // Echo suppression only matters when events can flow both + // ways on the same upstream. + val exchanged = if (up.direction == MirrorDirection.BOTH) RecentIds(EXCHANGED_IDS_CAPACITY) else null - override fun onCannotConnect( - relay: NormalizedRelayUrl, - message: String, - forFilters: List?, - ) { - Log.w("MirrorWorker") { "cannot reach upstream ${relay.url}: $message" } - } - } - // The operator's filter scopes the REQ; the mirror owns the - // time window (since) and never bounds the result (limit). - val reqFilter = + // The operator's filter scopes the subscriptions in both + // directions; the mirror owns the time window (since) and + // never bounds the result (limit). + val scopedFilter = (up.filter ?: Filter()).copy( since = since - up.backfillSeconds, limit = null, ) - client.subscribe( - subId = "geode-mirror-$i", - filters = mapOf(up.url to listOf(reqFilter)), - listener = listener, - ) + + if (up.direction != MirrorDirection.UP) { + startDown(i, up, scopedFilter, exchanged) + } + if (up.direction != MirrorDirection.DOWN) { + startUp(up, scopedFilter, exchanged) + } } client.connect() @@ -242,6 +286,87 @@ class MirrorWorker( } } + /** Down direction: subscribe to the upstream, ingest what it sends. */ + private fun startDown( + index: Int, + up: MirrorUpstream, + scopedFilter: Filter, + exchanged: RecentIds?, + ) { + val listener = + object : SubscriptionListener { + override fun onEvent( + event: Event, + isLive: Boolean, + relay: NormalizedRelayUrl, + forFilters: List?, + ) { + // strfry-router parity: never take the upstream's + // word for what matched. Re-checking the configured + // scope here means even a trusted (skip-verify) + // upstream can only inject events the operator + // declared — the REQ shapes what we ask for, this + // shapes what we accept. + if (up.filter != null && !up.filter.match(event)) { + filtered.incrementAndGet() + Log.d("MirrorWorker") { "out-of-scope from ${relay.url}: ${event.id}" } + return + } + // BOTH: an event we just pushed up is fanned back on + // this subscription — it already exists locally. + if (exchanged?.contains(event.id) == true) return + exchanged?.add(event.id) + inbound.trySend(Inbound(event, up.trusted)) + } + + override fun onCannotConnect( + relay: NormalizedRelayUrl, + message: String, + forFilters: List?, + ) { + Log.w("MirrorWorker") { "cannot reach upstream ${relay.url}: $message" } + } + } + client.subscribe( + subId = "geode-mirror-$index", + filters = mapOf(up.url to listOf(scopedFilter)), + listener = listener, + ) + } + + /** + * Up direction: an in-process session on the LOCAL relay subscribes + * with the same scoped filter — stored replay covers the backfill + * window, the live tail covers everything after — and each matching + * event is handed to the client's outbox for [MirrorUpstream.url]. + * The outbox owns delivery: it re-sends on reconnect until the + * upstream OKs, and the upstream's own duplicate handling absorbs + * replays. Going through a real session (not the store) means the + * relay's policy chain gates what leaves, same as any client. + */ + private fun startUp( + up: MirrorUpstream, + scopedFilter: Filter, + exchanged: RecentIds?, + ) { + val session = + server.connect { json -> + if (!json.startsWith("[\"EVENT\"")) return@connect + val event = + runCatching { (OptimizedJsonMapper.fromJsonToMessage(json) as? EventMessage)?.event } + .getOrNull() ?: return@connect + // BOTH: don't push back what we just pulled down. + if (exchanged?.contains(event.id) == true) return@connect + exchanged?.add(event.id) + client.publish(event, setOf(up.url)) + sentUp.incrementAndGet() + } + upSessions += AutoCloseable { session.close() } + scope.launch { + session.receive(OptimizedJsonMapper.toJson(ReqCmd("geode-mirror-up", listOf(scopedFilter)))) + } + } + /** * Stops pulling from every upstream. In-flight ingest submissions * drain through the server's queue; events still buffered in @@ -252,6 +377,7 @@ class MirrorWorker( override fun close() { // Close the client first so no listener callback races the // channel close below. + upSessions.forEach { runCatching { it.close() } } runCatching { client.close() } inbound.close() scope.cancel() @@ -265,5 +391,13 @@ class MirrorWorker( /** How often the retry pump nudges disconnected upstreams. */ const val RECONNECT_POKE_MS = 5_000L + + /** + * Per-upstream echo-suppression LRU size (BOTH direction only). + * Covers the burst window between pulling an event down and the + * up-session seeing its local fanout; eviction only costs a + * duplicate round trip. + */ + const val EXCHANGED_IDS_CAPACITY = 8_192 } } diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt index e2f62e2531..49bcb87785 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTest.kt @@ -80,9 +80,19 @@ class MirrorWorkerTest { private fun startMirror( trusted: Boolean, filter: Filter? = null, + direction: MirrorDirection = MirrorDirection.DOWN, ): MirrorWorker = MirrorWorker( - upstreams = listOf(MirrorUpstream(upstreamUrl, trusted = trusted, backfillSeconds = 3600, filter = filter)), + upstreams = + listOf( + MirrorUpstream( + upstreamUrl, + trusted = trusted, + backfillSeconds = 3600, + filter = filter, + direction = direction, + ), + ), server = downstream.server, websocketBuilder = hub, ).also { @@ -173,6 +183,63 @@ class MirrorWorkerTest { assertTrue(stored.all { it.kind == 1 }) } + @Test + fun upDirectionPushesLocalEventsToTheUpstream() = + runBlocking { + // Pre-existing local event: the up replay (backfill window) + // must carry it; then a live local publish must follow. + val preexisting = forgedEvent(1) + downstream.preload(preexisting) + + startMirror(trusted = false, direction = MirrorDirection.UP) + + val upstreamStore = hub.getOrCreate(upstreamUrl).store + await { upstreamStore.count(Filter()) >= 1 } + + // Live events go through the verifying publish path, so they + // must be genuinely signed (preload bypasses verification). + val live = signedEvent("up live") + downstream.publish(live) + await { upstreamStore.count(Filter()) >= 2 } + + val ids = upstreamStore.query(Filter()).map { it.id }.toSet() + assertEquals(setOf(preexisting.id, live.id), ids) + } + + @Test + fun bothDirectionConvergesWithoutPingPong() = + runBlocking { + // One event only the upstream has, one only the local relay + // has. BOTH must converge the two stores; echo suppression + // (plus store dedup as the backstop) must keep the shared + // events from bouncing. + val upstreamOnly = forgedEvent(1) + val localOnly = forgedEvent(2) + hub.getOrCreate(upstreamUrl).preload(upstreamOnly) + downstream.preload(localOnly) + + val mirror = startMirror(trusted = true, direction = MirrorDirection.BOTH) + + val upstreamStore = hub.getOrCreate(upstreamUrl).store + await { downstreamStore.count(Filter()) == 2 && upstreamStore.count(Filter()) == 2 } + + val expected = setOf(upstreamOnly.id, localOnly.id) + assertEquals(expected, downstreamStore.query(Filter()).map { it.id }.toSet()) + assertEquals(expected, upstreamStore.query(Filter()).map { it.id }.toSet()) + + // Live: an event published locally reaches the upstream AND + // its echo back down doesn't disturb either store. Signed, + // because the local publish path verifies. + val live = signedEvent("both live") + downstream.publish(live) + await { upstreamStore.count(Filter()) == 3 } + // Let any echo settle, then confirm counts are exact. + delay(500) + assertEquals(3, downstreamStore.count(Filter())) + assertEquals(3, upstreamStore.count(Filter())) + assertTrue(mirror.sentUp.get() >= 2) + } + @Test fun untrustedUpstreamStillVerifiesEverything() = runBlocking { From 6ac2903a3a52132984711ecab8eb4a999f22fde1 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 02:31:40 +0000 Subject: [PATCH 17/21] =?UTF-8?q?fix(quartz):=20live=20negentropy=20index?= =?UTF-8?q?=20=E2=80=94=20same-batch=20displacement,=20no-op=20kind-5,=20r?= =?UTF-8?q?ebuild=20cap?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Audit follow-ups on the live NIP-77 index, all with the store/scan equivalence test extended to cover them: - Same-batch replaceable displacement left a dead id in the index. applyAfterCommit applied all removes before all adds, so when a later row in one transaction displaced an earlier row of the same batch (two versions of one replaceable — the mirror-backfill hot path), the displaced row's remove no-op'd against an index that hadn't taken its add yet, then the add re-inserted it: the index advertised an id the trigger had already deleted. recordAccepted now cancels the pending add instead of queueing a remove (added is a LinkedHashSet for O(1) cancel). - A kind-5 that deleted nothing (a delete broadcast for events this relay never stored — the common case) still invalidated the whole index, forcing a full-scan rebuild under the writer mutex on the next NEG-OPEN. DeletionRequestModule.insert now returns the rows it deleted; recordAccepted only invalidates when that count is > 0, else records the kind-5 as a plain row. - The first NEG-OPEN over a corpus larger than the serve cap scanned the whole table uncapped, built a full index that could never produce a snapshot, and then maintained it forever for zero benefit. liveNegentropySnapshot now caps the rebuild scan at maxEntries + 1 and leaves the index unpopulated when the corpus is over-cap (the scan path answers NEG-ERR, as before). - delete/deleteExpired/clearDB invalidate() moved inside the writer mutex so no NEG-OPEN can seal a snapshot of just-deleted rows, and no concurrent rebuild can be discarded by a late invalidate. Also corrects the ~40 B/event heap figure to ~140 B (IdAndTime keeps the id as a 64-char hex string, not 32 bytes) in the strategy/plan docs. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../geode/RelayIndexingStrategy.kt | 4 +- ...26-07-03-incremental-negentropy-storage.md | 6 +- .../store/sqlite/DeletionRequestModule.kt | 17 ++- .../store/sqlite/IndexingStrategy.kt | 5 +- .../store/sqlite/SQLiteEventStore.kt | 94 +++++++++---- .../sqlite/LiveNegentropyIndexStoreTest.kt | 130 ++++++++++++++++++ 6 files changed, 214 insertions(+), 42 deletions(-) diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt index 9b3a116f3f..33fa9f4441 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt @@ -43,8 +43,8 @@ import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy * @param liveNegentropyIndex keep the always-current `(created_at, id)` * set that serves full-corpus NIP-77 NEG-OPENs without a scan + seal * (strfry answers those off its live tree; the scan+seal path measured - * ~340 ms per cold open at 50k events). Costs ~40 B/event of heap and - * one indexed pre-SELECT per replaceable insert. + * ~340 ms per cold open at 50k events). Costs ~140 B/event of JVM heap + * (hex-string ids) and one indexed pre-SELECT per replaceable insert. * `[negentropy].live_index = false` turns it off. */ fun relayIndexingStrategy( diff --git a/quartz/plans/2026-07-03-incremental-negentropy-storage.md b/quartz/plans/2026-07-03-incremental-negentropy-storage.md index ca10f6b091..d3e7ffcc80 100644 --- a/quartz/plans/2026-07-03-incremental-negentropy-storage.md +++ b/quartz/plans/2026-07-03-incremental-negentropy-storage.md @@ -60,8 +60,10 @@ are effectively always cold; any new filter is cold by definition. ### 1. `LiveNegentropyIndex` (quartz, server-only, opt-in) An always-current sorted set of `(createdAt, id₃₂)` maintained from the -store's write path — strfry's `MemoryView` equivalent, ~40 B/entry -(1M events ≈ 40 MB; capped by `negentropy.max_sync_events`). +store's write path — strfry's `MemoryView` equivalent, ~140 B/entry on +the JVM (`IdAndTime` keeps the id as a 64-char hex string; 1M events ≈ +140 MB; the index is only built when the corpus fits +`negentropy.max_sync_events`, so that also caps the heap). - **Structure**: single sorted array with binary-search insert. Nostr inserts are near-tail (created_at ≈ now), so the memmove is tiny in the common diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/DeletionRequestModule.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/DeletionRequestModule.kt index 30bb25f99e..448296a4ae 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/DeletionRequestModule.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/DeletionRequestModule.kt @@ -82,15 +82,18 @@ class DeletionRequestModule( fun insert( event: Event, db: SQLiteConnection, - ) { - if (event is DeletionEvent) { - val idValues = event.deleteEventIds() - val addresses = event.deleteAddresses() + ): Int { + if (event !is DeletionEvent) return 0 - deleteSQL(event.pubKey, event.createdAt, idValues, addresses, hasher(db)).forEach { delete -> - db.execSQL(delete.sql, delete.args) - } + val idValues = event.deleteEventIds() + val addresses = event.deleteAddresses() + + var deletedRows = 0 + deleteSQL(event.pubKey, event.createdAt, idValues, addresses, hasher(db)).forEach { delete -> + db.execSQL(delete.sql, delete.args) + deletedRows += db.changes() } + return deletedRows } /** diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt index aa7d38ca25..816f50f0f5 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt @@ -123,8 +123,9 @@ interface IndexingStrategy { * Maintain an always-current in-memory `(created_at, id)` set (a * [com.vitorpamplona.quartz.nip77Negentropy.LiveNegentropyIndex]) so * NIP-77 NEG-OPENs over the full corpus skip the scan + O(n log n) - * seal — strfry answers those off its live tree. Costs ~40 B/event - * of heap plus one indexed pre-SELECT per replaceable insert, which + * seal — strfry answers those off its live tree. Costs ~140 B/event + * of JVM heap (the id is kept as a 64-char hex string, not 32 bytes) + * plus one indexed pre-SELECT per replaceable insert, which * only makes sense on a *relay*; client-side stores don't serve * NEG-OPENs, so the default is **off**. */ diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt index 94179e9406..8ca31300f7 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SQLiteEventStore.kt @@ -214,8 +214,8 @@ class SQLiteEventStore( suspend fun clearDB() { pool.useWriter { db -> modules.reversed().forEach { it.deleteAll(db) } + liveNegentropyIndex?.invalidate() } - liveNegentropyIndex?.invalidate() } suspend fun vacuum() = @@ -278,7 +278,12 @@ class SQLiteEventStore( * when [liveNegentropyIndex] is off: zero cost on the write path. */ internal class LiveIndexDelta { - val added = ArrayList() + /** + * A set (not a list) so a row displaced by a LATER row of the + * same transaction can be cancelled in O(1) — see + * [recordAccepted]. Unique ids make duplicates impossible. + */ + val added = LinkedHashSet() val removed = ArrayList() /** @@ -308,13 +313,28 @@ class SQLiteEventStore( private fun LiveIndexDelta.recordAccepted( event: Event, displaced: IdAndTime?, + deletedRows: Int, ) { - if (event is DeletionEvent || (event is RequestToVanishEvent && event.shouldVanishFrom(relay))) { - // Their own row lands too, but the rebuild scan picks it up. + if (event is RequestToVanishEvent && event.shouldVanishFrom(relay)) { + // Vanish deletes by pubkey inside a trigger; not itemizable. + // Its own row lands too, but the rebuild scan picks it up. invalidateAll = true return } - displaced?.let { removed += it } + if (event is DeletionEvent && deletedRows > 0) { + // Only a kind-5 that actually removed rows costs a rebuild. + // The common case — a delete broadcast for events this relay + // never stored — deletes nothing and records as a plain row, + // so it can't be used to thrash the index. + invalidateAll = true + return + } + // A row displaced by this insert that was added earlier in the + // SAME transaction (two versions of one replaceable in one + // batch) was never in the live index: cancel its pending add. + // Queueing a remove instead would no-op against the index and + // let the stale add through — advertising a dead id. + displaced?.let { if (!added.remove(it)) removed += it } added += IdAndTime(event.createdAt, event.id) } @@ -386,11 +406,11 @@ class SQLiteEventStore( ) { val displaced = if (delta != null) displacedBy(event, db) else null val headerId = eventIndexModule.insert(event, db) - deletionModule.insert(event, db) + val deletedRows = deletionModule.insert(event, db) expirationModule.insert(event, headerId, db) fullTextSearchModule.insert(event, headerId, db) rightToVanishModule.insert(event, relay, headerId, db) - delta?.recordAccepted(event, displaced) + delta?.recordAccepted(event, displaced, deletedRows) } suspend fun insertEvent(event: Event) { @@ -540,35 +560,41 @@ class SQLiteEventStore( maxEntries: Int? = null, ): List = pool.useReader { queryBuilder.snapshotIdsForNegentropy(filters, it, maxEntries) } + // The invalidate() calls below run INSIDE useWriter: dropping the + // index while still holding the mutex means no NEG-OPEN can seal a + // snapshot that still advertises the just-deleted rows, and a + // concurrent rebuild (also writer-mutexed) can't complete only to + // have its fresh content thrown away by a late invalidate. + suspend fun delete(filter: Filter) { - pool.useWriter { queryBuilder.delete(filter, it) } - // Delete-by-filter can't itemize what it removed; drop the live - // index and let the next NEG-OPEN rebuild from one scan. - liveNegentropyIndex?.invalidate() + pool.useWriter { + queryBuilder.delete(filter, it) + // Delete-by-filter can't itemize what it removed; drop the + // live index and let the next NEG-OPEN rebuild from one scan. + liveNegentropyIndex?.invalidate() + } } suspend fun delete(filters: List) { - pool.useWriter { queryBuilder.delete(filters, it) } - liveNegentropyIndex?.invalidate() + pool.useWriter { + queryBuilder.delete(filters, it) + liveNegentropyIndex?.invalidate() + } } - suspend fun delete(id: HexKey): Int { - val changes = - pool.useWriter { db -> - db.execSQL("DELETE FROM event_headers WHERE id = ?", arrayOf(id)) - db.changes() - } - if (changes > 0) liveNegentropyIndex?.invalidate() - return changes - } + suspend fun delete(id: HexKey): Int = + pool.useWriter { db -> + db.execSQL("DELETE FROM event_headers WHERE id = ?", arrayOf(id)) + val changes = db.changes() + if (changes > 0) liveNegentropyIndex?.invalidate() + changes + } suspend fun deleteExpiredEvents() { - val swept = - pool.useWriter { db -> - expirationModule.deleteExpiredEvents(db) - db.changes() - } - if (swept > 0) liveNegentropyIndex?.invalidate() + pool.useWriter { db -> + expirationModule.deleteExpiredEvents(db) + if (db.changes() > 0) liveNegentropyIndex?.invalidate() + } } /** @@ -585,7 +611,17 @@ class SQLiteEventStore( if (!index.isPopulated()) { pool.useWriter { db -> if (!index.isPopulated()) { - index.rebuild(queryBuilder.snapshotIdsForNegentropy(listOf(Filter()), db, null)) + // The scan is capped at maxEntries + 1: a corpus over + // the serve cap can never produce a snapshot (callers + // answer NEG-ERR from the scan path instead), so + // building — and then maintaining — a full index for + // it would cost heap and write-path work for zero + // benefit. Leaving it unpopulated also keeps ingest + // delta-free (see newDeltaOrNull). + val scan = queryBuilder.snapshotIdsForNegentropy(listOf(Filter()), db, maxEntries) + if (scan.size <= maxEntries) { + index.rebuild(scan) + } } } } diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt index 3102f162de..ae04a670df 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/LiveNegentropyIndexStoreTest.kt @@ -193,6 +193,136 @@ class LiveNegentropyIndexStoreTest { store.close() } + @Test + fun sameBatchReplaceableOverwriteDropsTheDisplacedId() = + runTest { + val store = newStore() + store.insert(event(1)) + // Populate the index first so the batch runs the DELTA path + // (an unpopulated index would be covered by the rebuild scan + // and could never expose this). + assertIndexMatchesScan(store) + + // Three versions of one replaceable in ONE batch: each later + // row displaces the previous one INSIDE the same transaction. + // The displaced rows were never in the index, so their + // pending adds must be cancelled — a remove would no-op and + // leave the index advertising dead ids. + val outcomes = + store.batchInsert( + listOf( + event(2, kind = 0, createdAt = 100, pubKey = pubkey(7)), + event(3, kind = 0, createdAt = 200, pubKey = pubkey(7)), + event(4, kind = 0, createdAt = 300, pubKey = pubkey(7)), + ), + ) + assertTrue(outcomes.all { it == IEventStore.InsertOutcome.Accepted }) + assertEquals(1, store.count(Filter(kinds = listOf(0)))) + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun sameTransactionReplaceableOverwriteDropsTheDisplacedId() = + runTest { + val store = newStore() + store.insert(event(1)) + assertIndexMatchesScan(store) + + store.transaction { + insert(event(2, kind = 10002, createdAt = 100, pubKey = pubkey(8))) + insert(event(3, kind = 10002, createdAt = 200, pubKey = pubkey(8))) + } + assertEquals(1, store.count(Filter(kinds = listOf(10002)))) + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun sameBatchAddressableOverwriteDropsTheDisplacedId() = + runTest { + val store = newStore() + store.insert(event(1)) + assertIndexMatchesScan(store) + + val author = pubkey(9) + store.batchInsert( + listOf( + event(2, kind = 30023, createdAt = 100, pubKey = author, tags = arrayOf(arrayOf("d", "post-a"))), + event(3, kind = 30023, createdAt = 200, pubKey = author, tags = arrayOf(arrayOf("d", "post-a"))), + ), + ) + assertEquals(1, store.count(Filter(kinds = listOf(30023)))) + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun noOpDeletionKeepsTheIndexLive() = + runTest { + val store = newStore() + store.insert(event(1)) + assertIndexMatchesScan(store) + + // A kind-5 referencing an id this store never had — the + // common case for broadcast deletes — removes nothing, so it + // must NOT invalidate the index (that would be a free way to + // force full-scan rebuilds); it lands as a plain row. + store.insert( + event(2, kind = 5, createdAt = 120, tags = arrayOf(arrayOf("e", hexId(999)))), + ) + val index = assertNotNull(store.store.liveNegentropyIndex) + assertTrue(index.isPopulated(), "no-op kind-5 must not invalidate the live index") + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun effectiveDeletionStillInvalidates() = + runTest { + val store = newStore() + val author = pubkey(3) + val target = event(1, createdAt = 100, pubKey = author) + store.insert(target) + assertIndexMatchesScan(store) + + store.insert( + event(2, kind = 5, createdAt = 120, pubKey = author, tags = arrayOf(arrayOf("e", target.id))), + ) + val index = assertNotNull(store.store.liveNegentropyIndex) + assertTrue(!index.isPopulated(), "a kind-5 that deleted rows must invalidate") + assertIndexMatchesScan(store) + + store.close() + } + + @Test + fun overCapCorpusNeverBuildsTheIndex() = + runTest { + val store = newStore() + store.insert(event(1)) + store.insert(event(2)) + store.insert(event(3)) + + // First-ever NEG-OPEN over a corpus above the serve cap: the + // rebuild scan is capped, the index stays unpopulated (no + // heap, no write-path deltas), and the caller falls back. + assertNull(store.liveNegentropySnapshot(2)) + val index = assertNotNull(store.store.liveNegentropyIndex) + assertTrue(!index.isPopulated(), "an over-cap corpus must not build the index") + + // A cap the corpus fits builds and serves as usual. + assertNotNull(store.liveNegentropySnapshot(3)) + assertTrue(index.isPopulated()) + assertIndexMatchesScan(store) + + store.close() + } + @Test fun overCapFallsBackToNull() = runTest { From 2be0c9f8806a721c36b81a0f0901c79796e4e073 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 02:31:54 +0000 Subject: [PATCH 18/21] fix(quartz): NostrServer.ingest verifies inline when the queue hook is off MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit ingest() bypasses the per-connection policy chain, which is where VerifyPolicy lives. The IngestQueue verify hook only exists when parallelVerify is true, so with parallelVerify = false and skipVerify = false, ingest() previously verified nothing — an untrusted mirror upstream on a relay running the legacy in-policy verify path could inject forgeries. ingest() now verifies inline in that configuration (same rejection reason as the queue), so the documented "default keeps verify-everything semantics" holds regardless of parallelVerify. KDoc also spells out that ingest() skips the entire policy chain (blacklists, size limits), which callers must screen for themselves. Test covers the parallelVerify = false server: forged rejected, trusted skip and valid still land. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../nip01Core/relay/server/NostrServer.kt | 29 +++++++++++----- .../relay/server/NostrServerIngestTest.kt | 33 +++++++++++++++++-- 2 files changed, 52 insertions(+), 10 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt index 81326a5bbf..9ec35db402 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServer.kt @@ -65,7 +65,7 @@ class NostrServer( private val store: IEventStore, policyBuilder: () -> IRelayPolicy = { VerifyPolicy }, parentContext: CoroutineContext = SupervisorJob(), - parallelVerify: Boolean = false, + private val parallelVerify: Boolean = false, negentropySettings: NegentropySettings = NegentropySettings.Default, listener: RelayServerListener = RelayServerListener.None, limits: RelayLimits? = null, @@ -105,19 +105,32 @@ class NostrServer( * connection — e.g. a mirror worker streaming a trusted upstream * relay, or an import job. Routes through the same group-commit * [IngestQueue] and live fanout as a client EVENT publish, but skips - * the per-connection policy chain (there is no connection). + * the **entire** per-connection policy chain (there is no + * connection): no [VerifyPolicy], no allow/deny lists, no size + * limits. Callers own that screening — scope what may enter this + * path (e.g. geode's per-upstream mirror filters) accordingly. * - * [skipVerify] exempts the event from the parallel signature-verify - * hook — the relay-to-relay trust model: pass `true` only for events - * from an explicitly configured upstream that already verified them - * (Schnorr verify profiles at ~8% of busy ingest CPU). The default - * `false` keeps verify-everything semantics. + * [skipVerify] exempts the event from signature verification — the + * relay-to-relay trust model: pass `true` only for events from an + * explicitly configured upstream that already verified them (Schnorr + * verify profiles at ~8% of busy ingest CPU). The default `false` + * keeps verify-everything semantics regardless of configuration: + * when the [IngestQueue] hook is on ([parallelVerify]) it verifies + * there, otherwise this method verifies inline — without this, a + * server whose verification lives in the (bypassed) policy chain + * would silently ingest forgeries. */ suspend fun ingest( event: Event, skipVerify: Boolean = false, onComplete: (IEventStore.InsertOutcome) -> Unit, - ) = liveStore.submit(event, skipVerify, onComplete) + ) { + if (!skipVerify && !parallelVerify && !event.verify()) { + onComplete(IEventStore.InsertOutcome.Rejected("invalid: bad signature or id")) + return + } + liveStore.submit(event, skipVerify, onComplete) + } init { // Deferred-FTS catch-up worker: tokenizes in the gaps between diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt index 085edb3098..f1a881432e 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/relay/server/NostrServerIngestTest.kt @@ -59,12 +59,15 @@ class NostrServerIngestTest { sig = "f".repeat(128), ) - private fun createServer(dispatcher: CoroutineDispatcher): NostrServer = + private fun createServer( + dispatcher: CoroutineDispatcher, + parallelVerify: Boolean = true, + ): NostrServer = NostrServer( store = EventStore(null), policyBuilder = { EmptyPolicy }, parentContext = dispatcher, - parallelVerify = true, + parallelVerify = parallelVerify, ) private suspend fun NostrServer.ingestOutcome( @@ -118,6 +121,32 @@ class NostrServerIngestTest { server.close() } + @Test + fun forgedEventIsRejectedEvenWithoutTheQueueVerifyHook() = + runTest { + // With parallelVerify = false the IngestQueue has no verify + // hook AND ingest() bypasses the policy chain where + // VerifyPolicy would live — the inline fallback is the only + // thing standing between an untrusted mirror and forgeries. + val server = createServer(UnconfinedTestDispatcher(testScheduler), parallelVerify = false) + + val outcome = server.ingestOutcome(forgedEvent(), skipVerify = false) + assertTrue(outcome is IEventStore.InsertOutcome.Rejected) + assertTrue(outcome.reason.contains("signature")) + + // The trust switch and valid events still work on this config. + assertEquals( + IEventStore.InsertOutcome.Accepted, + server.ingestOutcome(forgedEvent(2), skipVerify = true), + ) + assertEquals( + IEventStore.InsertOutcome.Accepted, + server.ingestOutcome(signedEvent(), skipVerify = false), + ) + + server.close() + } + @Test fun trustIsPerSubmissionNotPerQueue() = runTest { From 9bb75d9bbdc32190d281fef7debcb778592ad71f Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 02:32:08 +0000 Subject: [PATCH 19/21] fix(quartz/desktop): close socket on connect failure; NODELAY for desktop pre-init client TcpNoDelaySocketFactory's connecting overloads used `socket().apply { connect(...) }`, which leaks the file descriptor if bind/connect throws (the JDK's connecting Socket constructors close on failure; ours didn't). Wrapped in a helper that closes on throw. OkHttp only calls the no-arg overload, so this guards any other direct caller. DesktopHttpClient's pre-init `simpleClient` (direct relay sockets opened before setInstance) now gets the same TcpNoDelaySocketFactory as directClient. failClosedClient is left as-is: it's a SOCKS client and OkHttp bypasses the socket factory for SOCKS proxies. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../desktop/network/DesktopHttpClient.kt | 6 ++++ .../sockets/okhttp/TcpNoDelaySocketFactory.kt | 35 ++++++++++++++----- 2 files changed, 33 insertions(+), 8 deletions(-) diff --git a/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt b/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt index 6f4c290668..ea31213db2 100644 --- a/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt +++ b/desktopApp/src/jvmMain/kotlin/com/vitorpamplona/amethyst/desktop/network/DesktopHttpClient.kt @@ -154,6 +154,12 @@ class DesktopHttpClient( private val simpleClient: OkHttpClient by lazy { OkHttpClient .Builder() + // Direct relay sockets opened before setInstance() should + // get the same TCP_NODELAY as directClient — see + // TcpNoDelaySocketFactory. (failClosedClient is SOCKS, and + // OkHttp bypasses the socket factory for SOCKS proxies, so + // it doesn't need this.) + .socketFactory(TcpNoDelaySocketFactory) .connectTimeout(BASE_TIMEOUT_SECONDS, TimeUnit.SECONDS) .readTimeout(BASE_TIMEOUT_SECONDS, TimeUnit.SECONDS) .writeTimeout(BASE_TIMEOUT_SECONDS, TimeUnit.SECONDS) diff --git a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt index 2f175a3b3b..6f581eb893 100644 --- a/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt +++ b/quartz/src/jvmAndroid/kotlin/com/vitorpamplona/quartz/nip01Core/relay/sockets/okhttp/TcpNoDelaySocketFactory.kt @@ -47,12 +47,31 @@ import javax.net.SocketFactory object TcpNoDelaySocketFactory : SocketFactory() { private fun socket() = Socket().apply { tcpNoDelay = true } + /** + * Runs [block] on a fresh NODELAY socket, closing it if the + * bind/connect throws — the JDK's connecting `Socket(...)` + * constructors do this, but our `Socket().apply { connect(...) }` + * form would otherwise leak the descriptor on a failed connect. + * (OkHttp only calls the no-arg overload, so this guards the other + * overloads for any direct caller.) + */ + private inline fun connecting(block: (Socket) -> Unit): Socket { + val s = socket() + try { + block(s) + } catch (t: Throwable) { + runCatching { s.close() } + throw t + } + return s + } + override fun createSocket(): Socket = socket() override fun createSocket( host: String?, port: Int, - ): Socket = socket().apply { connect(InetSocketAddress(host, port)) } + ): Socket = connecting { it.connect(InetSocketAddress(host, port)) } override fun createSocket( host: String?, @@ -60,15 +79,15 @@ object TcpNoDelaySocketFactory : SocketFactory() { localHost: InetAddress?, localPort: Int, ): Socket = - socket().apply { - bind(InetSocketAddress(localHost, localPort)) - connect(InetSocketAddress(host, port)) + connecting { + it.bind(InetSocketAddress(localHost, localPort)) + it.connect(InetSocketAddress(host, port)) } override fun createSocket( host: InetAddress?, port: Int, - ): Socket = socket().apply { connect(InetSocketAddress(host, port)) } + ): Socket = connecting { it.connect(InetSocketAddress(host, port)) } override fun createSocket( address: InetAddress?, @@ -76,8 +95,8 @@ object TcpNoDelaySocketFactory : SocketFactory() { localAddress: InetAddress?, localPort: Int, ): Socket = - socket().apply { - bind(InetSocketAddress(localAddress, localPort)) - connect(InetSocketAddress(address, port)) + connecting { + it.bind(InetSocketAddress(localAddress, localPort)) + it.connect(InetSocketAddress(address, port)) } } From 401f36ee82761d9ced67ed243aed472f1ee99495 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 02:32:21 +0000 Subject: [PATCH 20/21] fix(geode): mirror trust bound to origin relay; reconnect advances since watermark Two mirror hardening fixes from the audit: - Trust leak: the skip-verify decision was keyed on the subscription id, but the client pool dispatches EVENTs by subscription id alone and every [[mirror]] upstream shares one client. A hostile untrusted upstream could answer with the trusted upstream's subscription id and ride its skip-verify into the store. startDown now drops any event whose delivering relay isn't the one that subscription dialed (relay != up.url). New MirrorWorkerTrustOriginTest injects a foreign subId frame from a hostile relay and asserts it never lands. - Reconnect replay: `since` was frozen at boot, so a long-lived daemon re-streamed the whole backfill window on every upstream flap. The down path now tracks the newest ingested created_at and, on the connected->disconnected edge, advances the REQ's since to watermark - overlap before the reconnect re-sends it. Advancing only on disconnect means a stable link never re-queries. The reconnect test now asserts the watermark advanced. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- .../geode/mirror/MirrorWorker.kt | 123 +++++++++++-- .../geode/mirror/MirrorWorkerReconnectTest.kt | 7 + .../mirror/MirrorWorkerTrustOriginTest.kt | 165 ++++++++++++++++++ 3 files changed, 280 insertions(+), 15 deletions(-) create mode 100644 geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTrustOriginTest.kt diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt index 60643a24a7..eae9d20b0c 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/mirror/MirrorWorker.kt @@ -180,15 +180,63 @@ class MirrorWorker( val rejected = AtomicLong(0) /** - * Deliveries dropped by the [MirrorUpstream.filter] re-check before - * ever reaching the store — an upstream sending these is answering - * outside the REQ it was given. + * Deliveries dropped before ever reaching the store: events outside + * the [MirrorUpstream.filter] scope (an upstream answering outside + * the REQ it was given) and events delivered by a different relay + * than the one the subscription dialed (a peer answering with a + * subscription id that isn't its own). */ val filtered = AtomicLong(0) /** Local events handed to the client's outbox for an up-direction upstream. */ val sentUp = AtomicLong(0) + /** + * How many times a down subscription advanced its `since` watermark + * on a reconnect (test/observability hook). Each advance is one + * avoided full-window replay. + */ + val sinceAdvances = AtomicLong(0) + + /** + * One down subscription's re-subscribe state. The upstream replays + * everything at or after the REQ's `since` on every (re)connect; left + * at the boot-time value, a month-old daemon re-streams a month of + * events on every flap. So we track the newest `created_at` ingested + * from this upstream and, on a disconnect, advance the REQ's `since` + * to `watermark - overlap` before the reconnect re-sends it. The + * overlap re-requests a small tail (dup-safe against the store's + * unique-id constraint) to cover out-of-order streaming and clock + * skew. Advancing only on disconnect keeps a healthy connection from + * ever re-querying — no steady-state cost. + */ + private inner class DownSub( + val subId: String, + val up: MirrorUpstream, + val scopedBase: Filter, + val listener: SubscriptionListener, + initialSince: Long, + val watermark: AtomicLong, + ) { + @Volatile + var issuedSince: Long = initialSince + + fun advanceSinceOnReconnect() { + val candidate = watermark.get() - WATERMARK_OVERLAP_SECS + if (candidate > issuedSince) { + issuedSince = candidate + client.subscribe( + subId = subId, + filters = mapOf(up.url to listOf(scopedBase.copy(since = candidate))), + listener = listener, + ) + sinceAdvances.incrementAndGet() + } + } + } + + private val downSubs = mutableListOf() + /** * Recently exchanged event ids, one set per up-capable upstream — * the echo suppressor for [MirrorDirection.BOTH]. An event pulled @@ -254,22 +302,37 @@ class MirrorWorker( // The operator's filter scopes the subscriptions in both // directions; the mirror owns the time window (since) and - // never bounds the result (limit). - val scopedFilter = - (up.filter ?: Filter()).copy( - since = since - up.backfillSeconds, - limit = null, - ) + // never bounds the result (limit). scopedBase carries no + // since — the down path applies (and later advances) it via + // the watermark; the up path uses the fixed initial value. + val scopedBase = (up.filter ?: Filter()).copy(since = null, limit = null) + val initialSince = since - up.backfillSeconds if (up.direction != MirrorDirection.UP) { - startDown(i, up, scopedFilter, exchanged) + downSubs += startDown(i, up, scopedBase, initialSince, exchanged) } if (up.direction != MirrorDirection.DOWN) { - startUp(up, scopedFilter, exchanged) + startUp(up, scopedBase.copy(since = initialSince), exchanged) } } client.connect() + // Watermark advance: when an upstream drops, bump its REQ's since + // to the newest event we ingested from it (minus an overlap) so + // the reconnect doesn't replay the whole window since boot. Only + // fires on the connected→disconnected edge, so a stable link + // never re-queries. + scope.launch { + var prev = emptySet() + client.connectedRelaysFlow().collect { current -> + val dropped = prev - current + if (dropped.isNotEmpty()) { + downSubs.forEach { if (it.up.url in dropped) it.advanceSinceOnReconnect() } + } + prev = current + } + } + // Retry pump. NostrClient re-dials once on disconnect and then // relies on its 60s keep-alive — measured as a 61s mirror blackout // when an upstream restarts and the immediate re-dial races the @@ -290,9 +353,14 @@ class MirrorWorker( private fun startDown( index: Int, up: MirrorUpstream, - scopedFilter: Filter, + scopedBase: Filter, + initialSince: Long, exchanged: RecentIds?, - ) { + ): DownSub { + // watermark tracks the newest created_at ingested from this + // upstream; seeded at initialSince so a still-catching-up + // backfill can never advance the since backwards. + val watermark = AtomicLong(initialSince) val listener = object : SubscriptionListener { override fun onEvent( @@ -301,6 +369,18 @@ class MirrorWorker( relay: NormalizedRelayUrl, forFilters: List?, ) { + // The trusted identity is the relay THIS subscription + // dialed. The pool dispatches EVENTs by subscription + // id alone and every upstream shares this one client, + // so a hostile co-configured upstream could answer + // with another subscription's id and ride its trust — + // bind the decision to the delivering relay, not the + // sub id. + if (relay != up.url) { + filtered.incrementAndGet() + Log.w("MirrorWorker") { "dropped event delivered by ${relay.url} on ${up.url.url}'s subscription: ${event.id}" } + return + } // strfry-router parity: never take the upstream's // word for what matched. Re-checking the configured // scope here means even a trusted (skip-verify) @@ -316,6 +396,8 @@ class MirrorWorker( // this subscription — it already exists locally. if (exchanged?.contains(event.id) == true) return exchanged?.add(event.id) + // Advance the reconnect watermark past this event. + watermark.updateAndGet { if (event.createdAt > it) event.createdAt else it } inbound.trySend(Inbound(event, up.trusted)) } @@ -327,11 +409,13 @@ class MirrorWorker( Log.w("MirrorWorker") { "cannot reach upstream ${relay.url}: $message" } } } + val subId = "geode-mirror-$index" client.subscribe( - subId = "geode-mirror-$index", - filters = mapOf(up.url to listOf(scopedFilter)), + subId = subId, + filters = mapOf(up.url to listOf(scopedBase.copy(since = initialSince))), listener = listener, ) + return DownSub(subId, up, scopedBase, listener, initialSince, watermark) } /** @@ -392,6 +476,15 @@ class MirrorWorker( /** How often the retry pump nudges disconnected upstreams. */ const val RECONNECT_POKE_MS = 5_000L + /** + * Overlap (seconds) subtracted from the reconnect `since` + * watermark. Re-requests a small tail on reconnect to cover + * out-of-order streaming and clock skew; the store's unique-id + * constraint drops the duplicates. Bounds worst-case replay after + * a flap to this window instead of everything since boot. + */ + const val WATERMARK_OVERLAP_SECS = 300L + /** * Per-upstream echo-suppression LRU size (BOTH direction only). * Covers the burst window between pulling an event down and the diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt index 9268ca0384..bb6b007287 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerReconnectTest.kt @@ -34,6 +34,7 @@ import kotlinx.coroutines.withTimeout import org.junit.After import kotlin.test.Test import kotlin.test.assertEquals +import kotlin.test.assertTrue /** * The mirror is a long-running daemon, so surviving its upstream is not @@ -152,6 +153,12 @@ class MirrorWorkerReconnectTest { // the store, not double-inserted. assertEquals(2, downstreamStore.count(Filter())) + // The disconnect advanced the REQ's since watermark: the + // reconnect subscribed from ~(newest event - overlap), not + // from the boot-time `now - backfill`. Without this the + // reconnect would replay the whole backfill window every flap. + assertTrue(mirror.sinceAdvances.get() >= 1) + second.stop(gracePeriodMillis = 0, timeoutMillis = 1_000) } } diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTrustOriginTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTrustOriginTest.kt new file mode 100644 index 0000000000..28d7701ebc --- /dev/null +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/mirror/MirrorWorkerTrustOriginTest.kt @@ -0,0 +1,165 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.geode.mirror + +import com.vitorpamplona.geode.InProcessRelays +import com.vitorpamplona.geode.RelayEngine +import com.vitorpamplona.geode.testing.publish +import com.vitorpamplona.quartz.nip01Core.core.Event +import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer +import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocket +import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebSocketListener +import com.vitorpamplona.quartz.nip01Core.relay.sockets.WebsocketBuilder +import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore +import com.vitorpamplona.quartz.utils.TimeUtils +import kotlinx.coroutines.delay +import kotlinx.coroutines.runBlocking +import kotlinx.coroutines.withTimeout +import org.junit.After +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * The `trusted = true` skip-verify decision must be bound to the relay + * that DELIVERED the event, not to the subscription id it arrived on. + * The client pool dispatches EVENT messages by subscription id alone and + * every `[[mirror]]` upstream shares one client — so a hostile untrusted + * upstream that answers with the *trusted* upstream's subscription id + * would otherwise ride its skip-verify straight into the store. + */ +class MirrorWorkerTrustOriginTest { + private val trustedUrl = RelayUrlNormalizer.normalize("ws://trusted.relay/") + private val hostileUrl = RelayUrlNormalizer.normalize("ws://hostile.relay/") + private val downstreamUrl = RelayUrlNormalizer.normalize("ws://downstream.relay/") + + private val hub = InProcessRelays() + + private val downstreamStore = EventStore(null) + private val downstream = + RelayEngine( + url = downstreamUrl, + store = downstreamStore, + parallelVerify = true, + ) + + private var worker: MirrorWorker? = null + + @After + fun tearDown() { + worker?.close() + downstream.close() + hub.close() + } + + private fun forgedEvent(idSeed: Int): Event = + Event( + id = idSeed.toString().padStart(64, '0'), + pubKey = "1".repeat(64), + createdAt = TimeUtils.now() - idSeed, + kind = 1, + tags = emptyArray(), + content = "forged $idSeed", + sig = "f".repeat(128), + ) + + /** + * A relay that never answers the REQ it was given; instead, on + * connect it injects one EVENT frame carrying a subscription id that + * belongs to a DIFFERENT upstream's subscription. + */ + private class HijackingWebSocket( + private val out: WebSocketListener, + private val frame: String, + ) : WebSocket { + private var connected = false + + override fun needsReconnect(): Boolean = !connected + + override fun connect() { + connected = true + out.onOpen(0, false) + out.onMessage(frame) + } + + override fun disconnect() { + connected = false + } + + override fun send(msg: String): Boolean = true + } + + @Test + fun hostileUpstreamCannotRideAnotherSubscriptionsTrust() = + runBlocking { + val forged = forgedEvent(1) + // The trusted upstream is configured FIRST, so its down + // subscription id is "geode-mirror-0" — which the hostile + // relay claims in its injected frame. + val hijackFrame = """["EVENT","geode-mirror-0",${forged.toJson()}]""" + + val builder = + object : WebsocketBuilder { + override fun build( + url: NormalizedRelayUrl, + out: WebSocketListener, + ): WebSocket = + if (url == hostileUrl) { + HijackingWebSocket(out, hijackFrame) + } else { + hub.build(url, out) + } + } + + val mirror = + MirrorWorker( + upstreams = + listOf( + MirrorUpstream(trustedUrl, trusted = true, backfillSeconds = 3600), + MirrorUpstream(hostileUrl, trusted = false, backfillSeconds = 3600), + ), + server = downstream.server, + websocketBuilder = builder, + ).also { worker = it } + mirror.start() + + // The injected frame must be dropped at the origin check — + // observable as a `filtered` tick, never as a stored row. + withTimeout(15_000) { + while (mirror.filtered.get() < 1) delay(25) + } + assertTrue(downstreamStore.query(Filter(ids = listOf(forged.id))).isEmpty()) + + // The genuinely trusted upstream still works on the same + // subscription id the hostile relay tried to claim. + val real = forgedEvent(2) + hub.getOrCreate(trustedUrl).publish(real) + withTimeout(15_000) { + while (downstreamStore.count(Filter()) < 1) delay(25) + } + assertEquals( + listOf(real.id), + downstreamStore.query(Filter()).map { it.id }, + ) + } +} From bae2031cf34e974c4bd245eee7dab9a58e3c932c Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 4 Jul 2026 02:32:35 +0000 Subject: [PATCH 21/21] fix(geode): validate config knobs and mirror filter at boot MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fail-loud on config that would silently degrade the running relay: - readers = 0 makes every query hang forever on an empty reader pool; readers < 0 crashes with an unrelated message. optimize_interval_seconds <= 0 busy-loops PRAGMA optimize under the writer mutex. StaticConfig .validate() (called at boot) rejects both. - A typo in [[mirror]].filter (e.g. `kindss`) parsed to an empty match-everything filter through the tolerant deserializer — silently widening a trusted upstream's skip-verify scope to the whole firehose. MirrorFilterValidator strict-checks the filter JSON at boot: unknown keys and non-array list fields fail startup. - The self-mirror guard now compares scheme-insensitively (ws:// vs wss:// for the same host is still us) via displayUrl(). - The maintenance loop rethrows CancellationException and no longer swallows Errors. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01TtDNpayEYvJH7QuPswND3A --- geode/config.example.toml | 4 +- .../kotlin/com/vitorpamplona/geode/Main.kt | 32 +++++++- .../geode/config/MirrorFilterValidator.kt | 78 +++++++++++++++++++ .../geode/config/StaticConfig.kt | 24 +++++- .../geode/config/StaticConfigTest.kt | 53 +++++++++++++ 5 files changed, 187 insertions(+), 4 deletions(-) create mode 100644 geode/src/main/kotlin/com/vitorpamplona/geode/config/MirrorFilterValidator.kt diff --git a/geode/config.example.toml b/geode/config.example.toml index 1e5d9df79e..8ae62edc10 100644 --- a/geode/config.example.toml +++ b/geode/config.example.toml @@ -127,7 +127,9 @@ require_auth = false # inject events inside the declared scope. `since`/`limit` inside it # are ignored (backfill_seconds owns the time window). Omit to mirror # everything; for several disjoint scopes, repeat [[mirror]] with the -# same url. +# same url. The keys are validated at boot (a typo like `kindss` or a +# scalar where an array belongs fails startup) precisely because this +# filter is the trust boundary for `trusted = true`. # # `dir` (strfry-router parity) sets the flow direction: "down" pulls # from the upstream (default), "up" pushes this relay's matching diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt index 8bb7895249..2ff1e4c866 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/Main.kt @@ -21,6 +21,7 @@ package com.vitorpamplona.geode import com.vitorpamplona.geode.config.BannedEntry +import com.vitorpamplona.geode.config.MirrorFilterValidator import com.vitorpamplona.geode.config.RuntimeConfig import com.vitorpamplona.geode.config.RuntimeConfigData import com.vitorpamplona.geode.config.StaticConfig @@ -30,6 +31,7 @@ import com.vitorpamplona.geode.mirror.MirrorWorker import com.vitorpamplona.quartz.nip01Core.core.OptimizedJsonMapper import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import com.vitorpamplona.quartz.nip01Core.relay.normalizer.NormalizedRelayUrl +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.displayUrl import com.vitorpamplona.quartz.nip01Core.relay.normalizer.normalizeRelayUrl import com.vitorpamplona.quartz.nip01Core.relay.server.policies.EmptyPolicy import com.vitorpamplona.quartz.nip01Core.relay.server.policies.FullAuthPolicy @@ -40,6 +42,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.server.policies.VerifyAuthOnlyPo import com.vitorpamplona.quartz.nip01Core.relay.server.policies.VerifyPolicy import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore import com.vitorpamplona.quartz.nip77Negentropy.NegentropySettings +import kotlinx.coroutines.CancellationException import kotlinx.coroutines.CoroutineScope import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.SupervisorJob @@ -95,6 +98,7 @@ fun main(args: Array) { .opt("--config") ?.let { StaticConfig.fromFile(File(it)) } ?: StaticConfig() + config.validate() val host = a.opt("--host") ?: config.network.host val port = a.opt("--port")?.toInt() ?: config.network.port @@ -210,6 +214,14 @@ fun main(args: Array) { // (backfill_seconds) and never bounds the subscription. val scope = m.filter?.let { json -> + // Strict-validate FIRST: the deserializer is tolerant + // (unknown keys skipped, wrong-typed entries dropped), + // so a typo like `{"kindss":[4]}` would silently parse + // to an empty filter — widening a trusted upstream's + // scope to the whole firehose. This filter is the + // containment boundary for `trusted = true`, so a typo + // must fail the boot, not the boundary. + MirrorFilterValidator.validate(m.url, json) val parsed = try { OptimizedJsonMapper.fromJsonTo(json) @@ -234,7 +246,14 @@ fun main(args: Array) { direction = direction, ) } - require(upstreams.none { it.url == advertisedUrl }) { + // Never mirror ourselves — a self-URL echoes every local publish + // back forever. Compare scheme-insensitively (ws:// vs wss:// for the + // same host is still us) and ignoring the trailing slash. This can't + // catch a public URL that resolves to this bind behind a proxy, nor a + // `--port 0` autobind, so it's a guardrail against the obvious typo, + // not a proof of non-self-reference. + val advertisedIdentity = advertisedUrl.displayUrl() + require(upstreams.none { it.url.displayUrl() == advertisedIdentity }) { "[[mirror]] must not list this relay's own URL ($advertisedUrl)" } val mirror = @@ -252,7 +271,16 @@ fun main(args: Array) { maintenanceScope.launch { while (true) { delay(secs * 1000) - runCatching { store.optimize() } + try { + store.optimize() + } catch (e: CancellationException) { + throw e // shutdown cancelled us; don't swallow it + } catch (e: Exception) { + // A missed refresh only means slightly staler planner + // stats until the next tick — log and keep the loop. + // Errors (OOM, etc.) are NOT swallowed. + println("PRAGMA optimize failed: ${e.message}") + } } } } diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/MirrorFilterValidator.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/MirrorFilterValidator.kt new file mode 100644 index 0000000000..ab474cf596 --- /dev/null +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/MirrorFilterValidator.kt @@ -0,0 +1,78 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.geode.config + +import com.fasterxml.jackson.databind.ObjectMapper +import com.fasterxml.jackson.databind.node.ObjectNode + +/** + * Strict boot-time check for a `[[mirror]].filter` JSON string. The + * NIP-01 [com.vitorpamplona.quartz.nip01Core.relay.filters.Filter] + * deserializer is tolerant by design — it skips unknown keys and drops + * wrong-typed entries — which is right for untrusted wire input but + * wrong for operator config: a typo like `{"kindss":[4]}` would silently + * parse to an empty (match-everything) filter. Because this filter is + * the containment boundary for a `trusted = true` upstream (it caps what + * a skip-verify upstream may inject), a silent mis-parse would widen + * that trust to the whole firehose. So we reject the obvious mistakes at + * boot instead of degrading the boundary. + */ +object MirrorFilterValidator { + /** NIP-01 filter keys the deserializer recognizes; anything else is a typo. */ + private val KNOWN_KEYS = setOf("ids", "authors", "kinds", "since", "until", "limit", "search") + + /** Keys whose value must be a JSON array. */ + private val ARRAY_KEYS = setOf("ids", "authors", "kinds") + + private val mapper = ObjectMapper() + + /** + * Throws [IllegalArgumentException] when [json] is not a JSON object, + * carries an unrecognized top-level key, or gives a list-typed field + * (`ids`/`authors`/`kinds`, or a `#tag`/`&tag`) a non-array value. + */ + fun validate( + url: String, + json: String, + ) { + val node = + try { + mapper.readTree(json) + } catch (e: Exception) { + throw IllegalArgumentException("[[mirror]] filter for $url is not valid JSON: $json", e) + } + require(node is ObjectNode) { + "[[mirror]] filter for $url must be a JSON object (a NIP-01 filter), got: $json" + } + node.fieldNames().forEach { field -> + val isTagKey = field.length > 1 && (field[0] == '#' || field[0] == '&') + require(field in KNOWN_KEYS || isTagKey) { + "[[mirror]] filter for $url has unknown key \"$field\" — a NIP-01 filter uses " + + "ids/authors/kinds/since/until/limit/search or #tag/&tag" + } + if (field in ARRAY_KEYS || isTagKey) { + require(node.get(field).isArray) { + "[[mirror]] filter for $url: \"$field\" must be a JSON array" + } + } + } + } +} diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt index c1559f6726..1145a36a1c 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/config/StaticConfig.kt @@ -180,7 +180,10 @@ data class StaticConfig( /** * Keep an always-current in-memory `(created_at, id)` set so * full-corpus NEG-OPENs skip the table scan + seal (strfry - * parity). ~40 B per stored event of heap; on by default. + * parity). ~140 B per stored event of heap; on by default. Only + * built once the first full-corpus NEG-OPEN arrives, and only + * when the corpus fits `max_sync_events` (an over-cap corpus + * answers NEG-ERR from a capped scan instead). */ val live_index: Boolean = true, ) @@ -250,6 +253,25 @@ data class StaticConfig( val state_file: String? = null, ) + /** + * Boot-time sanity check for values the TOML types can't constrain. + * Throws [IllegalArgumentException] (fail-loud at startup) rather + * than letting a nonsensical knob degrade the running relay — a zero + * reader pool hangs every query, a non-positive optimize interval + * busy-loops the writer. Call once after parsing, before building + * the store. + */ + fun validate() { + database.readers?.let { + require(it >= 1) { "[database].readers must be >= 1 (got $it); a 0/negative pool can never answer a query" } + } + database.optimize_interval_seconds?.let { + require(it > 0) { + "[database].optimize_interval_seconds must be > 0 (got $it); a non-positive interval busy-loops PRAGMA optimize under the writer mutex" + } + } + } + companion object { private val mapper = tomlMapper { } diff --git a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt index 5974fb9294..d7698300d7 100644 --- a/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt +++ b/geode/src/test/kotlin/com/vitorpamplona/geode/config/StaticConfigTest.kt @@ -25,6 +25,7 @@ import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter import java.io.File import kotlin.test.Test import kotlin.test.assertEquals +import kotlin.test.assertFailsWith import kotlin.test.assertNotNull import kotlin.test.assertTrue @@ -81,6 +82,58 @@ class StaticConfigTest { assertTrue(StaticConfig.fromToml("").mirror.isEmpty()) } + @Test + fun validateRejectsNonPositiveReaders() { + assertFailsWith { + StaticConfig.fromToml("[database]\nreaders = 0").validate() + } + assertFailsWith { + StaticConfig.fromToml("[database]\nreaders = -1").validate() + } + // A sane pool passes. + StaticConfig.fromToml("[database]\nreaders = 1").validate() + // Unset passes (quartz default applies). + StaticConfig.fromToml("").validate() + } + + @Test + fun validateRejectsNonPositiveOptimizeInterval() { + assertFailsWith { + StaticConfig.fromToml("[database]\noptimize_interval_seconds = 0").validate() + } + assertFailsWith { + StaticConfig.fromToml("[database]\noptimize_interval_seconds = -5").validate() + } + StaticConfig.fromToml("[database]\noptimize_interval_seconds = 3600").validate() + } + + @Test + fun mirrorFilterValidatorRejectsTyposAndScalars() { + val url = "wss://up.example/" + + // Unknown key (a typo) — must fail, not silently widen scope. + assertFailsWith { + MirrorFilterValidator.validate(url, """{"kindss":[4]}""") + } + // List field given a scalar. + assertFailsWith { + MirrorFilterValidator.validate(url, """{"authors":"abc"}""") + } + // Not an object. + assertFailsWith { + MirrorFilterValidator.validate(url, """["kinds",1]""") + } + // Malformed JSON. + assertFailsWith { + MirrorFilterValidator.validate(url, """{"kinds":[1,}""") + } + + // Valid shapes pass: recognized scalar + array + tag keys. + MirrorFilterValidator.validate(url, """{"kinds":[0,1,3],"#t":["nostr"],"since":123,"limit":5,"search":"x"}""") + MirrorFilterValidator.validate(url, """{"&p":["abc"]}""") + MirrorFilterValidator.validate(url, "{}") + } + @Test fun mirrorFilterJsonParsesToANip01Filter() { // The exact parse Main.kt runs on [[mirror]].filter at boot.