From 57ffb3386d59c6693897c36c41f74a185da99353 Mon Sep 17 00:00:00 2001 From: Claude Date: Tue, 21 Jul 2026 15:07:12 +0000 Subject: [PATCH] feat(geode): enable the tag+kind+pubkey index, refresh measured docs TagAuthorIndexBenchmark at 1M events settles the flag: the DM-room shape (kinds + authors + #p, 65 client assembler call sites) drops 14.2 ms -> 0.66 ms (~21x, growing with corpus size) while batch-insert cost stays inside run noise (49.0 vs 47.4 us/event). Existing relay DBs build the index on next open via ensureOptionalIndexes. Also refreshes the docs the numbers made stale: IndexingStrategy KDoc now records the 200k and 1M measurements instead of a TODO, MergeQueryExecutor's tag-merge note points at the new relayBench reactions-watch scenario, FsQueryPlanner/FsDriverSelectionBenchmark reflect the landed cost-based pick (149 ms -> 4.0 ms at 30k events), and RELAY.md documents that strategy flag flips materialize indexes on the next open. Verified: quartz jvmTest store suites, geode test (126), desktopApp LocalRelayStore tests (5, incl. reopening a default-strategy DB with the new pubkey-alone flag), relayBench compiles. Co-Authored-By: Claude Fable 5 Claude-Session: https://claude.ai/code/session_01RzkoN3SJHCZWRAiadXXG4w --- .../geode/RelayIndexingStrategy.kt | 8 ++++ quartz/RELAY.md | 2 + .../store/sqlite/IndexingStrategy.kt | 13 +++-- .../store/sqlite/MergeQueryExecutor.kt | 11 +++-- .../nip01Core/store/fs/FsQueryPlanner.kt | 5 +- .../store/fs/FsDriverSelectionBenchmark.kt | 47 ++++++++++--------- 6 files changed, 51 insertions(+), 35 deletions(-) diff --git a/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt b/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt index 33fa9f4441..97dde44807 100644 --- a/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt +++ b/geode/src/main/kotlin/com/vitorpamplona/geode/RelayIndexingStrategy.kt @@ -57,6 +57,14 @@ fun relayIndexingStrategy( // index unconditionally; without it the filter walks the whole // time index. indexEventsByPubkeyAlone = true, + // The tag ∩ author ∩ kind shape (DM rooms, reports-by-follows, + // follows-scoped community feeds — 65 client assembler call sites) + // otherwise reads every row for the tag/kind before filtering the + // author. TagAuthorIndexBenchmark @ 1M events: 14.2 ms -> 0.66 ms + // (~21x, growing with corpus size) with insert cost inside run noise + // (49.0 vs 47.4 µs/event). Existing DBs build the index on next open + // via ensureOptionalIndexes. + indexTagsWithKindAndPubkey = true, indexFullTextSearch = fullTextSearch, // Tokenize off the commit path; NostrServer drives the catch-up // worker and search queries drain it first, so NIP-50 stays diff --git a/quartz/RELAY.md b/quartz/RELAY.md index 807b65caee..01bc5fc736 100644 --- a/quartz/RELAY.md +++ b/quartz/RELAY.md @@ -83,6 +83,8 @@ val store = EventStore( By default, all single-letter tags with values are indexed. Override `shouldIndex(kind, tag)` for custom behavior. More indexes = faster queries but larger database. +Flag flips are safe on existing databases: any flag-gated index the strategy wants but the on-disk schema lacks is created on the next open (idempotent `CREATE INDEX IF NOT EXISTS`, one-time build cost) — no schema version bump involved. Disabling a flag never drops an existing index. + `indexFullTextSearch` defaults to `true` and controls the NIP-50 full-text index (`event_fts`). Set it to `false` when search is served elsewhere (e.g. a Vespa backend, or a `SearchEventSource` as shown below): inserts skip the FTS tokenization cost, no `event_fts` table/trigger is created, and any filter carrying a non-empty `search` term returns no matches. ## Non-Storage Relays (search, redirector, computed) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt index 4f2e848c12..01362e03fb 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/IndexingStrategy.kt @@ -81,11 +81,14 @@ interface IndexingStrategy { * `(tag_hash, kind)` and reads every row for that tag/kind before * filtering the author. * - * TODO: re-evaluate the off-by-default choice (especially for geode) - * with `TagAuthorIndexBenchmark` (jvmTest prodbench). At 200k events: - * DM-room query 9.4 ms → 0.6 ms (~15×) with the flag on, for a batch - * insert cost of 41.5 → 47.3 µs/event (+14%) — measure at target - * corpus size before flipping, since the index competes for page cache. + * Measured by `TagAuthorIndexBenchmark` (jvmTest prodbench): the + * DM-room query drops 9.4 ms → 0.6 ms (~15×) at 200k events and + * 14.2 ms → 0.66 ms (~21×) at 1M — the gap grows with corpus size — + * while batch-insert cost stays inside run noise (49.0 vs 47.4 + * µs/event at 1M). geode enables it; the client default stays off + * because a client store's per-tag row counts are bounded by one + * user's data. Flipping it on an existing DB is safe: the index is + * built on next open by `EventIndexesModule.ensureOptionalIndexes`. * * Keep in mind that activating too many indexes increases the size of the * DB so much that the indexes themselves won't fit in memory, requiring diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/MergeQueryExecutor.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/MergeQueryExecutor.kt index 24b62c56d5..34ada1e64c 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/MergeQueryExecutor.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/MergeQueryExecutor.kt @@ -57,11 +57,12 @@ internal object MergeQueryExecutor { // reactions/replies watcher archetype) unions per-value streams that are // each sorted off `(tag_hash, kind, created_at)` and sorts the union. // `streamCount` currently rejects any filter with tags, so those queries - // never merge. Measured by `TagAuthorIndexBenchmark` at 200k events: - // `#e IN 300, limit 500` costs 12.8 ms cold / 6.0 ms since-bounded — - // tolerable client-side, but it scales with matching history like the - // follow-feed shape did; extend the merge to per-tag-value streams if - // relay-scale runs (relayBench) show it in the profile. + // never merge. Measured by `TagAuthorIndexBenchmark`: `#e IN 300, + // limit 500` costs 12.8 ms cold at 200k events and 14.2 ms at 1M + // (6.7 ms with indexTagsWithKindAndPubkey on) — tolerable, but it + // scales with matching history like the follow-feed shape did; extend + // the merge to per-tag-value streams if the relayBench + // `reactions-watch` scenario shows it in the profile vs strfry. const val COLS = "id, pubkey, created_at, kind, tags, content, sig" /** diff --git a/quartz/src/jvmMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsQueryPlanner.kt b/quartz/src/jvmMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsQueryPlanner.kt index 431ccafe85..f650bfd613 100644 --- a/quartz/src/jvmMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsQueryPlanner.kt +++ b/quartz/src/jvmMain/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsQueryPlanner.kt @@ -49,8 +49,9 @@ import kotlin.io.path.exists * with a million entries) is therefore never read past ~the smallest * candidate's size. Before the pick, the fixed tags → kinds → authors * order sent `authors + kinds + limit` — the most common CLI shape — - * through the kind tree: 149 ms vs 3.4 ms (~44×) at 30k events per - * `FsDriverSelectionBenchmark`. All FilterMatcher semantics (tag AND/OR, + * through the kind tree: 149 ms fixed-order vs 4.0 ms cost-based at 30k + * events per `FsDriverSelectionBenchmark` (floor: author-only at + * 1.4 ms). All FilterMatcher semantics (tag AND/OR, * since/until, id, author, kind cross-checks) are enforced in the * orchestrator, so any driver pick is correctness-safe. */ diff --git a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsDriverSelectionBenchmark.kt b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsDriverSelectionBenchmark.kt index 54058191ba..320e3fcb9f 100644 --- a/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsDriverSelectionBenchmark.kt +++ b/quartz/src/jvmTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/fs/FsDriverSelectionBenchmark.kt @@ -30,23 +30,24 @@ import kotlin.io.path.exists import kotlin.test.Test /** - * Quantifies [FsQueryPlanner]'s "first available driver wins" ordering - * (tags → kinds → authors) on the `authors + kinds + limit` shape — the - * most common CLI query (27 assembler call sites; every `amy feed`-style - * author timeline over non-replaceable kinds). + * Guards [FsQueryPlanner]'s cost-based driver pick on the + * `authors + kinds + limit` shape — the most common CLI query (27 + * assembler call sites; every `amy feed`-style author timeline over + * non-replaceable kinds). * - * `Filter(authors=[pk], kinds=[1], limit=n)` drives from `idx/kind/1/` - * (the biggest tree in any real store) and post-filters the author, even - * though `idx/author//` holds exactly that author's events. The - * benchmark times: + * Under the pre-pick fixed order (tags → kinds → authors), + * `Filter(authors=[pk], kinds=[1], limit=n)` drove from `idx/kind/1/` + * (the biggest tree in any real store) and post-filtered the author: + * 149 ms at 30k events. The lockstep pick drives from the author tree + * and runs at ~4 ms. The benchmark times: * - * - **kind-driver (current)**: the filter as the planner runs it today. - * - **author-driver (proposed)**: same result set, but driven from the - * author tree with the kind check as a post-filter — what a cost-based - * picker (compare candidate directory sizes) would choose. - * - * Also reports the author-only shape (`authors + limit`) as the floor: the - * planner already picks the author tree there, so its time is the target. + * - **planner (cost-based pick)**: the filter as the planner runs it — + * should sit near the floor, far below a kind-tree walk. + * - **author-driver emulation**: the author tree walked via an + * authors-only query with the kind check applied by the caller — the + * reference the pick is expected to match or beat. + * - **author-only floor**: `authors + limit` with no kind, the cheapest + * possible walk of the same tree. * * Size the seed with `-DfsBenchScale=N` (default 1 ≈ ~30k events; each * event is a file + ~3 hardlinks, so seeding dominates wall time). @@ -115,14 +116,14 @@ class FsDriverSelectionBenchmark { val insertMs = (System.nanoTime() - t0) / 1e6 println("─ FsDriverSelectionBenchmark: ${bg.size} events (scale=$SCALE), seed %.0f ms ─".format(insertMs)) - // Current planner: kinds present → kind tree drives, author - // is a post-filter over the whole kind-1 listing. + // The planner's own pick — expected to choose the author + // tree over the ~30k-entry kind-1 tree. val kindDriven = Filter(authors = listOf(target), kinds = listOf(1), limit = 50) - time(store, "kind-driver (current planner)") { store.query(kindDriven).size } + time(store, "planner (cost-based pick)") { store.query(kindDriven).size } - // Proposed: drive from the author tree, post-filter kind — - // same semantics, what a cost-based picker would run. - time(store, "author-driver (proposed)") { + // Reference: author tree walked explicitly, kind checked + // by the caller — the pick should match or beat this. + time(store, "author-driver emulation") { store .query(Filter(authors = listOf(target), limit = 250)) .asSequence() @@ -131,9 +132,9 @@ class FsDriverSelectionBenchmark { .count() } - // Floor: author-only shape, planner already optimal here. + // Floor: author-only shape, the cheapest walk of the tree. val authorOnly = Filter(authors = listOf(target), limit = 50) - time(store, "author-only (planner floor)") { store.query(authorOnly).size } + time(store, "author-only floor") { store.query(authorOnly).size } } finally { store.close() if (root.exists()) {