test(store): add tag∩author and FS driver-selection benchmarks

The 2026-07 client filter-assembler survey mapped 551 Filter
constructions to ~12 query archetypes. Two hot shapes had no benchmark
coverage in prodbench or relayBench, and the FS store had none at all:

- TagAuthorIndexBenchmark: the DM-room shape (kinds + authors + #p,
  65 assembler call sites) with indexTagsWithKindAndPubkey off vs on,
  including the insert-cost delta of the extra index; plus the
  reactions watcher (kinds=[7], #e IN 300, limit) cold and
  since-bounded, which has no tag-side k-way merge today.

- FsDriverSelectionBenchmark: FsQueryPlanner's fixed driver order
  (tags → kinds → authors) on authors+kinds+limit — the most common
  CLI shape — comparing the current kind-tree driver against an
  author-tree driver with kind post-filter.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01RzkoN3SJHCZWRAiadXXG4w
This commit is contained in:
Claude
2026-07-21 14:34:57 +00:00
parent 72e27a3530
commit 5988db010d
2 changed files with 342 additions and 0 deletions
@@ -0,0 +1,184 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip01Core.relay.prodbench
import com.vitorpamplona.quartz.nip01Core.core.Event
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
import com.vitorpamplona.quartz.nip01Core.store.sqlite.DefaultIndexingStrategy
import com.vitorpamplona.quartz.nip01Core.store.sqlite.EventStore
import com.vitorpamplona.quartz.utils.EventFactory
import kotlinx.coroutines.runBlocking
import kotlin.test.Test
/**
* Measures the two tag-path query shapes the client filter-assembler survey
* (2026-07) found hot but that no existing benchmark covers:
*
* 1. **tag ∩ author (DM-room shape)** — `kinds=[4] AND authors=[peer] AND
* #p=[me] LIMIT n`. 65 assembler call sites build this shape (every
* NIP-04 chat room, reports-by-follows, follows-scoped community feeds).
* [com.vitorpamplona.quartz.nip01Core.store.sqlite.IndexingStrategy.indexTagsWithKindAndPubkey]
* gates a covering `(tag_hash, kind, pubkey_hash, created_at)` index for
* it, but the flag is off everywhere (including geode). Without it the
* plan seeks `(tag_hash, kind)` and reads EVERY DM the user has ever
* received before filtering to the one peer. This compares query latency
* with the flag off vs on, and the batch-insert cost the extra index adds.
*
* 2. **large-IN tag watcher (reactions shape)** — `kinds=[7] AND
* #e=[hundreds of note ids] LIMIT n`. The per-value streams come sorted
* off `(tag_hash, kind, created_at)`, but their union does not, so SQLite
* collects every matching row and TEMP-B-TREE sorts to the limit — the
* tag-index analogue of the follow-feed regression
* [MergeQueryExecutor] fixed for author streams. Reported with and
* without a `since` bound to show what EOSE-warm steady state hides.
*
* Size the seed with `-DtagBenchScale=N` (default 1 ≈ ~200k events).
*/
class TagAuthorIndexBenchmark {
companion object {
val SCALE = System.getProperty("tagBenchScale")?.toInt() ?: 1
}
private val hex = "0123456789abcdef"
private fun mix(seed: Long): Long {
var z = seed + -0x61c8864680b583ebL
z = (z xor (z ushr 30)) * -0x40a7b892e31b1a47L
z = (z xor (z ushr 27)) * -0x6b2fb644ecceee15L
return z xor (z ushr 31)
}
private fun hex64(
salt: Long,
index: Int,
): String {
val out = CharArray(64)
for (w in 0 until 4) {
val v = mix(salt * 1_000_003 + index.toLong() * 4 + w)
for (b in 0 until 8) {
val byte = ((v ushr (b * 8)) and 0xFF).toInt()
out[(w * 8 + b) * 2] = hex[byte ushr 4]
out[(w * 8 + b) * 2 + 1] = hex[byte and 0xF]
}
}
return String(out)
}
private val sig = "0".repeat(128)
private var idSeq = 0
private fun ev(
pubkey: String,
createdAt: Long,
kind: Int,
tags: Array<Array<String>>,
): Event = EventFactory.create(hex64(7, idSeq++), pubkey, createdAt, kind, tags, "", sig)
private fun seedEvents(): List<Event> {
idSeq = 0
val base = 1_700_000_000L
val span = 3_000_000L // ~35 days
val me = hex64(9, 0)
val events = ArrayList<Event>(220_000 * SCALE)
// DM inbox: 200 peers, 300 DMs each → 60k kind-4 rows sharing the
// same (p:me) tag hash. The room query wants one peer's 300.
val peers = (0 until 200).map { hex64(2, it) }
for ((i, peer) in peers.withIndex()) {
repeat(300 * SCALE) {
val ts = base + (mix(i * 131L + it) and 0x7fffffff) % span
events.add(ev(peer, ts, 4, arrayOf(arrayOf("p", me))))
}
}
// Notification noise: 2000 authors mention me in kind-1 notes, so
// (p:me) spans multiple kinds like a real inbox does.
repeat(40_000 * SCALE) {
val author = hex64(3, it % 2_000)
val ts = base + (mix(it * 17L) and 0x7fffffff) % span
events.add(ev(author, ts, 1, arrayOf(arrayOf("p", me))))
}
// Reactions: 100k kind-7 events spread over 5000 target notes, for
// the large-IN watcher shape.
val noteIds = (0 until 5_000).map { hex64(5, it) }
repeat(100_000 * SCALE) {
val author = hex64(4, it % 3_000)
val ts = base + (mix(it * 29L) and 0x7fffffff) % span
events.add(ev(author, ts, 7, arrayOf(arrayOf("e", noteIds[it % noteIds.size]))))
}
return events
}
@Test
fun compareTagAuthorIndex() =
runBlocking {
val events = seedEvents()
val me = hex64(9, 0)
val peers = (0 until 200).map { hex64(2, it) }
val noteIds = (0 until 5_000).map { hex64(5, it) }
println("─ TagAuthorIndexBenchmark: ${events.size} events (scale=$SCALE) ─")
val strategies =
listOf(
"flag-off" to DefaultIndexingStrategy(indexFullTextSearch = false),
"flag-on " to DefaultIndexingStrategy(indexFullTextSearch = false, indexTagsWithKindAndPubkey = true),
)
for ((label, strategy) in strategies) {
val store = EventStore(dbName = null, indexStrategy = strategy)
val t0 = System.nanoTime()
events.chunked(10_000).forEach { store.batchInsert(it) }
val insertMs = (System.nanoTime() - t0) / 1e6
println("$label ═ insert: %.0f ms (%.1f µs/event)".format(insertMs, insertMs * 1000 / events.size))
// 1. DM room: one peer's DMs out of the whole (p:me) inbox.
val room = Filter(kinds = listOf(4), authors = listOf(peers[42]), tags = mapOf("p" to listOf(me)), limit = 100)
time(store, "dm-room (#p ∩ author ∩ kind, limit 100)", room)
// 2. Reactions watcher: 300 note ids, cold (no since).
val watcher = Filter(kinds = listOf(7), tags = mapOf("e" to noteIds.take(300)), limit = 500)
time(store, "reactions (#e IN 300, limit 500, cold)", watcher)
// 3. Same watcher, EOSE-warm (since bounds the window).
val warm = watcher.copy(since = 1_700_000_000L + 2_900_000L)
time(store, "reactions (#e IN 300, limit 500, since)", warm)
store.close()
}
}
private suspend fun time(
store: EventStore,
label: String,
filter: Filter,
) {
repeat(3) { store.query<Event>(filter) }
val runs = 10
var rows = 0
val start = System.nanoTime()
repeat(runs) { rows = store.query<Event>(filter).size }
val ms = (System.nanoTime() - start) / 1e6 / runs
println(" %-42s %8.2f ms (%d rows)".format(label, ms, rows))
}
}
@@ -0,0 +1,158 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip01Core.store.fs
import com.vitorpamplona.quartz.nip01Core.core.Event
import com.vitorpamplona.quartz.nip01Core.relay.filters.Filter
import com.vitorpamplona.quartz.utils.EventFactory
import kotlinx.coroutines.runBlocking
import java.nio.file.Files
import java.nio.file.Path
import kotlin.io.path.exists
import kotlin.test.Test
/**
* Quantifies [FsQueryPlanner]'s "first available driver wins" ordering
* (tags → kinds → authors) on the `authors + kinds + limit` shape — the
* most common CLI query (27 assembler call sites; every `amy feed`-style
* author timeline over non-replaceable kinds).
*
* `Filter(authors=[pk], kinds=[1], limit=n)` drives from `idx/kind/1/`
* (the biggest tree in any real store) and post-filters the author, even
* though `idx/author/<pk>/` holds exactly that author's events. The
* benchmark times:
*
* - **kind-driver (current)**: the filter as the planner runs it today.
* - **author-driver (proposed)**: same result set, but driven from the
* author tree with the kind check as a post-filter — what a cost-based
* picker (compare candidate directory sizes) would choose.
*
* Also reports the author-only shape (`authors + limit`) as the floor: the
* planner already picks the author tree there, so its time is the target.
*
* Size the seed with `-DfsBenchScale=N` (default 1 ≈ ~30k events; each
* event is a file + ~3 hardlinks, so seeding dominates wall time).
*/
class FsDriverSelectionBenchmark {
companion object {
val SCALE = System.getProperty("fsBenchScale")?.toInt() ?: 1
}
private val hex = "0123456789abcdef"
private fun mix(seed: Long): Long {
var z = seed + -0x61c8864680b583ebL
z = (z xor (z ushr 30)) * -0x40a7b892e31b1a47L
z = (z xor (z ushr 27)) * -0x6b2fb644ecceee15L
return z xor (z ushr 31)
}
private fun hex64(
salt: Long,
index: Int,
): String {
val out = CharArray(64)
for (w in 0 until 4) {
val v = mix(salt * 1_000_003 + index.toLong() * 4 + w)
for (b in 0 until 8) {
val byte = ((v ushr (b * 8)) and 0xFF).toInt()
out[(w * 8 + b) * 2] = hex[byte ushr 4]
out[(w * 8 + b) * 2 + 1] = hex[byte and 0xF]
}
}
return String(out)
}
private val sig = "0".repeat(128)
private var idSeq = 0
private fun ev(
pubkey: String,
createdAt: Long,
kind: Int,
): Event = EventFactory.create(hex64(7, idSeq++), pubkey, createdAt, kind, emptyArray(), "", sig)
@Test
fun compareDrivers() =
runBlocking {
val root: Path = Files.createTempDirectory("fs-driver-bench-")
val store = FsEventStore(root)
try {
val base = 1_700_000_000L
val span = 3_000_000L
val target = hex64(9, 0)
// Background: 300 authors × 100 kind-1 notes.
val bg = ArrayList<Event>(30_000 * SCALE + 300)
repeat(30_000 * SCALE) {
val author = hex64(1, it % 300)
bg.add(ev(author, base + (mix(it * 31L) and 0x7fffffff) % span, 1))
}
// Target author: 200 kind-1 notes + 50 kind-7 reactions.
repeat(200) { bg.add(ev(target, base + (mix(it * 131L) and 0x7fffffff) % span, 1)) }
repeat(50) { bg.add(ev(target, base + (mix(it * 61L) and 0x7fffffff) % span, 7)) }
val t0 = System.nanoTime()
store.transaction { bg.forEach { insert(it) } }
val insertMs = (System.nanoTime() - t0) / 1e6
println("─ FsDriverSelectionBenchmark: ${bg.size} events (scale=$SCALE), seed %.0f ms ─".format(insertMs))
// Current planner: kinds present → kind tree drives, author
// is a post-filter over the whole kind-1 listing.
val kindDriven = Filter(authors = listOf(target), kinds = listOf(1), limit = 50)
time(store, "kind-driver (current planner)") { store.query<Event>(kindDriven).size }
// Proposed: drive from the author tree, post-filter kind —
// same semantics, what a cost-based picker would run.
time(store, "author-driver (proposed)") {
store
.query<Event>(Filter(authors = listOf(target), limit = 250))
.asSequence()
.filter { it.kind == 1 }
.take(50)
.count()
}
// Floor: author-only shape, planner already optimal here.
val authorOnly = Filter(authors = listOf(target), limit = 50)
time(store, "author-only (planner floor)") { store.query<Event>(authorOnly).size }
} finally {
store.close()
if (root.exists()) {
Files.walk(root).use { it.sorted(Comparator.reverseOrder()).forEach { p -> Files.deleteIfExists(p) } }
}
}
}
private inline fun time(
store: FsEventStore,
label: String,
run: () -> Int,
) {
repeat(3) { run() }
val runs = 10
var rows = 0
val start = System.nanoTime()
repeat(runs) { rows = run() }
val ms = (System.nanoTime() - start) / 1e6 / runs
println(" %-32s %8.2f ms (%d rows)".format(label, ms, rows))
}
}