From a88ad5b39d792ad315973e02b90e5c411503f5d1 Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Thu, 13 Aug 2026 14:59:21 -0400 Subject: [PATCH 1/5] perf: scan note content by literal instead of driving the regex engine MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The NIP-19, hashtag and legacy-#[n] scans all ran `Regex.findAll` over the whole of a note's content. `findAll` restarts the regex engine at every position in the string, so cost scaled with content length regardless of whether the content had anything to find — and most notes have nothing to find. Each of these patterns has a literal that must be present for any match: - `nip19regex` can only match at one of eight entity prefixes, all starting `n` - `hashtagSearch` requires `(?:\s|\A)` immediately before a `#` - `tagSearch` requires the same before `#[` - `RichTextParser.tagIndex` requires the literal `#[` So the scans now jump between candidate positions with `indexOf`/char compares and apply the regex ANCHORED there via `matchAt`. For the NIP-19 scan, `(nostr:)?@?` are optional, so anchoring at the entity's own `n` matches the same entities and captures the same groups the callers read. Measured on the production content distribution — median 529 B with a tail to 767 KB, taken from an on-device heap dump where 2,541 of 4,573 live Matchers were running the NIP-19 regex: before after (no match / with matches) Nip19Parser.parseAll 19 MB/s 922 / 61-654 MB/s findHashtags 68 MB/s ~19,000 / ~1,240 MB/s IndexedTags 63 MB/s ~19,900 / ~12,100 MB/s RichTextParser '#[' - 4.93x on the #[n] step Worst case improves most: a 767 KB long-form note went from 40.6 ms to 0.83 ms per NIP-19 scan. Two further notes on the NIP-19 scan. Every prefix starts with `n` and English prose is ~7% `n`, so testing all eight prefixes at each `n` dominated the scan; dispatching on the second character first cuts that to at most two and is worth 2.1x on its own. And `RichTextParser` gates on `contains("#[")`, not `startsWith`, because `find()` also matches `#[n]` mid-word. Behaviour is unchanged. `RegexContentBenchmark` (new, in commons) pins each rewritten scan against a verbatim copy of the implementation it replaces, plus ~6,000 fuzz cases, the boundary cases where `n`/`#` is the last character, and hardcoded segment expectations for the RichTextParser '#' path. Co-Authored-By: Claude Opus 5 (1M context) --- .../commons/richtext/RichTextParser.kt | 6 +- .../prodbench/RegexContentBenchmark.kt | 651 ++++++++++++++++++ .../nip10Notes/content/ContentHashTags.kt | 38 +- .../quartz/nip10Notes/content/IndexedTags.kt | 43 +- .../quartz/nip19Bech32/Nip19Parser.kt | 68 +- 5 files changed, 785 insertions(+), 21 deletions(-) create mode 100644 commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt diff --git a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt index 8d5fd30da0..c4bdde30a6 100644 --- a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt +++ b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt @@ -416,8 +416,12 @@ class RichTextParser { tags: ImmutableListOfLists, ): Segment { // First #[n] + // [tagIndex] requires the literal "#[", so a word without it can never match. + // Every plain "#hashtag" reaches this function, and the regex scan is far more + // expensive than the substring check that rules it out — so gate on the literal. + // Uses contains(), not startsWith(), because find() also matches "#[n]" mid-word. try { - val matcher = tagIndex.find(word) + val matcher = if (word.contains("#[")) tagIndex.find(word) else null if (matcher != null) { val index = matcher.groups[1]?.value?.toInt() val suffix = matcher.groups[2]?.value diff --git a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt new file mode 100644 index 0000000000..756e17f82c --- /dev/null +++ b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt @@ -0,0 +1,651 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.commons.prodbench + +import com.vitorpamplona.amethyst.commons.model.toImmutableListOfLists +import com.vitorpamplona.amethyst.commons.richtext.RichTextParser +import com.vitorpamplona.amethyst.commons.richtext.RichTextParser.Companion.tagIndex +import com.vitorpamplona.quartz.nip10Notes.content.findHashtags +import com.vitorpamplona.quartz.nip10Notes.content.findIndexTagsWithEventsOrAddresses +import com.vitorpamplona.quartz.nip10Notes.content.findIndexTagsWithPeople +import com.vitorpamplona.quartz.nip10Notes.content.findNostrUris +import com.vitorpamplona.quartz.nip10Notes.content.hashtagSearch +import com.vitorpamplona.quartz.nip10Notes.content.tagSearch +import com.vitorpamplona.quartz.nip19Bech32.Nip19Parser +import kotlin.test.Test + +/** + * Measures the regex scans Amethyst runs over note **content**. + * + * Why: an on-device heap dump (SM-T220, Dr. Edo's account) found 2,541 of 4,573 + * live `java.util.regex.Matcher` instances running [Nip19Parser.nip19regex], + * reached from `BaseNoteEvent.citedNIP19()` during ingest. Kotlin's `findAll` + * allocates a **new Matcher per match**, each retaining the whole input + * (`jvmMain/kotlin/text/regex/Regex.kt`: `matcher.pattern().matcher(input)`), + * so a long article with N mentions builds N matchers over the full text. + * + * Corpus sizes come from that same dump's measured distribution of the strings + * these matchers held: median 529 B, p90/max 767 KB (long-form articles — the + * v1.13.0 release notes at 68 KB, an essay at 50 KB). + * + * Deterministic and offline. Prints ns/op and MB/s; no assertions on wall time + * (CI machines vary) beyond a sanity check that the scans return results. + */ +class RegexContentBenchmark { + companion object { + private const val NPUB = "npub180cvv07tjdrrgpa0j7j7tmnyl2yr6yr7l8j4s3evf6u64th6gkwsyjh6w6" + private const val NEVENT = "nevent1qqstna2yrezu5wghjvswqqculvvwxsrcvu7uc0f78gan4xqhvz49d9spr3mhxue69uhkummnw3ez6un9d3shjtn4de6x2argwghx6egpr4mhxue69uhkummnw3ez6ur4vgh8wetvd3hhyer9wghxuet5nxnepm" + + /** A note with exactly [mentions] nostr: URIs, padded with prose to ~[targetBytes]. */ + fun note( + targetBytes: Int, + mentions: Int, + hashtags: Boolean = true, + ): String { + val filler = + if (hashtags) { + "A few months ago a nostrich was switching from iOS to Android and asked for " + + "suggestions for #Nostr apps to try out. Here is what came back, with notes. " + } else { + "A few months ago a nostrich was switching from iOS to Android and asked for " + + "suggestions for great apps to try out. Here is what came back, with notes. " + } + val sb = StringBuilder(targetBytes + 4096) + var placed = 0 + val stride = if (mentions > 0) targetBytes / (mentions + 1) else Int.MAX_VALUE + while (sb.length < targetBytes) { + sb.append(filler) + if (placed < mentions && sb.length >= (placed + 1).toLong() * stride) { + sb.append("nostr:").append(if (placed % 2 == 0) NPUB else NEVENT).append(' ') + placed++ + } + } + while (placed < mentions) { + sb.append("nostr:").append(if (placed % 2 == 0) NPUB else NEVENT).append(' ') + placed++ + } + return sb.toString() + } + + private val PREFIXES = + arrayOf("npub1", "nsec1", "note1", "nevent1", "naddr1", "nprofile1", "nrelay1", "nembed1") + + /** + * One O(n) pass over the chars: stop at every 'n'/'N' and test the 8 entity + * prefixes with regionMatches(ignoreCase). Cheap char compares instead of + * driving the regex NFA from every position in the string. + */ + fun hasNip19Candidate(content: String): Boolean { + var i = 0 + val n = content.length + while (i < n) { + val c = content[i] + if (c == 'n' || c == 'N') { + for (p in PREFIXES) { + if (content.regionMatches(i, p, 0, p.length, ignoreCase = true)) return true + } + } + i++ + } + return false + } + + /** + * Variant 2: never let the NFA scan. Walk to each candidate position with + * cheap char compares, then apply the regex ANCHORED there via matchAt. + * `(nostr:)?@?` are optional, so anchoring at the entity's 'n' still matches + * and still captures the type/key/trailing groups parseAll uses. + */ + fun anchoredNip19(content: String): Int { + var i = 0 + var hits = 0 + val n = content.length + while (i < n) { + val c = content[i] + if (c == 'n' || c == 'N') { + var candidate = false + for (p in PREFIXES) { + if (content.regionMatches(i, p, 0, p.length, ignoreCase = true)) { + candidate = true + break + } + } + if (candidate) { + val m = Nip19Parser.nip19regex.matchAt(content, i) + if (m != null) { + hits++ + i = m.range.last + 1 + continue + } + } + } + i++ + } + return hits + } + + /** + * Candidate: the regex requires `(?:\s|\A)` immediately before the `#`, so + * every match starts at a whitespace (or at position 0). Jump between '#' + * occurrences with indexOf and anchor the regex there. + */ + fun fastFindHashtags(content: String): List { + if (content.isBlank()) return emptyList() + val out = mutableSetOf() + var h = content.indexOf('#') + while (h >= 0) { + if (h == 0 || content[h - 1].isWhitespace()) { + val m = hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1) + if (m != null) { + m.groups[1]?.value?.let { if (it.isNotBlank()) out.add(it) } + h = content.indexOf('#', m.range.last + 1) + continue + } + } + h = content.indexOf('#', h + 1) + } + return out.toList() + } + + /** + * REFERENCE: the pre-optimization `findHashtags`, verbatim. Production is + * compared against this so the guard can never become a tautology. + */ + fun referenceFindHashtags(content: String): List { + if (content.isBlank()) return emptyList() + val output = mutableSetOf() + hashtagSearch.findAll(content).forEach { + try { + val tag = it.groups[1]?.value + if (tag != null && tag.isNotBlank()) output.add(tag) + } catch (e: Exception) { + } + } + return output.toList() + } + + /** REFERENCE: pre-optimization nip19 scan (findAll), resolved via the public uriToRoute. */ + fun referenceNip19(content: String): List = + Nip19Parser.nip19regex + .findAll(content) + .mapNotNull { m -> + val type = m.groups[3]?.value ?: m.groups[5]?.value + val key = m.groups[4]?.value ?: m.groups[6]?.value + if (type != null && key != null) { + Nip19Parser.uriToRoute(type + key)?.entity?.let { type + key } + } else { + null + } + }.toList() + + /** A tag array with [n] p/e/a entries, for the #[index] scans. */ + fun tagArray(n: Int): Array> = + Array(n) { i -> + when (i % 3) { + 0 -> arrayOf("p", "%064x".format(i)) + 1 -> arrayOf("e", "%064x".format(i + 1000)) + else -> arrayOf("a", "30023:%064x:slug$i".format(i)) + } + } + + /** A note with [refs] legacy #[n] references, padded to ~[targetBytes]. */ + fun indexNote( + targetBytes: Int, + refs: Int, + tagCount: Int, + ): String { + val filler = "Legacy index refs used to link people and events inline in the content. " + val sb = StringBuilder(targetBytes + 4096) + var placed = 0 + val stride = if (refs > 0) targetBytes / (refs + 1) else Int.MAX_VALUE + while (sb.length < targetBytes) { + sb.append(filler) + if (placed < refs && sb.length >= (placed + 1).toLong() * stride) { + sb.append("#[").append(placed % tagCount).append("] ") + placed++ + } + } + while (placed < refs) { + sb.append("#[").append(placed % tagCount).append("] ") + placed++ + } + return sb.toString() + } + + /** REFERENCE: pre-optimization findIndexTagsWithPeople, verbatim. */ + fun referenceIndexPeople( + content: String, + tags: Array>, + ): List { + val output = mutableSetOf() + tagSearch.findAll(content).forEach { index -> + try { + val tag = index.groups[1]?.value?.let { tags[it.toInt()] } + if (tag != null && tag.size > 1 && tag[0] == "p") output.add(tag[1]) + } catch (e: Exception) { + } + } + return output.toList() + } + + /** REFERENCE: pre-optimization findIndexTagsWithEventsOrAddresses, verbatim. */ + fun referenceIndexEvents( + content: String, + tags: Array>, + ): Set { + val output = mutableSetOf() + tagSearch.findAll(content).forEach { index -> + try { + val tag = index.groups[1]?.value?.let { tags[it.toInt()] } + if (tag != null && tag.size > 1 && tag[0] == "e") output.add(tag[1]) + if (tag != null && tag.size > 1 && tag[0] == "a") output.add(tag[1]) + } catch (e: Exception) { + } + } + return output + } + + fun bench( + label: String, + input: String, + reps: Int, + op: (String) -> Int, + ) { + repeat(maxOf(reps / 4, 2)) { op(input) } // warmup + val t0 = System.nanoTime() + var sink = 0 + repeat(reps) { sink += op(input) } + val ns = (System.nanoTime() - t0) / reps + val mbps = input.length.toDouble() / ns * 1000.0 // bytes/ns -> MB/s + println( + String.format( + "%-34s %9d B %9d ns/op %8.1f MB/s (hits=%d)", + label, + input.length, + ns, + mbps, + sink / reps, + ), + ) + } + } + + @Test + fun contentScans() { + // (bytes, mentions, reps) — mirrors the measured distribution + val corpus = + listOf( + Triple(120, 1, 20_000), // short note + Triple(529, 2, 20_000), // MEDIAN of what matchers held + Triple(4_000, 5, 5_000), // long note + Triple(68_000, 40, 200), // v1.13.0 release notes + Triple(767_000, 120, 20), // observed MAX + ) + + // Guard the corpus: a benchmark that silently matches nothing measures nothing. + corpus.forEach { (n, m, _) -> + val found = Nip19Parser.parseAll(note(n, m)).size + check(found >= m) { "corpus broken: ${n}B/$m mentions parsed only $found entities" } + } + + println("\n=== nip19 findNostrUris — NO matches (the common case) ===") + corpus.forEach { (n, _, r) -> + bench("nip19 0 mentions", note(n, 0), r) { findNostrUris(it).size } + } + + println("\n=== nip19 findNostrUris — WITH matches (Matcher per match) ===") + corpus.forEach { (n, m, r) -> + bench("nip19 m=$m", note(n, m), r) { findNostrUris(it).size } + } + + println("\n=== findHashtags (ContentHashTags.hashtagSearch) ===") + corpus.forEach { (n, m, r) -> + bench("hashtags m=$m", note(n, m), r) { findHashtags(it).size } + } + + println("\n=== OPTIMIZED: literal pre-scan, then regex only if a candidate exists ===") + corpus.forEach { (n, _, r) -> + bench("opt nip19 0 mentions", note(n, 0), r) { if (hasNip19Candidate(it)) findNostrUris(it).size else 0 } + } + corpus.forEach { (n, m, r) -> + bench("opt nip19 m=$m", note(n, m), r) { if (hasNip19Candidate(it)) findNostrUris(it).size else 0 } + } + + println("\n=== findHashtags — content with NO '#' ===") + corpus.forEach { (n, _, r) -> + bench("hashtags none", note(n, 0, hashtags = false), r) { findHashtags(it).size } + } + println("\n=== findHashtags OPTIMIZED (indexOf('#') + anchored) ===") + corpus.forEach { (n, _, r) -> + bench("opt hashtags none", note(n, 0, hashtags = false), r) { fastFindHashtags(it).size } + } + corpus.forEach { (n, m, r) -> + bench("opt hashtags dense", note(n, m), r) { fastFindHashtags(it).size } + } + + println("\n=== IndexedTags findIndexTagsWithPeople ===") + val tags = tagArray(30) + corpus.forEach { (n, _, r) -> + bench("idxTags none", indexNote(n, 0, 30), r) { findIndexTagsWithPeople(it, tags).size } + } + corpus.forEach { (n, m, r) -> + bench("idxTags refs=$m", indexNote(n, m, 30), r) { findIndexTagsWithPeople(it, tags).size } + } + + println("\n=== RENDER PATH: RichTextParser.parseText (per note, per render) ===") + val rtTags = tagArray(30).toImmutableListOfLists() + val rtCorpus = + listOf( + Triple(120, 1, 3_000), + Triple(529, 2, 3_000), + Triple(4_000, 5, 500), + Triple(68_000, 40, 20), + ) + rtCorpus.forEach { (n, m, r) -> + bench("parseText hashtags m=$m", note(n, m), r) { + RichTextParser().parseText(it, rtTags, null).paragraphs.size + } + } + rtCorpus.forEach { (n, _, r) -> + bench("parseText no '#'", note(n, 0, hashtags = false), r) { + RichTextParser().parseText(it, rtTags, null).paragraphs.size + } + } + rtCorpus.forEach { (n, m, r) -> + bench("parseText #[n] refs", indexNote(n, m, 30), r) { + RichTextParser().parseText(it, rtTags, null).paragraphs.size + } + } + + println("\n=== OPTIMIZED v2: anchored matchAt at candidate positions (no NFA scan) ===") + corpus.forEach { (n, _, r) -> + bench("v2 nip19 0 mentions", note(n, 0), r) { anchoredNip19(it) } + } + corpus.forEach { (n, m, r) -> + bench("v2 nip19 m=$m", note(n, m), r) { anchoredNip19(it) } + } + } + + /** The fast hashtag scan must return exactly what the production scan returns. */ + @Test + fun fastHashtagsMatchesProduction() { + val cases = + listOf( + "#one #two #three", + "no hashtags here", + "#start of line", + "mid #tag then more", + "email a@b.com and #tag", + "punct #tag, #tag2. #tag3!", + "nospace#nottag should not match", + "\n#afterNewline", + "", + " ", + note(4_000, 0), + note(20_000, 3), + note(4_000, 0, hashtags = false), + ) + cases.forEach { c -> + val ref = referenceFindHashtags(c).sorted() + val prod = findHashtags(c).sorted() + check(ref == prod) { "mismatch on ${c.take(40).replace("\n", "\\n")}: reference=$ref production=$prod" } + } + + // fuzz: random text peppered with '#' in awkward positions + val rnd = kotlin.random.Random(20260813) + val alphabet = " \n\t#abcXYZ.,!@\u00a0-_0123" + repeat(3_000) { + val c = (1..rnd.nextInt(0, 80)).map { alphabet[rnd.nextInt(alphabet.length)] }.joinToString("") + val ref = referenceFindHashtags(c).sorted() + val prod = findHashtags(c).sorted() + check(ref == prod) { "fuzz mismatch on ${c.replace("\n", "\\n")}: reference=$ref production=$prod" } + } + } + + /** + * Isolated A/B for the `#[` gate, interleaved with medians. + * + * parseText is allocation-heavy and its whole-parse timings swing ±50% run to + * run (the unchanged no-'#' control arm moved that much), which is far larger + * than this effect — so measure the gated call on its own instead. + */ + @Test + fun hashGateIsolatedAb() { + // realistic mix: mostly plain hashtags, a few legacy #[n] refs + val words = + buildList { + repeat(90) { add("#hashtag$it") } + repeat(10) { add("#[$it]") } + } + + fun ungated(): Int { + var n = 0 + words.forEach { if (tagIndex.find(it) != null) n++ } + return n + } + + fun gated(): Int { + var n = 0 + words.forEach { if (it.contains("#[") && tagIndex.find(it) != null) n++ } + return n + } + check(ungated() == gated()) { "gate changed the result" } + repeat(2_000) { + ungated() + gated() + } // warmup both + val a = mutableListOf() + val b = mutableListOf() + repeat(21) { + var t = System.nanoTime() + repeat(2_000) { ungated() } + a.add(System.nanoTime() - t) + t = System.nanoTime() + repeat(2_000) { gated() } + b.add(System.nanoTime() - t) + } + a.sort() + b.sort() + val am = a[a.size / 2] / 2_000 + val bm = b[b.size / 2] / 2_000 + println( + "\nHASH GATE (100 words: 90 plain hashtags + 10 #[n]) median of 21:" + + "\n ungated tagIndex.find : $am ns/word-set" + + "\n gated with contains : $bm ns/word-set" + + "\n speedup : ${String.format("%.2f", am.toDouble() / bm)}x", + ) + } + + /** + * Segment-level guard for the RichTextParser '#' path. + * + * Expectations are HARDCODED from the behaviour before the `#[` gate was added, + * so this pins the parser against the code it replaced rather than against itself. + */ + @Test + fun richTextHashSegmentsUnchanged() { + val tags = tagArray(30).toImmutableListOfLists() + val expected = + listOf( + "#Nostr" to "HashTagSegment|#Nostr", + "#[0]" to "HashIndexUserSegment|#[0]", + "#[1]suffix" to "HashIndexEventSegment|#[1]suffix", + "#[999]" to "RegularTextSegment|#[999]", + "#[abc]" to "RegularTextSegment|#[abc]", + "#[]" to "RegularTextSegment|#[]", + "#" to "RegularTextSegment|#", + "##tag" to "HashTagSegment|##tag", + "#tag," to "HashTagSegment|#tag,", + "#tag." to "HashTagSegment|#tag.", + "#a#b" to "HashTagSegment|#a#b", + "#[2]#tail" to "HashIndexEventSegment|#[2]#tail", + "#\u00e9t\u00e9" to "HashTagSegment|#\u00e9t\u00e9", + "#123" to "HashTagSegment|#123", + "#[0]!" to "HashIndexUserSegment|#[0]!", + "plain" to "RegularTextSegment|plain", + "#tag)" to "HashTagSegment|#tag)", + "#-dash" to "HashTagSegment|#-dash", + "#_under" to "HashTagSegment|#_under", + // '#[' appearing mid-word must still reach the index parser + "#a#[0]" to "HashIndexUserSegment|#a#[0]", + ) + expected.forEach { (word, want) -> + val got = + RichTextParser() + .parseText(word, tags, null) + .paragraphs + .flatMap { p -> p.words.map { it::class.simpleName + "|" + it.segmentText } } + .joinToString(";") + check(got == want) { "segment mismatch for '$word': want=$want got=$got" } + } + } + + /** Both IndexedTags scans must equal the findAll-based references. */ + @Test + fun indexedTagsMatchReference() { + val tags = tagArray(30) + val cases = + listOf( + "#[0] #[1] #[2]", + "no refs here", + "#[0] at start", + "mid #[5] ref", + "nospace#[3] should not match", + "\n#[7] after newline", + "#[999] out of range", + "#[abc] not numeric", + "#[] empty", + "", + " ", + indexNote(2_000, 0, 30), + indexNote(4_000, 6, 30), + indexNote(20_000, 20, 30), + ) + cases.forEach { c -> + check(referenceIndexPeople(c, tags).sorted() == findIndexTagsWithPeople(c, tags).sorted()) { + "people mismatch on ${c.take(40).replace("\n", "\\n")}" + } + check(referenceIndexEvents(c, tags).sorted() == findIndexTagsWithEventsOrAddresses(c, tags).sorted()) { + "events mismatch on ${c.take(40).replace("\n", "\\n")}" + } + } + val rnd = kotlin.random.Random(20260813) + val alphabet = " \n#[]0123456789abc\u00a0" + repeat(3_000) { + val c = (1..rnd.nextInt(0, 60)).map { alphabet[rnd.nextInt(alphabet.length)] }.joinToString("") + check(referenceIndexPeople(c, tags).sorted() == findIndexTagsWithPeople(c, tags).sorted()) { + "fuzz people mismatch on ${c.replace("\n", "\\n")}" + } + check(referenceIndexEvents(c, tags).sorted() == findIndexTagsWithEventsOrAddresses(c, tags).sorted()) { + "fuzz events mismatch on ${c.replace("\n", "\\n")}" + } + } + } + + /** Production parseAll must equal the findAll-based reference scan. */ + @Test + fun nip19MatchesReferenceScan() { + val cases = + listOf( + "nostr:$NPUB", + "@$NPUB", + NPUB, + "text nostr:$NEVENT tail", + "two $NPUB and nostr:$NEVENT here", + "none at all", + "npub1tooshort", + "", + note(200, 1), + note(4_000, 5), + note(20_000, 12), + note(4_000, 0), + // every prefix standalone + "npub1", + "nsec1", + "note1", + "nevent1", + "naddr1", + "nprofile1", + "nrelay1", + "nembed1", + // 'n' as the LAST char: the second-char dispatch must not read past the end + "n", + "N", + "ends with n", + "trailing N", + "nn", + "np", + "ne", + "na", + "nr", + "ns", + "no", + // case + adjacency + "NOSTR:" + NPUB.uppercase(), + "x" + NPUB, + NPUB + NEVENT, + NPUB + " " + NEVENT, + ) + cases.forEach { c -> + val ref = referenceNip19(c) + val prod = Nip19Parser.parseAll(c).size + check(ref.size == prod) { "nip19 mismatch on ${c.take(50)}: reference=${ref.size} production=$prod" } + } + } + + /** v2 must find exactly the same number of entities as the production scan. */ + @Test + fun anchoredMatchesProductionCount() { + listOf(0, 1, 2, 5, 40).forEach { m -> + listOf(200, 4_000, 68_000).forEach { size -> + val c = note(size, m) + val real = Nip19Parser.nip19regex.findAll(c).count() + val fast = anchoredNip19(c) + check(real == fast) { "size=$size m=$m: production found $real, anchored found $fast" } + } + } + } + + /** + * Correctness gate for the pre-scan: it must never hide a real match. + * Anything the regex finds must also be reported as a candidate. + */ + @Test + fun preScanNeverMissesAMatch() { + val cases = + listOf( + "nostr:$NPUB", + "@$NPUB", + NPUB, + "text before nostr:$NEVENT and after", + "UPPER ${NPUB.uppercase()} case", + "no entities here at all, just prose about nostr and npubs", + "npub1tooshort", + "", + ) + cases.forEach { c -> + val real = Nip19Parser.parseAll(c).isNotEmpty() + if (real) check(hasNip19Candidate(c)) { "pre-scan missed a real match in: ${c.take(40)}" } + } + // and it must actually filter the common case + check(!hasNip19Candidate(note(4_000, 0))) { "pre-scan failed to reject prose with no entities" } + } +} diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt index f14b2ebe20..51a70d5331 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt @@ -22,21 +22,45 @@ package com.vitorpamplona.quartz.nip10Notes.content val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)") +/** + * Collects the hashtags in [content]. + * + * [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match + * starts either at position 0 or at a whitespace. That lets the scan jump between + * `#` occurrences with `indexOf` — an intrinsified char search — and apply the + * regex **anchored** at each, instead of letting `findAll` drive the regex engine + * from every position in the string. + * + * Measured on the production content distribution (median 529 B, tail to 767 KB): + * ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the + * content has no `#` at all (up to 300x). Equivalence with the previous `findAll` + * implementation is guarded by `RegexContentBenchmark` in `commons`. + */ fun findHashtags( content: String, output: MutableSet = mutableSetOf(), ): List { if (content.isBlank()) return emptyList() - val matcher = hashtagSearch.findAll(content) - matcher.forEach { - try { - val tag = it.groups[1]?.value - if (tag != null && tag.isNotBlank()) { - output.add(tag) + var h = content.indexOf('#') + while (h >= 0) { + if (h == 0 || content[h - 1].isWhitespace()) { + val match = + try { + hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1) + } catch (e: Exception) { + null + } + if (match != null) { + val tag = match.groups[1]?.value + if (tag != null && tag.isNotBlank()) { + output.add(tag) + } + h = content.indexOf('#', match.range.last + 1) + continue } - } catch (e: Exception) { } + h = content.indexOf('#', h + 1) } return output.toList() } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt index 0d7edb3584..cb6a2769a6 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt @@ -28,6 +28,43 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]") +/** + * Walks every `#[n]` reference in [content]. + * + * [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match + * starts at position 0 or at a whitespace. That lets the scan jump between `#` + * occurrences with `indexOf` — an intrinsified char search — and apply the regex + * **anchored** at each, instead of letting `findAll` drive the regex engine from + * every position in the string. + * + * Measured on the production content distribution (median 529 B, tail to 767 KB): + * ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense + * text. Equivalence with the previous `findAll` implementation (both callers) is + * guarded by `RegexContentBenchmark` in `commons`. + */ +private inline fun forEachIndexTag( + content: String, + action: (MatchResult) -> Unit, +) { + var h = content.indexOf('#') + while (h >= 0) { + if (h == 0 || content[h - 1].isWhitespace()) { + val match = + try { + tagSearch.matchAt(content, if (h == 0) 0 else h - 1) + } catch (e: Exception) { + null + } + if (match != null) { + action(match) + h = content.indexOf('#', match.range.last + 1) + continue + } + } + h = content.indexOf('#', h + 1) + } +} + /** * Returns the old-style [1] tag that pionts to an index in the tag array */ @@ -36,8 +73,7 @@ fun findIndexTagsWithPeople( tags: TagArray, output: MutableSet = mutableSetOf(), ): List { - val matcher = tagSearch.findAll(content) - matcher.forEach { index -> + forEachIndexTag(content) { index -> try { val tag = index.groups[1]?.value?.let { tags[it.toInt()] } if (tag != null && tag.size > 1 && tag[0] == "p") { @@ -58,8 +94,7 @@ fun findIndexTagsWithEventsOrAddresses( tags: TagArray, output: MutableSet = mutableSetOf(), ): Set { - val matcher = tagSearch.findAll(content) - matcher.forEach { index -> + forEachIndexTag(content) { index -> try { val tag = index.groups[1]?.value?.let { tags[it.toInt()] } if (tag != null && tag.size > 1 && tag[0] == "e") { diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index 866ee0869e..d87eacb874 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -133,21 +133,71 @@ object Nip19Parser { fun hasAny(content: String): Boolean = nip19regex.matches(content) + /** + * True when one of [nip19regex]'s entity prefixes starts at [i]. + * + * Every prefix begins with `n`, and English prose is ~7% `n`, so testing all + * eight prefixes at each `n` is the bulk of the scan's cost. Dispatching on the + * second character first narrows it to at most two candidates — most `n`s are + * rejected by a single char compare. + */ + private fun isCandidateAt( + content: String, + i: Int, + ): Boolean { + val c = content[i] + if (c != 'n' && c != 'N') return false + if (i + 1 >= content.length) return false + return when (content[i + 1]) { + 'p', 'P' -> + content.regionMatches(i, "npub1", 0, 5, ignoreCase = true) || + content.regionMatches(i, "nprofile1", 0, 9, ignoreCase = true) + 's', 'S' -> content.regionMatches(i, "nsec1", 0, 5, ignoreCase = true) + 'o', 'O' -> content.regionMatches(i, "note1", 0, 5, ignoreCase = true) + 'e', 'E' -> + content.regionMatches(i, "nevent1", 0, 7, ignoreCase = true) || + content.regionMatches(i, "nembed1", 0, 7, ignoreCase = true) + 'a', 'A' -> content.regionMatches(i, "naddr1", 0, 6, ignoreCase = true) + 'r', 'R' -> content.regionMatches(i, "nrelay1", 0, 7, ignoreCase = true) + else -> false + } + } + + /** + * Scans [content] for NIP-19 entities. + * + * Walks to each candidate position with cheap char compares and applies + * [nip19regex] **anchored** there, instead of letting `findAll` drive the + * regex engine from every position in the string. `(nostr:)?@?` are optional, + * so anchoring at the entity's own `n` matches the same entities and captures + * the same type/key/trailing groups this function reads. + * + * Measured 9–23x faster across the production content distribution (median + * 529 B, tail to 767 KB): ~19 MB/s -> 170–445 MB/s. Equivalence and speed are + * guarded by `RegexContentBenchmark` in `commons`. Motivation: on an SM-T220 + * heap dump, 2,541 of 4,573 live Matchers were running this regex. + */ fun parseAll(content: String): List { - val matchSequence = nip19regex.findAll(content) val returningList = mutableListOf() - matchSequence.forEach { matcher -> - val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 - val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 - val additionalChars = matcher.groups[7]?.value // additional chars + var i = 0 + val len = content.length + while (i < len) { + if (isCandidateAt(content, i)) { + val matcher = nip19regex.matchAt(content, i) + if (matcher != null) { + val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 + val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 + val additionalChars = matcher.groups[7]?.value // additional chars - if (type != null) { - val parsed = parseComponents(type, key, additionalChars)?.entity + if (type != null) { + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } + } - if (parsed != null) { - returningList.add(parsed) + i = matcher.range.last + 1 + continue } } + i++ } return returningList } From c355dd82772fec53e91561f6fb7d99dee6564961 Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Thu, 13 Aug 2026 15:18:34 -0400 Subject: [PATCH 2/5] test: cover the content parsers' behaviour, not just the rewrite's equivalence MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The optimization PR pinned each rewritten scan against a verbatim copy of the implementation it replaced. That proves "unchanged" but not "correct" — and a coverage review found the underlying behaviour was barely tested at all: - NIP19ParserTest has 37 tests but every one of them exercises `uriToRoute`. `parseAll` — the whole-content scan run on every ingested note — had no behavioural test. - `findHashtags` and both `IndexedTags` scans had no tests anywhere. Adds two commonTest suites (41 tests). commonTest on purpose: these are commonMain parsers and the scans use matchAt/regionMatches, so they must behave identically on JVM, Android, Apple and native — the existing jvmTest benchmark only covers one of those. Nip19ScanTest covers where an entity may sit (start/middle/end, multiline, glued to a preceding word), what may precede it (bare, `nostr:`, `@`, uppercase), what must NOT be mistaken for one (`npub1tooshort`, prose full of n-words, every two-letter prefix stem), invalid checksums, and the greedy trailing group that makes two adjacent entities parse as one. ContentScanTest covers what opens a hashtag (start of content vs each kind of whitespace, and that a non-breaking space does not), what terminates one, and the IndexedTags failure modes: out-of-range indices, non-numeric and unclosed refs, a number too large for Int, tag kinds other than p/e/a, and a tag with no value. Both include a case where an early non-matching `#` must not hide a later real one. Every case was written to fail against a mutated parser, and the two riskiest were verified to do so: removing the `i + 1` bounds guard -> contentEndingInNDoesNotReadPastTheEnd: StringIndexOutOfBoundsException dropping 'e' from the second-char dispatch (nevent1/nembed1) -> findsSeveralEntitiesInOneNote: expected:<3> but was:<2> quartz jvmTest 4,186 -> 4,240, 0 failures. Co-Authored-By: Claude Opus 5 (1M context) --- .../nip10Notes/content/ContentScanTest.kt | 216 ++++++++++++++++++ .../quartz/nip19Bech32/Nip19ScanTest.kt | 170 ++++++++++++++ 2 files changed, 386 insertions(+) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt new file mode 100644 index 0000000000..7e6035e23e --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt @@ -0,0 +1,216 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip10Notes.content + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Behavioural coverage for the two content scans that run on every ingested note. + * + * Both scans jump between `#` positions with `indexOf` and apply their regex + * anchored there, so the cases below deliberately cover what decides a match: + * what precedes the `#`, where it sits in the string, and what terminates the tag. + * + * In `commonTest` on purpose — these are `commonMain` parsers and the scans use + * `matchAt`, so they must behave identically on JVM, Android, Apple and native. + */ +class ContentScanTest { + // ---------- findHashtags ---------- + + @Test + fun hashtagAtStartOfContent() { + assertEquals(listOf("bitcoin"), findHashtags("#bitcoin")) + } + + @Test + fun hashtagAfterEachKindOfWhitespace() { + assertEquals(listOf("a"), findHashtags("x #a")) + assertEquals(listOf("b"), findHashtags("x\n#b")) + assertEquals(listOf("c"), findHashtags("x\t#c")) + assertEquals(listOf("d"), findHashtags("x\r\n#d")) + } + + @Test + fun hashtagMustBePrecededByWhitespaceOrStart() { + // glued to a word: not a hashtag + assertEquals(emptyList(), findHashtags("word#nottag")) + assertEquals(emptyList(), findHashtags("a#b")) + // a non-breaking space is not \s, so it does not open a hashtag either + assertEquals(emptyList(), findHashtags("x\u00a0#nope")) + } + + @Test + fun hashWithNoTagBodyIsNotAHashtag() { + assertEquals(emptyList(), findHashtags("#")) + assertEquals(emptyList(), findHashtags(" #")) + assertEquals(emptyList(), findHashtags("text # more")) + // the second '#' is an excluded char, so it cannot open the tag body + assertEquals(emptyList(), findHashtags("##tag")) + } + + @Test + fun punctuationTerminatesTheTag() { + assertEquals(listOf("tag"), findHashtags("#tag,")) + assertEquals(listOf("tag"), findHashtags("#tag.")) + assertEquals(listOf("tag"), findHashtags("#tag!")) + assertEquals(listOf("tag"), findHashtags("#tag)")) + assertEquals(listOf("tag"), findHashtags("#tag;")) + } + + @Test + fun tagsKeepCharactersOutsideTheExcludedSet() { + assertEquals(listOf("caf\u00e9"), findHashtags("#caf\u00e9")) + assertEquals(listOf("123"), findHashtags("#123")) + assertEquals(listOf("a-b_c"), findHashtags("#a-b_c")) + assertEquals(listOf("\u65e5\u672c"), findHashtags("#\u65e5\u672c")) + } + + @Test + fun multipleHashtagsAndDeduplication() { + assertEquals(listOf("a", "b", "c"), findHashtags("#a #b #c").sorted()) + assertEquals(listOf("dup"), findHashtags("#dup #dup #dup")) + } + + @Test + fun hashtagAtVeryEndOfContent() { + assertEquals(listOf("end"), findHashtags("something #end")) + } + + @Test + fun blankContentReturnsEmpty() { + assertEquals(emptyList(), findHashtags("")) + assertEquals(emptyList(), findHashtags(" ")) + assertEquals(emptyList(), findHashtags("\n\t ")) + } + + @Test + fun callerSuppliedOutputSetIsReusedAndAccumulates() { + val shared = mutableSetOf() + findHashtags("#one", shared) + findHashtags("#two", shared) + assertEquals(listOf("one", "two"), shared.toList().sorted()) + } + + @Test + fun scanDoesNotStopAtTheFirstNonMatchingHash() { + // a glued '#' must not hide a later real hashtag + assertEquals(listOf("real"), findHashtags("word#nottag #real")) + } + + // ---------- IndexedTags ---------- + + private val tags = + arrayOf( + arrayOf("p", "pubkey0"), + arrayOf("e", "event1"), + arrayOf("a", "30023:author:slug"), + arrayOf("p", "pubkey3"), + arrayOf("t", "topic"), + arrayOf("p"), + ) + + @Test + fun indexTagResolvesPeople() { + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#[0]", tags)) + assertEquals(listOf("pubkey3"), findIndexTagsWithPeople("hi #[3]", tags)) + } + + @Test + fun indexTagResolvesEventsAndAddresses() { + assertEquals(setOf("event1"), findIndexTagsWithEventsOrAddresses("#[1]", tags)) + assertEquals(setOf("30023:author:slug"), findIndexTagsWithEventsOrAddresses("#[2]", tags)) + } + + @Test + fun indexTagIgnoresWrongTagKinds() { + // "t" is neither p, e nor a + assertEquals(emptyList(), findIndexTagsWithPeople("#[4]", tags)) + assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[4]", tags)) + // a "p" tag with no value must not blow up or emit + assertEquals(emptyList(), findIndexTagsWithPeople("#[5]", tags)) + } + + @Test + fun indexTagOutOfRangeIsIgnored() { + assertEquals(emptyList(), findIndexTagsWithPeople("#[99]", tags)) + assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[99]", tags)) + } + + @Test + fun indexTagWithHugeNumberDoesNotThrow() { + // does not fit in an Int — must be swallowed, not propagated + assertEquals(emptyList(), findIndexTagsWithPeople("#[99999999999999999999]", tags)) + } + + @Test + fun malformedIndexRefsAreIgnored() { + assertEquals(emptyList(), findIndexTagsWithPeople("#[abc]", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("#[]", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("#[0", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("[0]", tags)) + } + + @Test + fun indexRefMustBePrecededByWhitespaceOrStart() { + assertEquals(emptyList(), findIndexTagsWithPeople("word#[0]", tags)) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("word #[0]", tags)) + } + + @Test + fun multipleIndexRefsAndDeduplication() { + assertEquals(listOf("pubkey0", "pubkey3"), findIndexTagsWithPeople("#[0] #[3]", tags).sorted()) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#[0] #[0]", tags)) + assertEquals( + setOf("30023:author:slug", "event1"), + findIndexTagsWithEventsOrAddresses("#[1] #[2]", tags), + ) + } + + @Test + fun indexScanDoesNotStopAtTheFirstNonMatchingHash() { + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("word#[9] #[0]", tags)) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#hashtag #[0]", tags)) + } + + @Test + fun emptyContentAndEmptyTagArray() { + assertEquals(emptyList(), findIndexTagsWithPeople("", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("#[0]", arrayOf())) + assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[0]", arrayOf())) + } + + @Test + fun hashtagAndIndexRefsCoexistInOneNote() { + val content = "#intro see #[0] and #[1] about #nostr" + assertEquals(listOf("intro", "nostr"), findHashtags(content).sorted()) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople(content, tags)) + assertEquals(setOf("event1"), findIndexTagsWithEventsOrAddresses(content, tags)) + } + + @Test + fun longContentWithNoMarkersTerminates() { + val prose = "the quick brown fox jumps over the lazy dog ".repeat(2_000) + assertTrue(findHashtags(prose).isEmpty()) + assertTrue(findIndexTagsWithPeople(prose, tags).isEmpty()) + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt new file mode 100644 index 0000000000..15fec6d315 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt @@ -0,0 +1,170 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32 + +import com.vitorpamplona.quartz.nip19Bech32.entities.NEvent +import com.vitorpamplona.quartz.nip19Bech32.entities.NNote +import com.vitorpamplona.quartz.nip19Bech32.entities.NPub +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Behavioural coverage for [Nip19Parser.parseAll] — the whole-content scan run on + * every ingested note. + * + * `NIP19ParserTest` covers [Nip19Parser.uriToRoute] (one entity, already isolated). + * This covers the other half: finding entities *inside* arbitrary text — where they + * may sit, what may precede them, and what must NOT be mistaken for one. + * + * In `commonTest` on purpose: [Nip19Parser] is `commonMain` and the scan uses + * `matchAt`/`regionMatches`, so it must behave identically on every target. + */ +class Nip19ScanTest { + companion object { + const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y" + const val NPUB_HEX = "bb3d6543d4a4f45f402b674ecb5cb2155ff12456fcdd2d66d9727660d3037365" + const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv" + const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy" + } + + @Test + fun findsBareEntity() { + val found = Nip19Parser.parseAll(NPUB) + assertEquals(1, found.size) + assertEquals(NPUB_HEX, (found[0] as NPub).hex) + } + + @Test + fun findsEntityWithNostrScheme() { + val found = Nip19Parser.parseAll("hello nostr:$NPUB world") + assertEquals(1, found.size) + assertEquals(NPUB_HEX, (found[0] as NPub).hex) + } + + @Test + fun findsEntityWithAtPrefix() { + val found = Nip19Parser.parseAll("cc @$NPUB thanks") + assertEquals(1, found.size) + assertEquals(NPUB_HEX, (found[0] as NPub).hex) + } + + @Test + fun findsEntityAtStartMiddleAndEndOfContent() { + assertEquals(1, Nip19Parser.parseAll("$NPUB trailing text").size) + assertEquals(1, Nip19Parser.parseAll("leading $NPUB trailing").size) + assertEquals(1, Nip19Parser.parseAll("leading text $NPUB").size) + } + + @Test + fun findsDifferentEntityTypes() { + assertTrue(Nip19Parser.parseAll("see $NOTE")[0] is NNote) + assertTrue(Nip19Parser.parseAll("see $NEVENT")[0] is NEvent) + } + + @Test + fun findsSeveralEntitiesInOneNote() { + val found = Nip19Parser.parseAll("$NPUB then $NOTE and nostr:$NEVENT") + assertEquals(3, found.size) + } + + @Test + fun entityGluedToAPrecedingWordIsStillFound() { + // the regex has no leading whitespace requirement + assertEquals(1, Nip19Parser.parseAll("x$NPUB").size) + } + + @Test + fun uppercaseEntityIsFound() { + // the regex is IGNORE_CASE; bech32 decode is case-insensitive too + assertEquals(1, Nip19Parser.parseAll("NOSTR:${NPUB.uppercase()}").size) + } + + @Test + fun invalidChecksumIsNotReturned() { + val broken = NPUB.dropLast(6) + "qqqqqq" + assertEquals(emptyList(), Nip19Parser.parseAll(broken)) + } + + @Test + fun tooShortNpubIsNotMatched() { + assertEquals(emptyList(), Nip19Parser.parseAll("npub1tooshort")) + assertEquals(emptyList(), Nip19Parser.parseAll("npub1")) + } + + @Test + fun proseWithoutEntitiesReturnsNothing() { + assertEquals(emptyList(), Nip19Parser.parseAll("no entities in this note at all")) + // words that start like a prefix but are not one + assertEquals(emptyList(), Nip19Parser.parseAll("nostrich nope never nan note nsec nprofile")) + } + + @Test + fun contentEndingInNDoesNotReadPastTheEnd() { + // the candidate scan dispatches on the character AFTER 'n' + assertEquals(emptyList(), Nip19Parser.parseAll("n")) + assertEquals(emptyList(), Nip19Parser.parseAll("N")) + assertEquals(emptyList(), Nip19Parser.parseAll("ends with n")) + assertEquals(emptyList(), Nip19Parser.parseAll("trailing N")) + listOf("np", "ne", "na", "nr", "ns", "no", "nn").forEach { + assertEquals(emptyList(), Nip19Parser.parseAll(it), "for '$it'") + assertEquals(emptyList(), Nip19Parser.parseAll("text $it"), "for 'text $it'") + } + } + + @Test + fun emptyContentReturnsNothing() { + assertEquals(emptyList(), Nip19Parser.parseAll("")) + assertEquals(emptyList(), Nip19Parser.parseAll(" ")) + } + + @Test + fun adjacentEntitiesWithNoSeparatorYieldOne() { + // the trailing ([\S]*) group is greedy, so it swallows the second entity — + // documented behaviour, pinned here so the scan rewrite cannot change it + assertEquals(1, Nip19Parser.parseAll(NPUB + NOTE).size) + // separated by a space, both are found + assertEquals(2, Nip19Parser.parseAll("$NPUB $NOTE").size) + } + + @Test + fun trailingPunctuationDoesNotBreakTheEntity() { + assertEquals(1, Nip19Parser.parseAll("thanks nostr:$NPUB!").size) + assertEquals(1, Nip19Parser.parseAll("(nostr:$NPUB)").size) + } + + @Test + fun entityInsideMultilineContent() { + assertEquals(1, Nip19Parser.parseAll("line one\nnostr:$NPUB\nline three").size) + } + + @Test + fun scanDoesNotStopAtTheFirstNonEntityCandidate() { + // "nostrich" and "nevermind" are 'n' candidates that fail; a real entity follows + assertEquals(1, Nip19Parser.parseAll("nostrich nevermind nap $NPUB").size) + } + + @Test + fun longProseWithNoEntitiesTerminates() { + val prose = "the nimble nocturnal nightingale never naps near noon ".repeat(2_000) + assertEquals(emptyList(), Nip19Parser.parseAll(prose)) + } +} From 9ea4b6366e154a9bd1aaf1aac88986d9f05cf9cb Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Thu, 13 Aug 2026 15:29:19 -0400 Subject: [PATCH 3/5] perf: share the anchored scan with parseAllEvents; cover it and uriToRoute MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit parseAllEvents had the same shape as parseAll before it was rewritten — a `findAll` over the whole of the content, restarting the regex engine at every position — and it runs in the composer on the message being typed (`findNostrEventUris`). Both now share one `forEachNip19Match(content, regex, action)` walk. The candidate check covers the union of the prefixes across the three NIP-19 regexes (adding `ncryptsec1`), so a narrower regex simply fails `matchAt` on a prefix it does not accept — still far cheaper than restarting the engine everywhere. That keeps one scan to reason about instead of three copies. parseAllEvents before after no entities 23 MB/s ~1,020 MB/s (44x) with entities 23 MB/s 146-706 MB/s 767 KB content 33.9 ms 0.75 ms parseAll is unchanged by the refactor (911 MB/s, same as before). uriToRoute is deliberately NOT changed. It takes one short URI or id per call — podcast person tags, address deserialization, search queries, deep links — and measures 0.6-4.0 us per call. `find` on a ~60 character string has nothing to win from an anchored walk, so leaving it alone avoids the risk for no gain. Benchmarked anyway so that stays true, and pinned by tests since the shared candidate scan sits next to it. Coverage: 5 new cases in Nip19ScanTest — that parseAllEvents finds event-ish entities, ignores npub/nsec/nprofile, keeps scanning past a profile entity to find a later event (the case where sharing a wider candidate set could have gone wrong), handles the same edges as parseAll, and that uriToRoute still resolves every form including null/empty input. Mutation: making the shared walk stop at the first candidate that fails to match is caught by parseAllEventsSkipsProfilesButStillFindsLaterEvents (expected:<1> but was:<0>). quartz jvmTest 4,240 -> 4,245, 0 failures. Co-Authored-By: Claude Opus 5 (1M context) --- .../prodbench/RegexContentBenchmark.kt | 19 ++++++ .../quartz/nip19Bech32/Nip19Parser.kt | 66 +++++++++++-------- .../quartz/nip19Bech32/Nip19ScanTest.kt | 44 +++++++++++++ 3 files changed, 103 insertions(+), 26 deletions(-) diff --git a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt index 756e17f82c..c0d8760f25 100644 --- a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt +++ b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt @@ -350,6 +350,25 @@ class RegexContentBenchmark { bench("idxTags refs=$m", indexNote(n, m, 30), r) { findIndexTagsWithPeople(it, tags).size } } + println("\n=== parseAllEvents (composer: findNostrEventUris) ===") + corpus.forEach { (n, _, r) -> + bench("parseAllEvents 0 mentions", note(n, 0), r) { Nip19Parser.parseAllEvents(it).size } + } + corpus.forEach { (n, m, r) -> + bench("parseAllEvents m=$m", note(n, m), r) { Nip19Parser.parseAllEvents(it).size } + } + + println("\n=== uriToRoute (short strings: one URI / id per call) ===") + listOf( + "nostr:$NPUB" to 200_000, + NPUB to 200_000, + "nostr:$NEVENT" to 200_000, + "30023:abc:slug" to 200_000, + "not an entity at all" to 200_000, + ).forEach { (uri, reps) -> + bench("uriToRoute ${uri.take(18)}", uri, reps) { if (Nip19Parser.uriToRoute(it) != null) 1 else 0 } + } + println("\n=== RENDER PATH: RichTextParser.parseText (per note, per render) ===") val rtTags = tagArray(30).toImmutableListOfLists() val rtCorpus = diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index d87eacb874..0d8fb3bb4d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -134,7 +134,7 @@ object Nip19Parser { fun hasAny(content: String): Boolean = nip19regex.matches(content) /** - * True when one of [nip19regex]'s entity prefixes starts at [i]. + * True when one of the NIP-19 entity prefixes starts at [i]. * * Every prefix begins with `n`, and English prose is ~7% `n`, so testing all * eight prefixes at each `n` is the bulk of the scan's cost. Dispatching on the @@ -159,10 +159,39 @@ object Nip19Parser { content.regionMatches(i, "nembed1", 0, 7, ignoreCase = true) 'a', 'A' -> content.regionMatches(i, "naddr1", 0, 6, ignoreCase = true) 'r', 'R' -> content.regionMatches(i, "nrelay1", 0, 7, ignoreCase = true) + 'c', 'C' -> content.regionMatches(i, "ncryptsec1", 0, 10, ignoreCase = true) else -> false } } + /** + * Applies [regex] anchored at every NIP-19 candidate position in [content]. + * + * [isCandidateAt] covers the union of the prefixes across the three NIP-19 + * regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does + * not accept — still far cheaper than `findAll` restarting the engine at + * every position in the string. + */ + private inline fun forEachNip19Match( + content: String, + regex: Regex, + action: (MatchResult) -> Unit, + ) { + var i = 0 + val len = content.length + while (i < len) { + if (isCandidateAt(content, i)) { + val match = regex.matchAt(content, i) + if (match != null) { + action(match) + i = match.range.last + 1 + continue + } + } + i++ + } + } + /** * Scans [content] for NIP-19 entities. * @@ -179,43 +208,28 @@ object Nip19Parser { */ fun parseAll(content: String): List { val returningList = mutableListOf() - var i = 0 - val len = content.length - while (i < len) { - if (isCandidateAt(content, i)) { - val matcher = nip19regex.matchAt(content, i) - if (matcher != null) { - val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 - val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 - val additionalChars = matcher.groups[7]?.value // additional chars + forEachNip19Match(content, nip19regex) { matcher -> + val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 + val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 + val additionalChars = matcher.groups[7]?.value // additional chars - if (type != null) { - parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } - } - - i = matcher.range.last + 1 - continue - } + if (type != null) { + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } - i++ } return returningList } + /** Same scan as [parseAll], restricted to the event-ish entities. */ fun parseAllEvents(content: String): List { - val matchSequence = nip19regexEvents.findAll(content) val returningList = mutableListOf() - matchSequence.forEach { matcher -> - val type = matcher.groups[2]?.value // npub1 + forEachNip19Match(content, nip19regexEvents) { matcher -> + val type = matcher.groups[2]?.value // nevent1 val key = matcher.groups[3]?.value // bech32 val additionalChars = matcher.groups[4]?.value // additional chars if (type != null) { - val parsed = parseComponents(type, key, additionalChars)?.entity - - if (parsed != null) { - returningList.add(parsed) - } + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } } return returningList diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt index 15fec6d315..474057d778 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt @@ -162,6 +162,50 @@ class Nip19ScanTest { assertEquals(1, Nip19Parser.parseAll("nostrich nevermind nap $NPUB").size) } + // ---------- parseAllEvents: same scan, event-ish entities only ---------- + + @Test + fun parseAllEventsFindsEventEntities() { + assertEquals(1, Nip19Parser.parseAllEvents("see nostr:$NEVENT").size) + assertEquals(1, Nip19Parser.parseAllEvents("see $NOTE").size) + } + + @Test + fun parseAllEventsIgnoresProfileEntities() { + // npub1/nsec1/nprofile1 are not in nip19regexEvents + assertEquals(emptyList(), Nip19Parser.parseAllEvents(NPUB)) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("hello nostr:$NPUB there")) + } + + @Test + fun parseAllEventsSkipsProfilesButStillFindsLaterEvents() { + // an npub is a candidate position that must not stop the scan + val found = Nip19Parser.parseAllEvents("$NPUB then nostr:$NEVENT") + assertEquals(1, found.size) + assertTrue(found[0] is NEvent) + } + + @Test + fun parseAllEventsHandlesEdgesLikeParseAll() { + assertEquals(emptyList(), Nip19Parser.parseAllEvents("")) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("n")) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("ends with n")) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("no entities here")) + assertEquals(2, Nip19Parser.parseAllEvents("$NOTE $NEVENT").size) + } + + // ---------- uriToRoute: unchanged, pinned against the shared candidate scan ---------- + + @Test + fun uriToRouteStillResolvesEachForm() { + assertTrue(Nip19Parser.uriToRoute("nostr:$NPUB")?.entity is NPub) + assertTrue(Nip19Parser.uriToRoute(NPUB)?.entity is NPub) + assertTrue(Nip19Parser.uriToRoute("nostr:$NEVENT")?.entity is NEvent) + assertEquals(null, Nip19Parser.uriToRoute(null)) + assertEquals(null, Nip19Parser.uriToRoute("not an entity")) + assertEquals(null, Nip19Parser.uriToRoute("")) + } + @Test fun longProseWithNoEntitiesTerminates() { val prose = "the nimble nocturnal nightingale never naps near noon ".repeat(2_000) From 996562c3a6381e0fa254856a0e94f10065952c7d Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Thu, 13 Aug 2026 15:42:47 -0400 Subject: [PATCH 4/5] test: cover the TLV decode paths and tryParseAndClean MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Closes the last gaps the coverage review found. These are read paths every ingested note can reach, and two of them had no positive test anywhere. nprofile1 / naddr1 / nrelay1 / nembed1 were only ever decoded through `uriToRoute`, which receives one isolated entity. Nothing asserted they survive being FOUND INSIDE arbitrary text — the path the content scan takes — or that the relay hints and identifiers they carry come back intact. Adds those, with nprofile round-tripped through `NProfile.create` so the relay list is checked rather than assumed. Two findings while writing them: - `nrelay1` has no encoder. The codebase can parse it (`NRelay.parse`) but there is no `toNRelay()` alongside toNsec/toNpub/toNote/toNEvent/toNProfile/ toNAddress/toNEmbed/toNCryptSec, and the only fixtures anywhere are negative ("nostr:nrelay" -> null). The test builds one from TLV + Bech32 directly, so this is the first positive nrelay case in the suite. Adding the missing encoder is left out of this PR deliberately — it is API surface, not coverage. - `nembed1` had no commonTest coverage at all; its fixtures live only in androidDeviceTest and iosTest, so the GZip + Event decode never ran on JVM. It does now. tryParseAndClean was untested. Covers scheme/@ stripping, null and non-entity input, that it accepts ncryptsec1 while the content scan deliberately does not (different regex), and that `type!! + key` can never render a literal "npub1null" suffix. Measured, not changed: tryParseAndClean runs 0.6-1.1 us per call on the short strings its callers pass — same shape as uriToRoute, nothing to win from an anchored walk. Mutation: pointing NProfile.parse at the wrong TLV field is caught by nprofileWithSeveralRelayHintsKeepsAllOfThem. quartz jvmTest 4,245 -> 4,256, 0 failures. Co-Authored-By: Claude Opus 5 (1M context) --- .../prodbench/RegexContentBenchmark.kt | 9 + .../quartz/nip19Bech32/Nip19DecodeTest.kt | 167 ++++++++++++++++++ 2 files changed, 176 insertions(+) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt diff --git a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt index c0d8760f25..0707816cf4 100644 --- a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt +++ b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt @@ -369,6 +369,15 @@ class RegexContentBenchmark { bench("uriToRoute ${uri.take(18)}", uri, reps) { if (Nip19Parser.uriToRoute(it) != null) 1 else 0 } } + println("\n=== tryParseAndClean (short strings, like uriToRoute) ===") + listOf( + "nostr:$NPUB" to 200_000, + NPUB to 200_000, + "not an entity at all" to 200_000, + ).forEach { (uri, reps) -> + bench("tryParseAndClean ${uri.take(14)}", uri, reps) { if (Nip19Parser.tryParseAndClean(it) != null) 1 else 0 } + } + println("\n=== RENDER PATH: RichTextParser.parseText (per note, per render) ===") val rtTags = tagArray(30).toImmutableListOfLists() val rtCorpus = diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt new file mode 100644 index 0000000000..27dd5d371d --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt @@ -0,0 +1,167 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32 + +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer +import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32 +import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress +import com.vitorpamplona.quartz.nip19Bech32.entities.NEmbed +import com.vitorpamplona.quartz.nip19Bech32.entities.NProfile +import com.vitorpamplona.quartz.nip19Bech32.entities.NRelay +import com.vitorpamplona.quartz.nip19Bech32.tlv.TlvBuilder +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNotNull +import kotlin.test.assertTrue + +/** + * Coverage for the TLV-backed entities reached *through the content scan*, and for + * [Nip19Parser.tryParseAndClean]. + * + * `NIP19ParserTest` decodes these through `uriToRoute`. This asserts the same + * entities survive being found inside arbitrary text, which is the path every + * ingested note takes, and pins the relay hints and identifiers they carry. + * + * `nembed1` had no `commonTest` coverage at all (its only fixtures live in + * `androidDeviceTest`/`iosTest`), and `nrelay1` had no *positive* case anywhere — + * the codebase can parse it but has no encoder for it, so one is built here from + * TLV directly. + */ +class Nip19DecodeTest { + companion object { + const val PUBKEY = "460c25e682fda7832b52d1f22d3d22b3176d972f60dcdc3212ed8c92ef85065c" + const val NADDR = + "naddr1qqyxzmt9w358jum5qyt8wumn8ghj7un9d3shjtnwdaehgu3wvfskueqzypd7v3r24z33cydnk3fmlrd0exe5dlej3506zxs05q4puerp765mzqcyqqq8scsq6mk7u" + const val NEMBED = "nembed1r79ssq9446hkwqhl642ukmku8qg0c92pu7w3j0jyfte8tc7tvg85vmrys8x3sqgle5vjy7jpjswqhphl0kd6yf4sz0n3peyjq5rp3zkat4w6c6j3f7um0724jmfu5456xxgg2yxkn8dp23j64xsn9npcggzafyh2effyntqrqxzja8dp52kpcvc9zqxlj86e8mx05vevzxkeprjkfs4wmppxm3p96vj6yvu2mqgf5l4v99492r2qsggquxuv93uzx244652h2kkj8xseg9xkq0afpygknjtty9j4ju5v0nm9mezux9wyl6s5wr7lzce7cj397mnu0u04ha7aq3w7exelrhe3zs3l3urwa9sp36u80npllrs0hmsxqdn0fsuyav3nv0azjs5suzuurg2uymncjxez8p9xksc2j6gw992enjflgrdd7n5uq2xrpvfrd3rckw624ey0elvm6grr27tyzlf4vaswgm5vc3hdyczsl983g2j8e67r6z5zt30lat84ma4wclkwwxxrcflvdsuwd7346h7zqav4vdwe3gkt9lr87sfk4aqd2aey03tt4eyspldrqcmkx9pqe2pn63rv7grwwalr86akuldnvjm6m87wrw9sdwns8wq0rnsmj57vqwtc3g7hkwum3vl2dda78dwkycgfzw6qna3ufhpatcvq5a4hm4ehl45an8umwt0clf7rn77ctke475qglwu86hhfwhn7dkca4pkfpyc4y75rll6nvr5qc8nlhf8mk22celn5mecvyuzxd830drhdck9tcdpcafymk8wajwu2w8ha8gatggjfvq0a4jlf2sdamzj0ysqks9dk8me3q7a0qpmf6vykurkrcls4pug3u4pn4u26ezx3h8e482n07x2nsmu80dpufxqc0ttcyzhnppguxma4d8aumdawnlsyy7yzcuxl7lw5y9p4nv5h8fn6u8anpm2tsze3p6mgxy9j9uuqfxg2jvlmtjpakna5m4hln0msmw804hnun96h66fh62270yhhljnmmdl7jln07ll5vft7e870hemcld34a09n943ed6629fgtctsftma9q6tf4jfm2p0ukd2j2n2dpz53fqrkk4ctdcy2j5jar095g5jntf6u807ggkzauzt6uqkwk4tg5w7w55kskspc9663zx5dzzzfwpg3q546g2ve4kukr70n0a46eyce2crsqqq247ql5" + const val NCRYPTSEC = "ncryptsec1qgg9947rlpvqu76pj5ecreduf9jxhselq2nae2kghhvd5g7dgjtcxfqtd67p9m0w57lspw8gsq6yphnm8623nsl8xn9j4jdzz84zm3frztj3z7s35vpzmqf6ksu8r89qk5z2zxfmu5gv8th8wclt0h4p" + } + + // ---------- nprofile: pubkey + relay hints ---------- + + @Test + fun nprofileWithRelayHintsSurvivesTheContentScan() { + val relay = RelayUrlNormalizer.normalize("wss://vitor.nostr1.com/") + val encoded = NProfile.create(PUBKEY, listOf(relay)) + + val found = Nip19Parser.parseAll("please follow nostr:$encoded today") + assertEquals(1, found.size) + val profile = found[0] as NProfile + assertEquals(PUBKEY, profile.hex) + assertEquals(listOf(relay), profile.relay) + } + + @Test + fun nprofileWithoutRelayHintsStillDecodes() { + val encoded = NProfile.create(PUBKEY, emptyList()) + val found = Nip19Parser.parseAll(encoded) + assertEquals(1, found.size) + assertEquals(PUBKEY, (found[0] as NProfile).hex) + assertTrue((found[0] as NProfile).relay.isEmpty()) + } + + @Test + fun nprofileWithSeveralRelayHintsKeepsAllOfThem() { + val a = RelayUrlNormalizer.normalize("wss://relay.one/") + val b = RelayUrlNormalizer.normalize("wss://relay.two/") + val encoded = NProfile.create(PUBKEY, listOf(a, b)) + val profile = Nip19Parser.parseAll("x $encoded").single() as NProfile + assertEquals(listOf(a, b), profile.relay) + } + + // ---------- naddr: kind:author:dTag ---------- + + @Test + fun naddrSurvivesTheContentScan() { + val found = Nip19Parser.parseAll("read nostr:$NADDR now") + assertEquals(1, found.size) + assertEquals( + "30818:5be6446aa8a31c11b3b453bf8dafc9b346ff328d1fa11a0fa02a1e6461f6a9b1:amethyst", + (found[0] as NAddress).aTag(), + ) + } + + // ---------- nrelay: no encoder in the codebase, so build one from TLV ---------- + + @Test + fun nrelayDecodesFromTheContentScan() { + val url = "wss://relay.example.com/" + val encoded = + TlvBuilder() + .apply { addString(TlvTypes.SPECIAL.id, url) } + .build() + .let { Bech32.encodeBytes(hrp = "nrelay", it, Bech32.Encoding.Bech32) } + + val found = Nip19Parser.parseAll("join nostr:$encoded please") + assertEquals(1, found.size) + assertEquals(listOf(url), (found[0] as NRelay).relay) + } + + // ---------- nembed: gzipped event, first commonTest coverage ---------- + + @Test + fun nembedDecodesFromTheContentScan() { + // fixture mirrored from NIP19EmbedTests (androidDeviceTest/iosTest), which + // never ran on JVM — the decode goes through GZip + Event parsing + val found = Nip19Parser.parseAll("look at nostr:$NEMBED") + assertEquals(1, found.size) + val embed = found[0] as NEmbed + assertNotNull(embed.event) + } + + @Test + fun truncatedNembedIsRejectedNotThrown() { + val broken = NEMBED.take(NEMBED.length / 2) + assertEquals(emptyList(), Nip19Parser.parseAll(broken)) + } + + // ---------- tryParseAndClean ---------- + + @Test + fun tryParseAndCleanStripsSchemeAndTrailingCharacters() { + val npub = Nip19ScanTest.NPUB + assertEquals(npub, Nip19Parser.tryParseAndClean("nostr:$npub")) + assertEquals(npub, Nip19Parser.tryParseAndClean(npub)) + assertEquals(npub, Nip19Parser.tryParseAndClean("@$npub")) + } + + @Test + fun tryParseAndCleanReturnsNullForNonEntities() { + assertEquals(null, Nip19Parser.tryParseAndClean(null)) + assertEquals(null, Nip19Parser.tryParseAndClean("")) + assertEquals(null, Nip19Parser.tryParseAndClean("just some words")) + assertEquals(null, Nip19Parser.tryParseAndClean("npub1tooshort")) + } + + @Test + fun tryParseAndCleanNeverReturnsALiteralNullSuffix() { + // `type!! + key` would render "npub1null" if the key group were ever absent + listOf("nostr:${Nip19ScanTest.NPUB}", Nip19ScanTest.NOTE, "nostr:${Nip19ScanTest.NEVENT}") + .forEach { assertTrue(Nip19Parser.tryParseAndClean(it)?.endsWith("null") != true, "for $it") } + } + + @Test + fun tryParseAndCleanAcceptsNcryptsecWhichTheContentScanDoesNot() { + // nip19PlusNip46regex includes ncryptsec1; nip19regex (used by parseAll) does not + val ncryptsec = NCRYPTSEC + assertEquals(ncryptsec, Nip19Parser.tryParseAndClean(ncryptsec)) + assertEquals(emptyList(), Nip19Parser.parseAll(ncryptsec)) + } +} From d921981e33751df313b7095a67c63922758fe270 Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Thu, 13 Aug 2026 16:01:35 -0400 Subject: [PATCH 5/5] test: drop the benchmark's copies of code that is now in production MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit While the optimizations were being compared, the benchmark carried its own implementations of each candidate (hasNip19Candidate, anchoredNip19, fastFindHashtags) so variants could be measured side by side. All three algorithms now live in production, so those copies were duplicating shipped code — and measuring the SUPERSEDED version of it, since they still used the eight-prefix loop rather than the second-char dispatch. A reader comparing rows would have concluded the "optimized" variant was slower than production. Removes the three copies, the three sections that measured them, and the two tests that compared production against them (which had become tautologies once production adopted the same algorithm). Nothing is left unguarded — verified rather than assumed, by mutating production and checking what fails: drop the position-0 case in findHashtags -> ContentScanTest.callerSuppliedOutputSetIsReusedAndAccumulates (quartz commonTest, runs on every target — better than the jvmTest-only benchmark case it replaces) stop the scan after the first hashtag -> hashtagsMatchReferenceUnderFuzz The randomised half is kept: `hashtagsMatchReferenceUnderFuzz` still fuzzes 3,000 inputs against a verbatim copy of the pre-optimization implementation, because that is what catches disagreements an anchored scan can have with `findAll` on awkward `#` placement. The explicit cases moved to ContentScanTest. Benchmark now measures only production: 679 -> 517 lines, 8 sections covering every scan the ingest and composer paths touch. Co-Authored-By: Claude Opus 5 (1M context) --- .../prodbench/RegexContentBenchmark.kt | 180 +----------------- 1 file changed, 9 insertions(+), 171 deletions(-) diff --git a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt index 0707816cf4..e8c9c05cbf 100644 --- a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt +++ b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt @@ -85,86 +85,6 @@ class RegexContentBenchmark { return sb.toString() } - private val PREFIXES = - arrayOf("npub1", "nsec1", "note1", "nevent1", "naddr1", "nprofile1", "nrelay1", "nembed1") - - /** - * One O(n) pass over the chars: stop at every 'n'/'N' and test the 8 entity - * prefixes with regionMatches(ignoreCase). Cheap char compares instead of - * driving the regex NFA from every position in the string. - */ - fun hasNip19Candidate(content: String): Boolean { - var i = 0 - val n = content.length - while (i < n) { - val c = content[i] - if (c == 'n' || c == 'N') { - for (p in PREFIXES) { - if (content.regionMatches(i, p, 0, p.length, ignoreCase = true)) return true - } - } - i++ - } - return false - } - - /** - * Variant 2: never let the NFA scan. Walk to each candidate position with - * cheap char compares, then apply the regex ANCHORED there via matchAt. - * `(nostr:)?@?` are optional, so anchoring at the entity's 'n' still matches - * and still captures the type/key/trailing groups parseAll uses. - */ - fun anchoredNip19(content: String): Int { - var i = 0 - var hits = 0 - val n = content.length - while (i < n) { - val c = content[i] - if (c == 'n' || c == 'N') { - var candidate = false - for (p in PREFIXES) { - if (content.regionMatches(i, p, 0, p.length, ignoreCase = true)) { - candidate = true - break - } - } - if (candidate) { - val m = Nip19Parser.nip19regex.matchAt(content, i) - if (m != null) { - hits++ - i = m.range.last + 1 - continue - } - } - } - i++ - } - return hits - } - - /** - * Candidate: the regex requires `(?:\s|\A)` immediately before the `#`, so - * every match starts at a whitespace (or at position 0). Jump between '#' - * occurrences with indexOf and anchor the regex there. - */ - fun fastFindHashtags(content: String): List { - if (content.isBlank()) return emptyList() - val out = mutableSetOf() - var h = content.indexOf('#') - while (h >= 0) { - if (h == 0 || content[h - 1].isWhitespace()) { - val m = hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1) - if (m != null) { - m.groups[1]?.value?.let { if (it.isNotBlank()) out.add(it) } - h = content.indexOf('#', m.range.last + 1) - continue - } - } - h = content.indexOf('#', h + 1) - } - return out.toList() - } - /** * REFERENCE: the pre-optimization `findHashtags`, verbatim. Production is * compared against this so the guard can never become a tautology. @@ -321,26 +241,6 @@ class RegexContentBenchmark { bench("hashtags m=$m", note(n, m), r) { findHashtags(it).size } } - println("\n=== OPTIMIZED: literal pre-scan, then regex only if a candidate exists ===") - corpus.forEach { (n, _, r) -> - bench("opt nip19 0 mentions", note(n, 0), r) { if (hasNip19Candidate(it)) findNostrUris(it).size else 0 } - } - corpus.forEach { (n, m, r) -> - bench("opt nip19 m=$m", note(n, m), r) { if (hasNip19Candidate(it)) findNostrUris(it).size else 0 } - } - - println("\n=== findHashtags — content with NO '#' ===") - corpus.forEach { (n, _, r) -> - bench("hashtags none", note(n, 0, hashtags = false), r) { findHashtags(it).size } - } - println("\n=== findHashtags OPTIMIZED (indexOf('#') + anchored) ===") - corpus.forEach { (n, _, r) -> - bench("opt hashtags none", note(n, 0, hashtags = false), r) { fastFindHashtags(it).size } - } - corpus.forEach { (n, m, r) -> - bench("opt hashtags dense", note(n, m), r) { fastFindHashtags(it).size } - } - println("\n=== IndexedTags findIndexTagsWithPeople ===") val tags = tagArray(30) corpus.forEach { (n, _, r) -> @@ -402,42 +302,18 @@ class RegexContentBenchmark { RichTextParser().parseText(it, rtTags, null).paragraphs.size } } - - println("\n=== OPTIMIZED v2: anchored matchAt at candidate positions (no NFA scan) ===") - corpus.forEach { (n, _, r) -> - bench("v2 nip19 0 mentions", note(n, 0), r) { anchoredNip19(it) } - } - corpus.forEach { (n, m, r) -> - bench("v2 nip19 m=$m", note(n, m), r) { anchoredNip19(it) } - } } - /** The fast hashtag scan must return exactly what the production scan returns. */ + /** + * Fuzz `findHashtags` against a verbatim copy of the implementation it replaced. + * + * The explicit behavioural cases live in quartz's `ContentScanTest` (commonTest, + * so they run on every target). What is kept here is the randomised half: text + * peppered with `#` in awkward positions, which is where an anchored scan can + * disagree with the original `findAll`. + */ @Test - fun fastHashtagsMatchesProduction() { - val cases = - listOf( - "#one #two #three", - "no hashtags here", - "#start of line", - "mid #tag then more", - "email a@b.com and #tag", - "punct #tag, #tag2. #tag3!", - "nospace#nottag should not match", - "\n#afterNewline", - "", - " ", - note(4_000, 0), - note(20_000, 3), - note(4_000, 0, hashtags = false), - ) - cases.forEach { c -> - val ref = referenceFindHashtags(c).sorted() - val prod = findHashtags(c).sorted() - check(ref == prod) { "mismatch on ${c.take(40).replace("\n", "\\n")}: reference=$ref production=$prod" } - } - - // fuzz: random text peppered with '#' in awkward positions + fun hashtagsMatchReferenceUnderFuzz() { val rnd = kotlin.random.Random(20260813) val alphabet = " \n\t#abcXYZ.,!@\u00a0-_0123" repeat(3_000) { @@ -638,42 +514,4 @@ class RegexContentBenchmark { check(ref.size == prod) { "nip19 mismatch on ${c.take(50)}: reference=${ref.size} production=$prod" } } } - - /** v2 must find exactly the same number of entities as the production scan. */ - @Test - fun anchoredMatchesProductionCount() { - listOf(0, 1, 2, 5, 40).forEach { m -> - listOf(200, 4_000, 68_000).forEach { size -> - val c = note(size, m) - val real = Nip19Parser.nip19regex.findAll(c).count() - val fast = anchoredNip19(c) - check(real == fast) { "size=$size m=$m: production found $real, anchored found $fast" } - } - } - } - - /** - * Correctness gate for the pre-scan: it must never hide a real match. - * Anything the regex finds must also be reported as a candidate. - */ - @Test - fun preScanNeverMissesAMatch() { - val cases = - listOf( - "nostr:$NPUB", - "@$NPUB", - NPUB, - "text before nostr:$NEVENT and after", - "UPPER ${NPUB.uppercase()} case", - "no entities here at all, just prose about nostr and npubs", - "npub1tooshort", - "", - ) - cases.forEach { c -> - val real = Nip19Parser.parseAll(c).isNotEmpty() - if (real) check(hasNip19Candidate(c)) { "pre-scan missed a real match in: ${c.take(40)}" } - } - // and it must actually filter the common case - check(!hasNip19Candidate(note(4_000, 0))) { "pre-scan failed to reject prose with no entities" } - } }