diff --git a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt index 8d5fd30da0..c4bdde30a6 100644 --- a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt +++ b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt @@ -416,8 +416,12 @@ class RichTextParser { tags: ImmutableListOfLists, ): Segment { // First #[n] + // [tagIndex] requires the literal "#[", so a word without it can never match. + // Every plain "#hashtag" reaches this function, and the regex scan is far more + // expensive than the substring check that rules it out — so gate on the literal. + // Uses contains(), not startsWith(), because find() also matches "#[n]" mid-word. try { - val matcher = tagIndex.find(word) + val matcher = if (word.contains("#[")) tagIndex.find(word) else null if (matcher != null) { val index = matcher.groups[1]?.value?.toInt() val suffix = matcher.groups[2]?.value diff --git a/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt new file mode 100644 index 0000000000..e8c9c05cbf --- /dev/null +++ b/commons/src/jvmTest/kotlin/com/vitorpamplona/amethyst/commons/prodbench/RegexContentBenchmark.kt @@ -0,0 +1,517 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.commons.prodbench + +import com.vitorpamplona.amethyst.commons.model.toImmutableListOfLists +import com.vitorpamplona.amethyst.commons.richtext.RichTextParser +import com.vitorpamplona.amethyst.commons.richtext.RichTextParser.Companion.tagIndex +import com.vitorpamplona.quartz.nip10Notes.content.findHashtags +import com.vitorpamplona.quartz.nip10Notes.content.findIndexTagsWithEventsOrAddresses +import com.vitorpamplona.quartz.nip10Notes.content.findIndexTagsWithPeople +import com.vitorpamplona.quartz.nip10Notes.content.findNostrUris +import com.vitorpamplona.quartz.nip10Notes.content.hashtagSearch +import com.vitorpamplona.quartz.nip10Notes.content.tagSearch +import com.vitorpamplona.quartz.nip19Bech32.Nip19Parser +import kotlin.test.Test + +/** + * Measures the regex scans Amethyst runs over note **content**. + * + * Why: an on-device heap dump (SM-T220, Dr. Edo's account) found 2,541 of 4,573 + * live `java.util.regex.Matcher` instances running [Nip19Parser.nip19regex], + * reached from `BaseNoteEvent.citedNIP19()` during ingest. Kotlin's `findAll` + * allocates a **new Matcher per match**, each retaining the whole input + * (`jvmMain/kotlin/text/regex/Regex.kt`: `matcher.pattern().matcher(input)`), + * so a long article with N mentions builds N matchers over the full text. + * + * Corpus sizes come from that same dump's measured distribution of the strings + * these matchers held: median 529 B, p90/max 767 KB (long-form articles — the + * v1.13.0 release notes at 68 KB, an essay at 50 KB). + * + * Deterministic and offline. Prints ns/op and MB/s; no assertions on wall time + * (CI machines vary) beyond a sanity check that the scans return results. + */ +class RegexContentBenchmark { + companion object { + private const val NPUB = "npub180cvv07tjdrrgpa0j7j7tmnyl2yr6yr7l8j4s3evf6u64th6gkwsyjh6w6" + private const val NEVENT = "nevent1qqstna2yrezu5wghjvswqqculvvwxsrcvu7uc0f78gan4xqhvz49d9spr3mhxue69uhkummnw3ez6un9d3shjtn4de6x2argwghx6egpr4mhxue69uhkummnw3ez6ur4vgh8wetvd3hhyer9wghxuet5nxnepm" + + /** A note with exactly [mentions] nostr: URIs, padded with prose to ~[targetBytes]. */ + fun note( + targetBytes: Int, + mentions: Int, + hashtags: Boolean = true, + ): String { + val filler = + if (hashtags) { + "A few months ago a nostrich was switching from iOS to Android and asked for " + + "suggestions for #Nostr apps to try out. Here is what came back, with notes. " + } else { + "A few months ago a nostrich was switching from iOS to Android and asked for " + + "suggestions for great apps to try out. Here is what came back, with notes. " + } + val sb = StringBuilder(targetBytes + 4096) + var placed = 0 + val stride = if (mentions > 0) targetBytes / (mentions + 1) else Int.MAX_VALUE + while (sb.length < targetBytes) { + sb.append(filler) + if (placed < mentions && sb.length >= (placed + 1).toLong() * stride) { + sb.append("nostr:").append(if (placed % 2 == 0) NPUB else NEVENT).append(' ') + placed++ + } + } + while (placed < mentions) { + sb.append("nostr:").append(if (placed % 2 == 0) NPUB else NEVENT).append(' ') + placed++ + } + return sb.toString() + } + + /** + * REFERENCE: the pre-optimization `findHashtags`, verbatim. Production is + * compared against this so the guard can never become a tautology. + */ + fun referenceFindHashtags(content: String): List { + if (content.isBlank()) return emptyList() + val output = mutableSetOf() + hashtagSearch.findAll(content).forEach { + try { + val tag = it.groups[1]?.value + if (tag != null && tag.isNotBlank()) output.add(tag) + } catch (e: Exception) { + } + } + return output.toList() + } + + /** REFERENCE: pre-optimization nip19 scan (findAll), resolved via the public uriToRoute. */ + fun referenceNip19(content: String): List = + Nip19Parser.nip19regex + .findAll(content) + .mapNotNull { m -> + val type = m.groups[3]?.value ?: m.groups[5]?.value + val key = m.groups[4]?.value ?: m.groups[6]?.value + if (type != null && key != null) { + Nip19Parser.uriToRoute(type + key)?.entity?.let { type + key } + } else { + null + } + }.toList() + + /** A tag array with [n] p/e/a entries, for the #[index] scans. */ + fun tagArray(n: Int): Array> = + Array(n) { i -> + when (i % 3) { + 0 -> arrayOf("p", "%064x".format(i)) + 1 -> arrayOf("e", "%064x".format(i + 1000)) + else -> arrayOf("a", "30023:%064x:slug$i".format(i)) + } + } + + /** A note with [refs] legacy #[n] references, padded to ~[targetBytes]. */ + fun indexNote( + targetBytes: Int, + refs: Int, + tagCount: Int, + ): String { + val filler = "Legacy index refs used to link people and events inline in the content. " + val sb = StringBuilder(targetBytes + 4096) + var placed = 0 + val stride = if (refs > 0) targetBytes / (refs + 1) else Int.MAX_VALUE + while (sb.length < targetBytes) { + sb.append(filler) + if (placed < refs && sb.length >= (placed + 1).toLong() * stride) { + sb.append("#[").append(placed % tagCount).append("] ") + placed++ + } + } + while (placed < refs) { + sb.append("#[").append(placed % tagCount).append("] ") + placed++ + } + return sb.toString() + } + + /** REFERENCE: pre-optimization findIndexTagsWithPeople, verbatim. */ + fun referenceIndexPeople( + content: String, + tags: Array>, + ): List { + val output = mutableSetOf() + tagSearch.findAll(content).forEach { index -> + try { + val tag = index.groups[1]?.value?.let { tags[it.toInt()] } + if (tag != null && tag.size > 1 && tag[0] == "p") output.add(tag[1]) + } catch (e: Exception) { + } + } + return output.toList() + } + + /** REFERENCE: pre-optimization findIndexTagsWithEventsOrAddresses, verbatim. */ + fun referenceIndexEvents( + content: String, + tags: Array>, + ): Set { + val output = mutableSetOf() + tagSearch.findAll(content).forEach { index -> + try { + val tag = index.groups[1]?.value?.let { tags[it.toInt()] } + if (tag != null && tag.size > 1 && tag[0] == "e") output.add(tag[1]) + if (tag != null && tag.size > 1 && tag[0] == "a") output.add(tag[1]) + } catch (e: Exception) { + } + } + return output + } + + fun bench( + label: String, + input: String, + reps: Int, + op: (String) -> Int, + ) { + repeat(maxOf(reps / 4, 2)) { op(input) } // warmup + val t0 = System.nanoTime() + var sink = 0 + repeat(reps) { sink += op(input) } + val ns = (System.nanoTime() - t0) / reps + val mbps = input.length.toDouble() / ns * 1000.0 // bytes/ns -> MB/s + println( + String.format( + "%-34s %9d B %9d ns/op %8.1f MB/s (hits=%d)", + label, + input.length, + ns, + mbps, + sink / reps, + ), + ) + } + } + + @Test + fun contentScans() { + // (bytes, mentions, reps) — mirrors the measured distribution + val corpus = + listOf( + Triple(120, 1, 20_000), // short note + Triple(529, 2, 20_000), // MEDIAN of what matchers held + Triple(4_000, 5, 5_000), // long note + Triple(68_000, 40, 200), // v1.13.0 release notes + Triple(767_000, 120, 20), // observed MAX + ) + + // Guard the corpus: a benchmark that silently matches nothing measures nothing. + corpus.forEach { (n, m, _) -> + val found = Nip19Parser.parseAll(note(n, m)).size + check(found >= m) { "corpus broken: ${n}B/$m mentions parsed only $found entities" } + } + + println("\n=== nip19 findNostrUris — NO matches (the common case) ===") + corpus.forEach { (n, _, r) -> + bench("nip19 0 mentions", note(n, 0), r) { findNostrUris(it).size } + } + + println("\n=== nip19 findNostrUris — WITH matches (Matcher per match) ===") + corpus.forEach { (n, m, r) -> + bench("nip19 m=$m", note(n, m), r) { findNostrUris(it).size } + } + + println("\n=== findHashtags (ContentHashTags.hashtagSearch) ===") + corpus.forEach { (n, m, r) -> + bench("hashtags m=$m", note(n, m), r) { findHashtags(it).size } + } + + println("\n=== IndexedTags findIndexTagsWithPeople ===") + val tags = tagArray(30) + corpus.forEach { (n, _, r) -> + bench("idxTags none", indexNote(n, 0, 30), r) { findIndexTagsWithPeople(it, tags).size } + } + corpus.forEach { (n, m, r) -> + bench("idxTags refs=$m", indexNote(n, m, 30), r) { findIndexTagsWithPeople(it, tags).size } + } + + println("\n=== parseAllEvents (composer: findNostrEventUris) ===") + corpus.forEach { (n, _, r) -> + bench("parseAllEvents 0 mentions", note(n, 0), r) { Nip19Parser.parseAllEvents(it).size } + } + corpus.forEach { (n, m, r) -> + bench("parseAllEvents m=$m", note(n, m), r) { Nip19Parser.parseAllEvents(it).size } + } + + println("\n=== uriToRoute (short strings: one URI / id per call) ===") + listOf( + "nostr:$NPUB" to 200_000, + NPUB to 200_000, + "nostr:$NEVENT" to 200_000, + "30023:abc:slug" to 200_000, + "not an entity at all" to 200_000, + ).forEach { (uri, reps) -> + bench("uriToRoute ${uri.take(18)}", uri, reps) { if (Nip19Parser.uriToRoute(it) != null) 1 else 0 } + } + + println("\n=== tryParseAndClean (short strings, like uriToRoute) ===") + listOf( + "nostr:$NPUB" to 200_000, + NPUB to 200_000, + "not an entity at all" to 200_000, + ).forEach { (uri, reps) -> + bench("tryParseAndClean ${uri.take(14)}", uri, reps) { if (Nip19Parser.tryParseAndClean(it) != null) 1 else 0 } + } + + println("\n=== RENDER PATH: RichTextParser.parseText (per note, per render) ===") + val rtTags = tagArray(30).toImmutableListOfLists() + val rtCorpus = + listOf( + Triple(120, 1, 3_000), + Triple(529, 2, 3_000), + Triple(4_000, 5, 500), + Triple(68_000, 40, 20), + ) + rtCorpus.forEach { (n, m, r) -> + bench("parseText hashtags m=$m", note(n, m), r) { + RichTextParser().parseText(it, rtTags, null).paragraphs.size + } + } + rtCorpus.forEach { (n, _, r) -> + bench("parseText no '#'", note(n, 0, hashtags = false), r) { + RichTextParser().parseText(it, rtTags, null).paragraphs.size + } + } + rtCorpus.forEach { (n, m, r) -> + bench("parseText #[n] refs", indexNote(n, m, 30), r) { + RichTextParser().parseText(it, rtTags, null).paragraphs.size + } + } + } + + /** + * Fuzz `findHashtags` against a verbatim copy of the implementation it replaced. + * + * The explicit behavioural cases live in quartz's `ContentScanTest` (commonTest, + * so they run on every target). What is kept here is the randomised half: text + * peppered with `#` in awkward positions, which is where an anchored scan can + * disagree with the original `findAll`. + */ + @Test + fun hashtagsMatchReferenceUnderFuzz() { + val rnd = kotlin.random.Random(20260813) + val alphabet = " \n\t#abcXYZ.,!@\u00a0-_0123" + repeat(3_000) { + val c = (1..rnd.nextInt(0, 80)).map { alphabet[rnd.nextInt(alphabet.length)] }.joinToString("") + val ref = referenceFindHashtags(c).sorted() + val prod = findHashtags(c).sorted() + check(ref == prod) { "fuzz mismatch on ${c.replace("\n", "\\n")}: reference=$ref production=$prod" } + } + } + + /** + * Isolated A/B for the `#[` gate, interleaved with medians. + * + * parseText is allocation-heavy and its whole-parse timings swing ±50% run to + * run (the unchanged no-'#' control arm moved that much), which is far larger + * than this effect — so measure the gated call on its own instead. + */ + @Test + fun hashGateIsolatedAb() { + // realistic mix: mostly plain hashtags, a few legacy #[n] refs + val words = + buildList { + repeat(90) { add("#hashtag$it") } + repeat(10) { add("#[$it]") } + } + + fun ungated(): Int { + var n = 0 + words.forEach { if (tagIndex.find(it) != null) n++ } + return n + } + + fun gated(): Int { + var n = 0 + words.forEach { if (it.contains("#[") && tagIndex.find(it) != null) n++ } + return n + } + check(ungated() == gated()) { "gate changed the result" } + repeat(2_000) { + ungated() + gated() + } // warmup both + val a = mutableListOf() + val b = mutableListOf() + repeat(21) { + var t = System.nanoTime() + repeat(2_000) { ungated() } + a.add(System.nanoTime() - t) + t = System.nanoTime() + repeat(2_000) { gated() } + b.add(System.nanoTime() - t) + } + a.sort() + b.sort() + val am = a[a.size / 2] / 2_000 + val bm = b[b.size / 2] / 2_000 + println( + "\nHASH GATE (100 words: 90 plain hashtags + 10 #[n]) median of 21:" + + "\n ungated tagIndex.find : $am ns/word-set" + + "\n gated with contains : $bm ns/word-set" + + "\n speedup : ${String.format("%.2f", am.toDouble() / bm)}x", + ) + } + + /** + * Segment-level guard for the RichTextParser '#' path. + * + * Expectations are HARDCODED from the behaviour before the `#[` gate was added, + * so this pins the parser against the code it replaced rather than against itself. + */ + @Test + fun richTextHashSegmentsUnchanged() { + val tags = tagArray(30).toImmutableListOfLists() + val expected = + listOf( + "#Nostr" to "HashTagSegment|#Nostr", + "#[0]" to "HashIndexUserSegment|#[0]", + "#[1]suffix" to "HashIndexEventSegment|#[1]suffix", + "#[999]" to "RegularTextSegment|#[999]", + "#[abc]" to "RegularTextSegment|#[abc]", + "#[]" to "RegularTextSegment|#[]", + "#" to "RegularTextSegment|#", + "##tag" to "HashTagSegment|##tag", + "#tag," to "HashTagSegment|#tag,", + "#tag." to "HashTagSegment|#tag.", + "#a#b" to "HashTagSegment|#a#b", + "#[2]#tail" to "HashIndexEventSegment|#[2]#tail", + "#\u00e9t\u00e9" to "HashTagSegment|#\u00e9t\u00e9", + "#123" to "HashTagSegment|#123", + "#[0]!" to "HashIndexUserSegment|#[0]!", + "plain" to "RegularTextSegment|plain", + "#tag)" to "HashTagSegment|#tag)", + "#-dash" to "HashTagSegment|#-dash", + "#_under" to "HashTagSegment|#_under", + // '#[' appearing mid-word must still reach the index parser + "#a#[0]" to "HashIndexUserSegment|#a#[0]", + ) + expected.forEach { (word, want) -> + val got = + RichTextParser() + .parseText(word, tags, null) + .paragraphs + .flatMap { p -> p.words.map { it::class.simpleName + "|" + it.segmentText } } + .joinToString(";") + check(got == want) { "segment mismatch for '$word': want=$want got=$got" } + } + } + + /** Both IndexedTags scans must equal the findAll-based references. */ + @Test + fun indexedTagsMatchReference() { + val tags = tagArray(30) + val cases = + listOf( + "#[0] #[1] #[2]", + "no refs here", + "#[0] at start", + "mid #[5] ref", + "nospace#[3] should not match", + "\n#[7] after newline", + "#[999] out of range", + "#[abc] not numeric", + "#[] empty", + "", + " ", + indexNote(2_000, 0, 30), + indexNote(4_000, 6, 30), + indexNote(20_000, 20, 30), + ) + cases.forEach { c -> + check(referenceIndexPeople(c, tags).sorted() == findIndexTagsWithPeople(c, tags).sorted()) { + "people mismatch on ${c.take(40).replace("\n", "\\n")}" + } + check(referenceIndexEvents(c, tags).sorted() == findIndexTagsWithEventsOrAddresses(c, tags).sorted()) { + "events mismatch on ${c.take(40).replace("\n", "\\n")}" + } + } + val rnd = kotlin.random.Random(20260813) + val alphabet = " \n#[]0123456789abc\u00a0" + repeat(3_000) { + val c = (1..rnd.nextInt(0, 60)).map { alphabet[rnd.nextInt(alphabet.length)] }.joinToString("") + check(referenceIndexPeople(c, tags).sorted() == findIndexTagsWithPeople(c, tags).sorted()) { + "fuzz people mismatch on ${c.replace("\n", "\\n")}" + } + check(referenceIndexEvents(c, tags).sorted() == findIndexTagsWithEventsOrAddresses(c, tags).sorted()) { + "fuzz events mismatch on ${c.replace("\n", "\\n")}" + } + } + } + + /** Production parseAll must equal the findAll-based reference scan. */ + @Test + fun nip19MatchesReferenceScan() { + val cases = + listOf( + "nostr:$NPUB", + "@$NPUB", + NPUB, + "text nostr:$NEVENT tail", + "two $NPUB and nostr:$NEVENT here", + "none at all", + "npub1tooshort", + "", + note(200, 1), + note(4_000, 5), + note(20_000, 12), + note(4_000, 0), + // every prefix standalone + "npub1", + "nsec1", + "note1", + "nevent1", + "naddr1", + "nprofile1", + "nrelay1", + "nembed1", + // 'n' as the LAST char: the second-char dispatch must not read past the end + "n", + "N", + "ends with n", + "trailing N", + "nn", + "np", + "ne", + "na", + "nr", + "ns", + "no", + // case + adjacency + "NOSTR:" + NPUB.uppercase(), + "x" + NPUB, + NPUB + NEVENT, + NPUB + " " + NEVENT, + ) + cases.forEach { c -> + val ref = referenceNip19(c) + val prod = Nip19Parser.parseAll(c).size + check(ref.size == prod) { "nip19 mismatch on ${c.take(50)}: reference=${ref.size} production=$prod" } + } + } +} diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt index f14b2ebe20..51a70d5331 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt @@ -22,21 +22,45 @@ package com.vitorpamplona.quartz.nip10Notes.content val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)") +/** + * Collects the hashtags in [content]. + * + * [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match + * starts either at position 0 or at a whitespace. That lets the scan jump between + * `#` occurrences with `indexOf` — an intrinsified char search — and apply the + * regex **anchored** at each, instead of letting `findAll` drive the regex engine + * from every position in the string. + * + * Measured on the production content distribution (median 529 B, tail to 767 KB): + * ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the + * content has no `#` at all (up to 300x). Equivalence with the previous `findAll` + * implementation is guarded by `RegexContentBenchmark` in `commons`. + */ fun findHashtags( content: String, output: MutableSet = mutableSetOf(), ): List { if (content.isBlank()) return emptyList() - val matcher = hashtagSearch.findAll(content) - matcher.forEach { - try { - val tag = it.groups[1]?.value - if (tag != null && tag.isNotBlank()) { - output.add(tag) + var h = content.indexOf('#') + while (h >= 0) { + if (h == 0 || content[h - 1].isWhitespace()) { + val match = + try { + hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1) + } catch (e: Exception) { + null + } + if (match != null) { + val tag = match.groups[1]?.value + if (tag != null && tag.isNotBlank()) { + output.add(tag) + } + h = content.indexOf('#', match.range.last + 1) + continue } - } catch (e: Exception) { } + h = content.indexOf('#', h + 1) } return output.toList() } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt index 0d7edb3584..cb6a2769a6 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt @@ -28,6 +28,43 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]") +/** + * Walks every `#[n]` reference in [content]. + * + * [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match + * starts at position 0 or at a whitespace. That lets the scan jump between `#` + * occurrences with `indexOf` — an intrinsified char search — and apply the regex + * **anchored** at each, instead of letting `findAll` drive the regex engine from + * every position in the string. + * + * Measured on the production content distribution (median 529 B, tail to 767 KB): + * ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense + * text. Equivalence with the previous `findAll` implementation (both callers) is + * guarded by `RegexContentBenchmark` in `commons`. + */ +private inline fun forEachIndexTag( + content: String, + action: (MatchResult) -> Unit, +) { + var h = content.indexOf('#') + while (h >= 0) { + if (h == 0 || content[h - 1].isWhitespace()) { + val match = + try { + tagSearch.matchAt(content, if (h == 0) 0 else h - 1) + } catch (e: Exception) { + null + } + if (match != null) { + action(match) + h = content.indexOf('#', match.range.last + 1) + continue + } + } + h = content.indexOf('#', h + 1) + } +} + /** * Returns the old-style [1] tag that pionts to an index in the tag array */ @@ -36,8 +73,7 @@ fun findIndexTagsWithPeople( tags: TagArray, output: MutableSet = mutableSetOf(), ): List { - val matcher = tagSearch.findAll(content) - matcher.forEach { index -> + forEachIndexTag(content) { index -> try { val tag = index.groups[1]?.value?.let { tags[it.toInt()] } if (tag != null && tag.size > 1 && tag[0] == "p") { @@ -58,8 +94,7 @@ fun findIndexTagsWithEventsOrAddresses( tags: TagArray, output: MutableSet = mutableSetOf(), ): Set { - val matcher = tagSearch.findAll(content) - matcher.forEach { index -> + forEachIndexTag(content) { index -> try { val tag = index.groups[1]?.value?.let { tags[it.toInt()] } if (tag != null && tag.size > 1 && tag[0] == "e") { diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index 866ee0869e..0d8fb3bb4d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -133,39 +133,103 @@ object Nip19Parser { fun hasAny(content: String): Boolean = nip19regex.matches(content) + /** + * True when one of the NIP-19 entity prefixes starts at [i]. + * + * Every prefix begins with `n`, and English prose is ~7% `n`, so testing all + * eight prefixes at each `n` is the bulk of the scan's cost. Dispatching on the + * second character first narrows it to at most two candidates — most `n`s are + * rejected by a single char compare. + */ + private fun isCandidateAt( + content: String, + i: Int, + ): Boolean { + val c = content[i] + if (c != 'n' && c != 'N') return false + if (i + 1 >= content.length) return false + return when (content[i + 1]) { + 'p', 'P' -> + content.regionMatches(i, "npub1", 0, 5, ignoreCase = true) || + content.regionMatches(i, "nprofile1", 0, 9, ignoreCase = true) + 's', 'S' -> content.regionMatches(i, "nsec1", 0, 5, ignoreCase = true) + 'o', 'O' -> content.regionMatches(i, "note1", 0, 5, ignoreCase = true) + 'e', 'E' -> + content.regionMatches(i, "nevent1", 0, 7, ignoreCase = true) || + content.regionMatches(i, "nembed1", 0, 7, ignoreCase = true) + 'a', 'A' -> content.regionMatches(i, "naddr1", 0, 6, ignoreCase = true) + 'r', 'R' -> content.regionMatches(i, "nrelay1", 0, 7, ignoreCase = true) + 'c', 'C' -> content.regionMatches(i, "ncryptsec1", 0, 10, ignoreCase = true) + else -> false + } + } + + /** + * Applies [regex] anchored at every NIP-19 candidate position in [content]. + * + * [isCandidateAt] covers the union of the prefixes across the three NIP-19 + * regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does + * not accept — still far cheaper than `findAll` restarting the engine at + * every position in the string. + */ + private inline fun forEachNip19Match( + content: String, + regex: Regex, + action: (MatchResult) -> Unit, + ) { + var i = 0 + val len = content.length + while (i < len) { + if (isCandidateAt(content, i)) { + val match = regex.matchAt(content, i) + if (match != null) { + action(match) + i = match.range.last + 1 + continue + } + } + i++ + } + } + + /** + * Scans [content] for NIP-19 entities. + * + * Walks to each candidate position with cheap char compares and applies + * [nip19regex] **anchored** there, instead of letting `findAll` drive the + * regex engine from every position in the string. `(nostr:)?@?` are optional, + * so anchoring at the entity's own `n` matches the same entities and captures + * the same type/key/trailing groups this function reads. + * + * Measured 9–23x faster across the production content distribution (median + * 529 B, tail to 767 KB): ~19 MB/s -> 170–445 MB/s. Equivalence and speed are + * guarded by `RegexContentBenchmark` in `commons`. Motivation: on an SM-T220 + * heap dump, 2,541 of 4,573 live Matchers were running this regex. + */ fun parseAll(content: String): List { - val matchSequence = nip19regex.findAll(content) val returningList = mutableListOf() - matchSequence.forEach { matcher -> + forEachNip19Match(content, nip19regex) { matcher -> val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 val additionalChars = matcher.groups[7]?.value // additional chars if (type != null) { - val parsed = parseComponents(type, key, additionalChars)?.entity - - if (parsed != null) { - returningList.add(parsed) - } + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } } return returningList } + /** Same scan as [parseAll], restricted to the event-ish entities. */ fun parseAllEvents(content: String): List { - val matchSequence = nip19regexEvents.findAll(content) val returningList = mutableListOf() - matchSequence.forEach { matcher -> - val type = matcher.groups[2]?.value // npub1 + forEachNip19Match(content, nip19regexEvents) { matcher -> + val type = matcher.groups[2]?.value // nevent1 val key = matcher.groups[3]?.value // bech32 val additionalChars = matcher.groups[4]?.value // additional chars if (type != null) { - val parsed = parseComponents(type, key, additionalChars)?.entity - - if (parsed != null) { - returningList.add(parsed) - } + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } } return returningList diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt new file mode 100644 index 0000000000..7e6035e23e --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanTest.kt @@ -0,0 +1,216 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip10Notes.content + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Behavioural coverage for the two content scans that run on every ingested note. + * + * Both scans jump between `#` positions with `indexOf` and apply their regex + * anchored there, so the cases below deliberately cover what decides a match: + * what precedes the `#`, where it sits in the string, and what terminates the tag. + * + * In `commonTest` on purpose — these are `commonMain` parsers and the scans use + * `matchAt`, so they must behave identically on JVM, Android, Apple and native. + */ +class ContentScanTest { + // ---------- findHashtags ---------- + + @Test + fun hashtagAtStartOfContent() { + assertEquals(listOf("bitcoin"), findHashtags("#bitcoin")) + } + + @Test + fun hashtagAfterEachKindOfWhitespace() { + assertEquals(listOf("a"), findHashtags("x #a")) + assertEquals(listOf("b"), findHashtags("x\n#b")) + assertEquals(listOf("c"), findHashtags("x\t#c")) + assertEquals(listOf("d"), findHashtags("x\r\n#d")) + } + + @Test + fun hashtagMustBePrecededByWhitespaceOrStart() { + // glued to a word: not a hashtag + assertEquals(emptyList(), findHashtags("word#nottag")) + assertEquals(emptyList(), findHashtags("a#b")) + // a non-breaking space is not \s, so it does not open a hashtag either + assertEquals(emptyList(), findHashtags("x\u00a0#nope")) + } + + @Test + fun hashWithNoTagBodyIsNotAHashtag() { + assertEquals(emptyList(), findHashtags("#")) + assertEquals(emptyList(), findHashtags(" #")) + assertEquals(emptyList(), findHashtags("text # more")) + // the second '#' is an excluded char, so it cannot open the tag body + assertEquals(emptyList(), findHashtags("##tag")) + } + + @Test + fun punctuationTerminatesTheTag() { + assertEquals(listOf("tag"), findHashtags("#tag,")) + assertEquals(listOf("tag"), findHashtags("#tag.")) + assertEquals(listOf("tag"), findHashtags("#tag!")) + assertEquals(listOf("tag"), findHashtags("#tag)")) + assertEquals(listOf("tag"), findHashtags("#tag;")) + } + + @Test + fun tagsKeepCharactersOutsideTheExcludedSet() { + assertEquals(listOf("caf\u00e9"), findHashtags("#caf\u00e9")) + assertEquals(listOf("123"), findHashtags("#123")) + assertEquals(listOf("a-b_c"), findHashtags("#a-b_c")) + assertEquals(listOf("\u65e5\u672c"), findHashtags("#\u65e5\u672c")) + } + + @Test + fun multipleHashtagsAndDeduplication() { + assertEquals(listOf("a", "b", "c"), findHashtags("#a #b #c").sorted()) + assertEquals(listOf("dup"), findHashtags("#dup #dup #dup")) + } + + @Test + fun hashtagAtVeryEndOfContent() { + assertEquals(listOf("end"), findHashtags("something #end")) + } + + @Test + fun blankContentReturnsEmpty() { + assertEquals(emptyList(), findHashtags("")) + assertEquals(emptyList(), findHashtags(" ")) + assertEquals(emptyList(), findHashtags("\n\t ")) + } + + @Test + fun callerSuppliedOutputSetIsReusedAndAccumulates() { + val shared = mutableSetOf() + findHashtags("#one", shared) + findHashtags("#two", shared) + assertEquals(listOf("one", "two"), shared.toList().sorted()) + } + + @Test + fun scanDoesNotStopAtTheFirstNonMatchingHash() { + // a glued '#' must not hide a later real hashtag + assertEquals(listOf("real"), findHashtags("word#nottag #real")) + } + + // ---------- IndexedTags ---------- + + private val tags = + arrayOf( + arrayOf("p", "pubkey0"), + arrayOf("e", "event1"), + arrayOf("a", "30023:author:slug"), + arrayOf("p", "pubkey3"), + arrayOf("t", "topic"), + arrayOf("p"), + ) + + @Test + fun indexTagResolvesPeople() { + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#[0]", tags)) + assertEquals(listOf("pubkey3"), findIndexTagsWithPeople("hi #[3]", tags)) + } + + @Test + fun indexTagResolvesEventsAndAddresses() { + assertEquals(setOf("event1"), findIndexTagsWithEventsOrAddresses("#[1]", tags)) + assertEquals(setOf("30023:author:slug"), findIndexTagsWithEventsOrAddresses("#[2]", tags)) + } + + @Test + fun indexTagIgnoresWrongTagKinds() { + // "t" is neither p, e nor a + assertEquals(emptyList(), findIndexTagsWithPeople("#[4]", tags)) + assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[4]", tags)) + // a "p" tag with no value must not blow up or emit + assertEquals(emptyList(), findIndexTagsWithPeople("#[5]", tags)) + } + + @Test + fun indexTagOutOfRangeIsIgnored() { + assertEquals(emptyList(), findIndexTagsWithPeople("#[99]", tags)) + assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[99]", tags)) + } + + @Test + fun indexTagWithHugeNumberDoesNotThrow() { + // does not fit in an Int — must be swallowed, not propagated + assertEquals(emptyList(), findIndexTagsWithPeople("#[99999999999999999999]", tags)) + } + + @Test + fun malformedIndexRefsAreIgnored() { + assertEquals(emptyList(), findIndexTagsWithPeople("#[abc]", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("#[]", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("#[0", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("[0]", tags)) + } + + @Test + fun indexRefMustBePrecededByWhitespaceOrStart() { + assertEquals(emptyList(), findIndexTagsWithPeople("word#[0]", tags)) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("word #[0]", tags)) + } + + @Test + fun multipleIndexRefsAndDeduplication() { + assertEquals(listOf("pubkey0", "pubkey3"), findIndexTagsWithPeople("#[0] #[3]", tags).sorted()) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#[0] #[0]", tags)) + assertEquals( + setOf("30023:author:slug", "event1"), + findIndexTagsWithEventsOrAddresses("#[1] #[2]", tags), + ) + } + + @Test + fun indexScanDoesNotStopAtTheFirstNonMatchingHash() { + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("word#[9] #[0]", tags)) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#hashtag #[0]", tags)) + } + + @Test + fun emptyContentAndEmptyTagArray() { + assertEquals(emptyList(), findIndexTagsWithPeople("", tags)) + assertEquals(emptyList(), findIndexTagsWithPeople("#[0]", arrayOf())) + assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[0]", arrayOf())) + } + + @Test + fun hashtagAndIndexRefsCoexistInOneNote() { + val content = "#intro see #[0] and #[1] about #nostr" + assertEquals(listOf("intro", "nostr"), findHashtags(content).sorted()) + assertEquals(listOf("pubkey0"), findIndexTagsWithPeople(content, tags)) + assertEquals(setOf("event1"), findIndexTagsWithEventsOrAddresses(content, tags)) + } + + @Test + fun longContentWithNoMarkersTerminates() { + val prose = "the quick brown fox jumps over the lazy dog ".repeat(2_000) + assertTrue(findHashtags(prose).isEmpty()) + assertTrue(findIndexTagsWithPeople(prose, tags).isEmpty()) + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt new file mode 100644 index 0000000000..27dd5d371d --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19DecodeTest.kt @@ -0,0 +1,167 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32 + +import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer +import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32 +import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress +import com.vitorpamplona.quartz.nip19Bech32.entities.NEmbed +import com.vitorpamplona.quartz.nip19Bech32.entities.NProfile +import com.vitorpamplona.quartz.nip19Bech32.entities.NRelay +import com.vitorpamplona.quartz.nip19Bech32.tlv.TlvBuilder +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertNotNull +import kotlin.test.assertTrue + +/** + * Coverage for the TLV-backed entities reached *through the content scan*, and for + * [Nip19Parser.tryParseAndClean]. + * + * `NIP19ParserTest` decodes these through `uriToRoute`. This asserts the same + * entities survive being found inside arbitrary text, which is the path every + * ingested note takes, and pins the relay hints and identifiers they carry. + * + * `nembed1` had no `commonTest` coverage at all (its only fixtures live in + * `androidDeviceTest`/`iosTest`), and `nrelay1` had no *positive* case anywhere — + * the codebase can parse it but has no encoder for it, so one is built here from + * TLV directly. + */ +class Nip19DecodeTest { + companion object { + const val PUBKEY = "460c25e682fda7832b52d1f22d3d22b3176d972f60dcdc3212ed8c92ef85065c" + const val NADDR = + "naddr1qqyxzmt9w358jum5qyt8wumn8ghj7un9d3shjtnwdaehgu3wvfskueqzypd7v3r24z33cydnk3fmlrd0exe5dlej3506zxs05q4puerp765mzqcyqqq8scsq6mk7u" + const val NEMBED = "nembed1r79ssq9446hkwqhl642ukmku8qg0c92pu7w3j0jyfte8tc7tvg85vmrys8x3sqgle5vjy7jpjswqhphl0kd6yf4sz0n3peyjq5rp3zkat4w6c6j3f7um0724jmfu5456xxgg2yxkn8dp23j64xsn9npcggzafyh2effyntqrqxzja8dp52kpcvc9zqxlj86e8mx05vevzxkeprjkfs4wmppxm3p96vj6yvu2mqgf5l4v99492r2qsggquxuv93uzx244652h2kkj8xseg9xkq0afpygknjtty9j4ju5v0nm9mezux9wyl6s5wr7lzce7cj397mnu0u04ha7aq3w7exelrhe3zs3l3urwa9sp36u80npllrs0hmsxqdn0fsuyav3nv0azjs5suzuurg2uymncjxez8p9xksc2j6gw992enjflgrdd7n5uq2xrpvfrd3rckw624ey0elvm6grr27tyzlf4vaswgm5vc3hdyczsl983g2j8e67r6z5zt30lat84ma4wclkwwxxrcflvdsuwd7346h7zqav4vdwe3gkt9lr87sfk4aqd2aey03tt4eyspldrqcmkx9pqe2pn63rv7grwwalr86akuldnvjm6m87wrw9sdwns8wq0rnsmj57vqwtc3g7hkwum3vl2dda78dwkycgfzw6qna3ufhpatcvq5a4hm4ehl45an8umwt0clf7rn77ctke475qglwu86hhfwhn7dkca4pkfpyc4y75rll6nvr5qc8nlhf8mk22celn5mecvyuzxd830drhdck9tcdpcafymk8wajwu2w8ha8gatggjfvq0a4jlf2sdamzj0ysqks9dk8me3q7a0qpmf6vykurkrcls4pug3u4pn4u26ezx3h8e482n07x2nsmu80dpufxqc0ttcyzhnppguxma4d8aumdawnlsyy7yzcuxl7lw5y9p4nv5h8fn6u8anpm2tsze3p6mgxy9j9uuqfxg2jvlmtjpakna5m4hln0msmw804hnun96h66fh62270yhhljnmmdl7jln07ll5vft7e870hemcld34a09n943ed6629fgtctsftma9q6tf4jfm2p0ukd2j2n2dpz53fqrkk4ctdcy2j5jar095g5jntf6u807ggkzauzt6uqkwk4tg5w7w55kskspc9663zx5dzzzfwpg3q546g2ve4kukr70n0a46eyce2crsqqq247ql5" + const val NCRYPTSEC = "ncryptsec1qgg9947rlpvqu76pj5ecreduf9jxhselq2nae2kghhvd5g7dgjtcxfqtd67p9m0w57lspw8gsq6yphnm8623nsl8xn9j4jdzz84zm3frztj3z7s35vpzmqf6ksu8r89qk5z2zxfmu5gv8th8wclt0h4p" + } + + // ---------- nprofile: pubkey + relay hints ---------- + + @Test + fun nprofileWithRelayHintsSurvivesTheContentScan() { + val relay = RelayUrlNormalizer.normalize("wss://vitor.nostr1.com/") + val encoded = NProfile.create(PUBKEY, listOf(relay)) + + val found = Nip19Parser.parseAll("please follow nostr:$encoded today") + assertEquals(1, found.size) + val profile = found[0] as NProfile + assertEquals(PUBKEY, profile.hex) + assertEquals(listOf(relay), profile.relay) + } + + @Test + fun nprofileWithoutRelayHintsStillDecodes() { + val encoded = NProfile.create(PUBKEY, emptyList()) + val found = Nip19Parser.parseAll(encoded) + assertEquals(1, found.size) + assertEquals(PUBKEY, (found[0] as NProfile).hex) + assertTrue((found[0] as NProfile).relay.isEmpty()) + } + + @Test + fun nprofileWithSeveralRelayHintsKeepsAllOfThem() { + val a = RelayUrlNormalizer.normalize("wss://relay.one/") + val b = RelayUrlNormalizer.normalize("wss://relay.two/") + val encoded = NProfile.create(PUBKEY, listOf(a, b)) + val profile = Nip19Parser.parseAll("x $encoded").single() as NProfile + assertEquals(listOf(a, b), profile.relay) + } + + // ---------- naddr: kind:author:dTag ---------- + + @Test + fun naddrSurvivesTheContentScan() { + val found = Nip19Parser.parseAll("read nostr:$NADDR now") + assertEquals(1, found.size) + assertEquals( + "30818:5be6446aa8a31c11b3b453bf8dafc9b346ff328d1fa11a0fa02a1e6461f6a9b1:amethyst", + (found[0] as NAddress).aTag(), + ) + } + + // ---------- nrelay: no encoder in the codebase, so build one from TLV ---------- + + @Test + fun nrelayDecodesFromTheContentScan() { + val url = "wss://relay.example.com/" + val encoded = + TlvBuilder() + .apply { addString(TlvTypes.SPECIAL.id, url) } + .build() + .let { Bech32.encodeBytes(hrp = "nrelay", it, Bech32.Encoding.Bech32) } + + val found = Nip19Parser.parseAll("join nostr:$encoded please") + assertEquals(1, found.size) + assertEquals(listOf(url), (found[0] as NRelay).relay) + } + + // ---------- nembed: gzipped event, first commonTest coverage ---------- + + @Test + fun nembedDecodesFromTheContentScan() { + // fixture mirrored from NIP19EmbedTests (androidDeviceTest/iosTest), which + // never ran on JVM — the decode goes through GZip + Event parsing + val found = Nip19Parser.parseAll("look at nostr:$NEMBED") + assertEquals(1, found.size) + val embed = found[0] as NEmbed + assertNotNull(embed.event) + } + + @Test + fun truncatedNembedIsRejectedNotThrown() { + val broken = NEMBED.take(NEMBED.length / 2) + assertEquals(emptyList(), Nip19Parser.parseAll(broken)) + } + + // ---------- tryParseAndClean ---------- + + @Test + fun tryParseAndCleanStripsSchemeAndTrailingCharacters() { + val npub = Nip19ScanTest.NPUB + assertEquals(npub, Nip19Parser.tryParseAndClean("nostr:$npub")) + assertEquals(npub, Nip19Parser.tryParseAndClean(npub)) + assertEquals(npub, Nip19Parser.tryParseAndClean("@$npub")) + } + + @Test + fun tryParseAndCleanReturnsNullForNonEntities() { + assertEquals(null, Nip19Parser.tryParseAndClean(null)) + assertEquals(null, Nip19Parser.tryParseAndClean("")) + assertEquals(null, Nip19Parser.tryParseAndClean("just some words")) + assertEquals(null, Nip19Parser.tryParseAndClean("npub1tooshort")) + } + + @Test + fun tryParseAndCleanNeverReturnsALiteralNullSuffix() { + // `type!! + key` would render "npub1null" if the key group were ever absent + listOf("nostr:${Nip19ScanTest.NPUB}", Nip19ScanTest.NOTE, "nostr:${Nip19ScanTest.NEVENT}") + .forEach { assertTrue(Nip19Parser.tryParseAndClean(it)?.endsWith("null") != true, "for $it") } + } + + @Test + fun tryParseAndCleanAcceptsNcryptsecWhichTheContentScanDoesNot() { + // nip19PlusNip46regex includes ncryptsec1; nip19regex (used by parseAll) does not + val ncryptsec = NCRYPTSEC + assertEquals(ncryptsec, Nip19Parser.tryParseAndClean(ncryptsec)) + assertEquals(emptyList(), Nip19Parser.parseAll(ncryptsec)) + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt new file mode 100644 index 0000000000..474057d778 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScanTest.kt @@ -0,0 +1,214 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32 + +import com.vitorpamplona.quartz.nip19Bech32.entities.NEvent +import com.vitorpamplona.quartz.nip19Bech32.entities.NNote +import com.vitorpamplona.quartz.nip19Bech32.entities.NPub +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Behavioural coverage for [Nip19Parser.parseAll] — the whole-content scan run on + * every ingested note. + * + * `NIP19ParserTest` covers [Nip19Parser.uriToRoute] (one entity, already isolated). + * This covers the other half: finding entities *inside* arbitrary text — where they + * may sit, what may precede them, and what must NOT be mistaken for one. + * + * In `commonTest` on purpose: [Nip19Parser] is `commonMain` and the scan uses + * `matchAt`/`regionMatches`, so it must behave identically on every target. + */ +class Nip19ScanTest { + companion object { + const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y" + const val NPUB_HEX = "bb3d6543d4a4f45f402b674ecb5cb2155ff12456fcdd2d66d9727660d3037365" + const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv" + const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy" + } + + @Test + fun findsBareEntity() { + val found = Nip19Parser.parseAll(NPUB) + assertEquals(1, found.size) + assertEquals(NPUB_HEX, (found[0] as NPub).hex) + } + + @Test + fun findsEntityWithNostrScheme() { + val found = Nip19Parser.parseAll("hello nostr:$NPUB world") + assertEquals(1, found.size) + assertEquals(NPUB_HEX, (found[0] as NPub).hex) + } + + @Test + fun findsEntityWithAtPrefix() { + val found = Nip19Parser.parseAll("cc @$NPUB thanks") + assertEquals(1, found.size) + assertEquals(NPUB_HEX, (found[0] as NPub).hex) + } + + @Test + fun findsEntityAtStartMiddleAndEndOfContent() { + assertEquals(1, Nip19Parser.parseAll("$NPUB trailing text").size) + assertEquals(1, Nip19Parser.parseAll("leading $NPUB trailing").size) + assertEquals(1, Nip19Parser.parseAll("leading text $NPUB").size) + } + + @Test + fun findsDifferentEntityTypes() { + assertTrue(Nip19Parser.parseAll("see $NOTE")[0] is NNote) + assertTrue(Nip19Parser.parseAll("see $NEVENT")[0] is NEvent) + } + + @Test + fun findsSeveralEntitiesInOneNote() { + val found = Nip19Parser.parseAll("$NPUB then $NOTE and nostr:$NEVENT") + assertEquals(3, found.size) + } + + @Test + fun entityGluedToAPrecedingWordIsStillFound() { + // the regex has no leading whitespace requirement + assertEquals(1, Nip19Parser.parseAll("x$NPUB").size) + } + + @Test + fun uppercaseEntityIsFound() { + // the regex is IGNORE_CASE; bech32 decode is case-insensitive too + assertEquals(1, Nip19Parser.parseAll("NOSTR:${NPUB.uppercase()}").size) + } + + @Test + fun invalidChecksumIsNotReturned() { + val broken = NPUB.dropLast(6) + "qqqqqq" + assertEquals(emptyList(), Nip19Parser.parseAll(broken)) + } + + @Test + fun tooShortNpubIsNotMatched() { + assertEquals(emptyList(), Nip19Parser.parseAll("npub1tooshort")) + assertEquals(emptyList(), Nip19Parser.parseAll("npub1")) + } + + @Test + fun proseWithoutEntitiesReturnsNothing() { + assertEquals(emptyList(), Nip19Parser.parseAll("no entities in this note at all")) + // words that start like a prefix but are not one + assertEquals(emptyList(), Nip19Parser.parseAll("nostrich nope never nan note nsec nprofile")) + } + + @Test + fun contentEndingInNDoesNotReadPastTheEnd() { + // the candidate scan dispatches on the character AFTER 'n' + assertEquals(emptyList(), Nip19Parser.parseAll("n")) + assertEquals(emptyList(), Nip19Parser.parseAll("N")) + assertEquals(emptyList(), Nip19Parser.parseAll("ends with n")) + assertEquals(emptyList(), Nip19Parser.parseAll("trailing N")) + listOf("np", "ne", "na", "nr", "ns", "no", "nn").forEach { + assertEquals(emptyList(), Nip19Parser.parseAll(it), "for '$it'") + assertEquals(emptyList(), Nip19Parser.parseAll("text $it"), "for 'text $it'") + } + } + + @Test + fun emptyContentReturnsNothing() { + assertEquals(emptyList(), Nip19Parser.parseAll("")) + assertEquals(emptyList(), Nip19Parser.parseAll(" ")) + } + + @Test + fun adjacentEntitiesWithNoSeparatorYieldOne() { + // the trailing ([\S]*) group is greedy, so it swallows the second entity — + // documented behaviour, pinned here so the scan rewrite cannot change it + assertEquals(1, Nip19Parser.parseAll(NPUB + NOTE).size) + // separated by a space, both are found + assertEquals(2, Nip19Parser.parseAll("$NPUB $NOTE").size) + } + + @Test + fun trailingPunctuationDoesNotBreakTheEntity() { + assertEquals(1, Nip19Parser.parseAll("thanks nostr:$NPUB!").size) + assertEquals(1, Nip19Parser.parseAll("(nostr:$NPUB)").size) + } + + @Test + fun entityInsideMultilineContent() { + assertEquals(1, Nip19Parser.parseAll("line one\nnostr:$NPUB\nline three").size) + } + + @Test + fun scanDoesNotStopAtTheFirstNonEntityCandidate() { + // "nostrich" and "nevermind" are 'n' candidates that fail; a real entity follows + assertEquals(1, Nip19Parser.parseAll("nostrich nevermind nap $NPUB").size) + } + + // ---------- parseAllEvents: same scan, event-ish entities only ---------- + + @Test + fun parseAllEventsFindsEventEntities() { + assertEquals(1, Nip19Parser.parseAllEvents("see nostr:$NEVENT").size) + assertEquals(1, Nip19Parser.parseAllEvents("see $NOTE").size) + } + + @Test + fun parseAllEventsIgnoresProfileEntities() { + // npub1/nsec1/nprofile1 are not in nip19regexEvents + assertEquals(emptyList(), Nip19Parser.parseAllEvents(NPUB)) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("hello nostr:$NPUB there")) + } + + @Test + fun parseAllEventsSkipsProfilesButStillFindsLaterEvents() { + // an npub is a candidate position that must not stop the scan + val found = Nip19Parser.parseAllEvents("$NPUB then nostr:$NEVENT") + assertEquals(1, found.size) + assertTrue(found[0] is NEvent) + } + + @Test + fun parseAllEventsHandlesEdgesLikeParseAll() { + assertEquals(emptyList(), Nip19Parser.parseAllEvents("")) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("n")) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("ends with n")) + assertEquals(emptyList(), Nip19Parser.parseAllEvents("no entities here")) + assertEquals(2, Nip19Parser.parseAllEvents("$NOTE $NEVENT").size) + } + + // ---------- uriToRoute: unchanged, pinned against the shared candidate scan ---------- + + @Test + fun uriToRouteStillResolvesEachForm() { + assertTrue(Nip19Parser.uriToRoute("nostr:$NPUB")?.entity is NPub) + assertTrue(Nip19Parser.uriToRoute(NPUB)?.entity is NPub) + assertTrue(Nip19Parser.uriToRoute("nostr:$NEVENT")?.entity is NEvent) + assertEquals(null, Nip19Parser.uriToRoute(null)) + assertEquals(null, Nip19Parser.uriToRoute("not an entity")) + assertEquals(null, Nip19Parser.uriToRoute("")) + } + + @Test + fun longProseWithNoEntitiesTerminates() { + val prose = "the nimble nocturnal nightingale never naps near noon ".repeat(2_000) + assertEquals(emptyList(), Nip19Parser.parseAll(prose)) + } +}