Merge pull request #3915 from vitorpamplona/perf/content-regex-scans

perf: scan note content by literal instead of driving the regex engine
This commit is contained in:
Vitor Pamplona
2026-08-13 16:09:11 -04:00
committed by GitHub
8 changed files with 1268 additions and 27 deletions
@@ -416,8 +416,12 @@ class RichTextParser {
tags: ImmutableListOfLists<String>,
): Segment {
// First #[n]
// [tagIndex] requires the literal "#[", so a word without it can never match.
// Every plain "#hashtag" reaches this function, and the regex scan is far more
// expensive than the substring check that rules it out — so gate on the literal.
// Uses contains(), not startsWith(), because find() also matches "#[n]" mid-word.
try {
val matcher = tagIndex.find(word)
val matcher = if (word.contains("#[")) tagIndex.find(word) else null
if (matcher != null) {
val index = matcher.groups[1]?.value?.toInt()
val suffix = matcher.groups[2]?.value
@@ -0,0 +1,517 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.amethyst.commons.prodbench
import com.vitorpamplona.amethyst.commons.model.toImmutableListOfLists
import com.vitorpamplona.amethyst.commons.richtext.RichTextParser
import com.vitorpamplona.amethyst.commons.richtext.RichTextParser.Companion.tagIndex
import com.vitorpamplona.quartz.nip10Notes.content.findHashtags
import com.vitorpamplona.quartz.nip10Notes.content.findIndexTagsWithEventsOrAddresses
import com.vitorpamplona.quartz.nip10Notes.content.findIndexTagsWithPeople
import com.vitorpamplona.quartz.nip10Notes.content.findNostrUris
import com.vitorpamplona.quartz.nip10Notes.content.hashtagSearch
import com.vitorpamplona.quartz.nip10Notes.content.tagSearch
import com.vitorpamplona.quartz.nip19Bech32.Nip19Parser
import kotlin.test.Test
/**
* Measures the regex scans Amethyst runs over note **content**.
*
* Why: an on-device heap dump (SM-T220, Dr. Edo's account) found 2,541 of 4,573
* live `java.util.regex.Matcher` instances running [Nip19Parser.nip19regex],
* reached from `BaseNoteEvent.citedNIP19()` during ingest. Kotlin's `findAll`
* allocates a **new Matcher per match**, each retaining the whole input
* (`jvmMain/kotlin/text/regex/Regex.kt`: `matcher.pattern().matcher(input)`),
* so a long article with N mentions builds N matchers over the full text.
*
* Corpus sizes come from that same dump's measured distribution of the strings
* these matchers held: median 529 B, p90/max 767 KB (long-form articles — the
* v1.13.0 release notes at 68 KB, an essay at 50 KB).
*
* Deterministic and offline. Prints ns/op and MB/s; no assertions on wall time
* (CI machines vary) beyond a sanity check that the scans return results.
*/
class RegexContentBenchmark {
companion object {
private const val NPUB = "npub180cvv07tjdrrgpa0j7j7tmnyl2yr6yr7l8j4s3evf6u64th6gkwsyjh6w6"
private const val NEVENT = "nevent1qqstna2yrezu5wghjvswqqculvvwxsrcvu7uc0f78gan4xqhvz49d9spr3mhxue69uhkummnw3ez6un9d3shjtn4de6x2argwghx6egpr4mhxue69uhkummnw3ez6ur4vgh8wetvd3hhyer9wghxuet5nxnepm"
/** A note with exactly [mentions] nostr: URIs, padded with prose to ~[targetBytes]. */
fun note(
targetBytes: Int,
mentions: Int,
hashtags: Boolean = true,
): String {
val filler =
if (hashtags) {
"A few months ago a nostrich was switching from iOS to Android and asked for " +
"suggestions for #Nostr apps to try out. Here is what came back, with notes. "
} else {
"A few months ago a nostrich was switching from iOS to Android and asked for " +
"suggestions for great apps to try out. Here is what came back, with notes. "
}
val sb = StringBuilder(targetBytes + 4096)
var placed = 0
val stride = if (mentions > 0) targetBytes / (mentions + 1) else Int.MAX_VALUE
while (sb.length < targetBytes) {
sb.append(filler)
if (placed < mentions && sb.length >= (placed + 1).toLong() * stride) {
sb.append("nostr:").append(if (placed % 2 == 0) NPUB else NEVENT).append(' ')
placed++
}
}
while (placed < mentions) {
sb.append("nostr:").append(if (placed % 2 == 0) NPUB else NEVENT).append(' ')
placed++
}
return sb.toString()
}
/**
* REFERENCE: the pre-optimization `findHashtags`, verbatim. Production is
* compared against this so the guard can never become a tautology.
*/
fun referenceFindHashtags(content: String): List<String> {
if (content.isBlank()) return emptyList()
val output = mutableSetOf<String>()
hashtagSearch.findAll(content).forEach {
try {
val tag = it.groups[1]?.value
if (tag != null && tag.isNotBlank()) output.add(tag)
} catch (e: Exception) {
}
}
return output.toList()
}
/** REFERENCE: pre-optimization nip19 scan (findAll), resolved via the public uriToRoute. */
fun referenceNip19(content: String): List<String> =
Nip19Parser.nip19regex
.findAll(content)
.mapNotNull { m ->
val type = m.groups[3]?.value ?: m.groups[5]?.value
val key = m.groups[4]?.value ?: m.groups[6]?.value
if (type != null && key != null) {
Nip19Parser.uriToRoute(type + key)?.entity?.let { type + key }
} else {
null
}
}.toList()
/** A tag array with [n] p/e/a entries, for the #[index] scans. */
fun tagArray(n: Int): Array<Array<String>> =
Array(n) { i ->
when (i % 3) {
0 -> arrayOf("p", "%064x".format(i))
1 -> arrayOf("e", "%064x".format(i + 1000))
else -> arrayOf("a", "30023:%064x:slug$i".format(i))
}
}
/** A note with [refs] legacy #[n] references, padded to ~[targetBytes]. */
fun indexNote(
targetBytes: Int,
refs: Int,
tagCount: Int,
): String {
val filler = "Legacy index refs used to link people and events inline in the content. "
val sb = StringBuilder(targetBytes + 4096)
var placed = 0
val stride = if (refs > 0) targetBytes / (refs + 1) else Int.MAX_VALUE
while (sb.length < targetBytes) {
sb.append(filler)
if (placed < refs && sb.length >= (placed + 1).toLong() * stride) {
sb.append("#[").append(placed % tagCount).append("] ")
placed++
}
}
while (placed < refs) {
sb.append("#[").append(placed % tagCount).append("] ")
placed++
}
return sb.toString()
}
/** REFERENCE: pre-optimization findIndexTagsWithPeople, verbatim. */
fun referenceIndexPeople(
content: String,
tags: Array<Array<String>>,
): List<String> {
val output = mutableSetOf<String>()
tagSearch.findAll(content).forEach { index ->
try {
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == "p") output.add(tag[1])
} catch (e: Exception) {
}
}
return output.toList()
}
/** REFERENCE: pre-optimization findIndexTagsWithEventsOrAddresses, verbatim. */
fun referenceIndexEvents(
content: String,
tags: Array<Array<String>>,
): Set<String> {
val output = mutableSetOf<String>()
tagSearch.findAll(content).forEach { index ->
try {
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == "e") output.add(tag[1])
if (tag != null && tag.size > 1 && tag[0] == "a") output.add(tag[1])
} catch (e: Exception) {
}
}
return output
}
fun bench(
label: String,
input: String,
reps: Int,
op: (String) -> Int,
) {
repeat(maxOf(reps / 4, 2)) { op(input) } // warmup
val t0 = System.nanoTime()
var sink = 0
repeat(reps) { sink += op(input) }
val ns = (System.nanoTime() - t0) / reps
val mbps = input.length.toDouble() / ns * 1000.0 // bytes/ns -> MB/s
println(
String.format(
"%-34s %9d B %9d ns/op %8.1f MB/s (hits=%d)",
label,
input.length,
ns,
mbps,
sink / reps,
),
)
}
}
@Test
fun contentScans() {
// (bytes, mentions, reps) — mirrors the measured distribution
val corpus =
listOf(
Triple(120, 1, 20_000), // short note
Triple(529, 2, 20_000), // MEDIAN of what matchers held
Triple(4_000, 5, 5_000), // long note
Triple(68_000, 40, 200), // v1.13.0 release notes
Triple(767_000, 120, 20), // observed MAX
)
// Guard the corpus: a benchmark that silently matches nothing measures nothing.
corpus.forEach { (n, m, _) ->
val found = Nip19Parser.parseAll(note(n, m)).size
check(found >= m) { "corpus broken: ${n}B/$m mentions parsed only $found entities" }
}
println("\n=== nip19 findNostrUris — NO matches (the common case) ===")
corpus.forEach { (n, _, r) ->
bench("nip19 0 mentions", note(n, 0), r) { findNostrUris(it).size }
}
println("\n=== nip19 findNostrUris — WITH matches (Matcher per match) ===")
corpus.forEach { (n, m, r) ->
bench("nip19 m=$m", note(n, m), r) { findNostrUris(it).size }
}
println("\n=== findHashtags (ContentHashTags.hashtagSearch) ===")
corpus.forEach { (n, m, r) ->
bench("hashtags m=$m", note(n, m), r) { findHashtags(it).size }
}
println("\n=== IndexedTags findIndexTagsWithPeople ===")
val tags = tagArray(30)
corpus.forEach { (n, _, r) ->
bench("idxTags none", indexNote(n, 0, 30), r) { findIndexTagsWithPeople(it, tags).size }
}
corpus.forEach { (n, m, r) ->
bench("idxTags refs=$m", indexNote(n, m, 30), r) { findIndexTagsWithPeople(it, tags).size }
}
println("\n=== parseAllEvents (composer: findNostrEventUris) ===")
corpus.forEach { (n, _, r) ->
bench("parseAllEvents 0 mentions", note(n, 0), r) { Nip19Parser.parseAllEvents(it).size }
}
corpus.forEach { (n, m, r) ->
bench("parseAllEvents m=$m", note(n, m), r) { Nip19Parser.parseAllEvents(it).size }
}
println("\n=== uriToRoute (short strings: one URI / id per call) ===")
listOf(
"nostr:$NPUB" to 200_000,
NPUB to 200_000,
"nostr:$NEVENT" to 200_000,
"30023:abc:slug" to 200_000,
"not an entity at all" to 200_000,
).forEach { (uri, reps) ->
bench("uriToRoute ${uri.take(18)}", uri, reps) { if (Nip19Parser.uriToRoute(it) != null) 1 else 0 }
}
println("\n=== tryParseAndClean (short strings, like uriToRoute) ===")
listOf(
"nostr:$NPUB" to 200_000,
NPUB to 200_000,
"not an entity at all" to 200_000,
).forEach { (uri, reps) ->
bench("tryParseAndClean ${uri.take(14)}", uri, reps) { if (Nip19Parser.tryParseAndClean(it) != null) 1 else 0 }
}
println("\n=== RENDER PATH: RichTextParser.parseText (per note, per render) ===")
val rtTags = tagArray(30).toImmutableListOfLists()
val rtCorpus =
listOf(
Triple(120, 1, 3_000),
Triple(529, 2, 3_000),
Triple(4_000, 5, 500),
Triple(68_000, 40, 20),
)
rtCorpus.forEach { (n, m, r) ->
bench("parseText hashtags m=$m", note(n, m), r) {
RichTextParser().parseText(it, rtTags, null).paragraphs.size
}
}
rtCorpus.forEach { (n, _, r) ->
bench("parseText no '#'", note(n, 0, hashtags = false), r) {
RichTextParser().parseText(it, rtTags, null).paragraphs.size
}
}
rtCorpus.forEach { (n, m, r) ->
bench("parseText #[n] refs", indexNote(n, m, 30), r) {
RichTextParser().parseText(it, rtTags, null).paragraphs.size
}
}
}
/**
* Fuzz `findHashtags` against a verbatim copy of the implementation it replaced.
*
* The explicit behavioural cases live in quartz's `ContentScanTest` (commonTest,
* so they run on every target). What is kept here is the randomised half: text
* peppered with `#` in awkward positions, which is where an anchored scan can
* disagree with the original `findAll`.
*/
@Test
fun hashtagsMatchReferenceUnderFuzz() {
val rnd = kotlin.random.Random(20260813)
val alphabet = " \n\t#abcXYZ.,!@\u00a0-_0123"
repeat(3_000) {
val c = (1..rnd.nextInt(0, 80)).map { alphabet[rnd.nextInt(alphabet.length)] }.joinToString("")
val ref = referenceFindHashtags(c).sorted()
val prod = findHashtags(c).sorted()
check(ref == prod) { "fuzz mismatch on ${c.replace("\n", "\\n")}: reference=$ref production=$prod" }
}
}
/**
* Isolated A/B for the `#[` gate, interleaved with medians.
*
* parseText is allocation-heavy and its whole-parse timings swing ±50% run to
* run (the unchanged no-'#' control arm moved that much), which is far larger
* than this effect — so measure the gated call on its own instead.
*/
@Test
fun hashGateIsolatedAb() {
// realistic mix: mostly plain hashtags, a few legacy #[n] refs
val words =
buildList {
repeat(90) { add("#hashtag$it") }
repeat(10) { add("#[$it]") }
}
fun ungated(): Int {
var n = 0
words.forEach { if (tagIndex.find(it) != null) n++ }
return n
}
fun gated(): Int {
var n = 0
words.forEach { if (it.contains("#[") && tagIndex.find(it) != null) n++ }
return n
}
check(ungated() == gated()) { "gate changed the result" }
repeat(2_000) {
ungated()
gated()
} // warmup both
val a = mutableListOf<Long>()
val b = mutableListOf<Long>()
repeat(21) {
var t = System.nanoTime()
repeat(2_000) { ungated() }
a.add(System.nanoTime() - t)
t = System.nanoTime()
repeat(2_000) { gated() }
b.add(System.nanoTime() - t)
}
a.sort()
b.sort()
val am = a[a.size / 2] / 2_000
val bm = b[b.size / 2] / 2_000
println(
"\nHASH GATE (100 words: 90 plain hashtags + 10 #[n]) median of 21:" +
"\n ungated tagIndex.find : $am ns/word-set" +
"\n gated with contains : $bm ns/word-set" +
"\n speedup : ${String.format("%.2f", am.toDouble() / bm)}x",
)
}
/**
* Segment-level guard for the RichTextParser '#' path.
*
* Expectations are HARDCODED from the behaviour before the `#[` gate was added,
* so this pins the parser against the code it replaced rather than against itself.
*/
@Test
fun richTextHashSegmentsUnchanged() {
val tags = tagArray(30).toImmutableListOfLists()
val expected =
listOf(
"#Nostr" to "HashTagSegment|#Nostr",
"#[0]" to "HashIndexUserSegment|#[0]",
"#[1]suffix" to "HashIndexEventSegment|#[1]suffix",
"#[999]" to "RegularTextSegment|#[999]",
"#[abc]" to "RegularTextSegment|#[abc]",
"#[]" to "RegularTextSegment|#[]",
"#" to "RegularTextSegment|#",
"##tag" to "HashTagSegment|##tag",
"#tag," to "HashTagSegment|#tag,",
"#tag." to "HashTagSegment|#tag.",
"#a#b" to "HashTagSegment|#a#b",
"#[2]#tail" to "HashIndexEventSegment|#[2]#tail",
"#\u00e9t\u00e9" to "HashTagSegment|#\u00e9t\u00e9",
"#123" to "HashTagSegment|#123",
"#[0]!" to "HashIndexUserSegment|#[0]!",
"plain" to "RegularTextSegment|plain",
"#tag)" to "HashTagSegment|#tag)",
"#-dash" to "HashTagSegment|#-dash",
"#_under" to "HashTagSegment|#_under",
// '#[' appearing mid-word must still reach the index parser
"#a#[0]" to "HashIndexUserSegment|#a#[0]",
)
expected.forEach { (word, want) ->
val got =
RichTextParser()
.parseText(word, tags, null)
.paragraphs
.flatMap { p -> p.words.map { it::class.simpleName + "|" + it.segmentText } }
.joinToString(";")
check(got == want) { "segment mismatch for '$word': want=$want got=$got" }
}
}
/** Both IndexedTags scans must equal the findAll-based references. */
@Test
fun indexedTagsMatchReference() {
val tags = tagArray(30)
val cases =
listOf(
"#[0] #[1] #[2]",
"no refs here",
"#[0] at start",
"mid #[5] ref",
"nospace#[3] should not match",
"\n#[7] after newline",
"#[999] out of range",
"#[abc] not numeric",
"#[] empty",
"",
" ",
indexNote(2_000, 0, 30),
indexNote(4_000, 6, 30),
indexNote(20_000, 20, 30),
)
cases.forEach { c ->
check(referenceIndexPeople(c, tags).sorted() == findIndexTagsWithPeople(c, tags).sorted()) {
"people mismatch on ${c.take(40).replace("\n", "\\n")}"
}
check(referenceIndexEvents(c, tags).sorted() == findIndexTagsWithEventsOrAddresses(c, tags).sorted()) {
"events mismatch on ${c.take(40).replace("\n", "\\n")}"
}
}
val rnd = kotlin.random.Random(20260813)
val alphabet = " \n#[]0123456789abc\u00a0"
repeat(3_000) {
val c = (1..rnd.nextInt(0, 60)).map { alphabet[rnd.nextInt(alphabet.length)] }.joinToString("")
check(referenceIndexPeople(c, tags).sorted() == findIndexTagsWithPeople(c, tags).sorted()) {
"fuzz people mismatch on ${c.replace("\n", "\\n")}"
}
check(referenceIndexEvents(c, tags).sorted() == findIndexTagsWithEventsOrAddresses(c, tags).sorted()) {
"fuzz events mismatch on ${c.replace("\n", "\\n")}"
}
}
}
/** Production parseAll must equal the findAll-based reference scan. */
@Test
fun nip19MatchesReferenceScan() {
val cases =
listOf(
"nostr:$NPUB",
"@$NPUB",
NPUB,
"text nostr:$NEVENT tail",
"two $NPUB and nostr:$NEVENT here",
"none at all",
"npub1tooshort",
"",
note(200, 1),
note(4_000, 5),
note(20_000, 12),
note(4_000, 0),
// every prefix standalone
"npub1",
"nsec1",
"note1",
"nevent1",
"naddr1",
"nprofile1",
"nrelay1",
"nembed1",
// 'n' as the LAST char: the second-char dispatch must not read past the end
"n",
"N",
"ends with n",
"trailing N",
"nn",
"np",
"ne",
"na",
"nr",
"ns",
"no",
// case + adjacency
"NOSTR:" + NPUB.uppercase(),
"x" + NPUB,
NPUB + NEVENT,
NPUB + " " + NEVENT,
)
cases.forEach { c ->
val ref = referenceNip19(c)
val prod = Nip19Parser.parseAll(c).size
check(ref.size == prod) { "nip19 mismatch on ${c.take(50)}: reference=${ref.size} production=$prod" }
}
}
}
@@ -22,21 +22,45 @@ package com.vitorpamplona.quartz.nip10Notes.content
val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)")
/**
* Collects the hashtags in [content].
*
* [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match
* starts either at position 0 or at a whitespace. That lets the scan jump between
* `#` occurrences with `indexOf` — an intrinsified char search — and apply the
* regex **anchored** at each, instead of letting `findAll` drive the regex engine
* from every position in the string.
*
* Measured on the production content distribution (median 529 B, tail to 767 KB):
* ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the
* content has no `#` at all (up to 300x). Equivalence with the previous `findAll`
* implementation is guarded by `RegexContentBenchmark` in `commons`.
*/
fun findHashtags(
content: String,
output: MutableSet<String> = mutableSetOf(),
): List<String> {
if (content.isBlank()) return emptyList()
val matcher = hashtagSearch.findAll(content)
matcher.forEach {
try {
val tag = it.groups[1]?.value
if (tag != null && tag.isNotBlank()) {
output.add(tag)
var h = content.indexOf('#')
while (h >= 0) {
if (h == 0 || content[h - 1].isWhitespace()) {
val match =
try {
hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1)
} catch (e: Exception) {
null
}
if (match != null) {
val tag = match.groups[1]?.value
if (tag != null && tag.isNotBlank()) {
output.add(tag)
}
h = content.indexOf('#', match.range.last + 1)
continue
}
} catch (e: Exception) {
}
h = content.indexOf('#', h + 1)
}
return output.toList()
}
@@ -28,6 +28,43 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray
val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]")
/**
* Walks every `#[n]` reference in [content].
*
* [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match
* starts at position 0 or at a whitespace. That lets the scan jump between `#`
* occurrences with `indexOf` — an intrinsified char search — and apply the regex
* **anchored** at each, instead of letting `findAll` drive the regex engine from
* every position in the string.
*
* Measured on the production content distribution (median 529 B, tail to 767 KB):
* ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense
* text. Equivalence with the previous `findAll` implementation (both callers) is
* guarded by `RegexContentBenchmark` in `commons`.
*/
private inline fun forEachIndexTag(
content: String,
action: (MatchResult) -> Unit,
) {
var h = content.indexOf('#')
while (h >= 0) {
if (h == 0 || content[h - 1].isWhitespace()) {
val match =
try {
tagSearch.matchAt(content, if (h == 0) 0 else h - 1)
} catch (e: Exception) {
null
}
if (match != null) {
action(match)
h = content.indexOf('#', match.range.last + 1)
continue
}
}
h = content.indexOf('#', h + 1)
}
}
/**
* Returns the old-style [1] tag that pionts to an index in the tag array
*/
@@ -36,8 +73,7 @@ fun findIndexTagsWithPeople(
tags: TagArray,
output: MutableSet<String> = mutableSetOf<String>(),
): List<String> {
val matcher = tagSearch.findAll(content)
matcher.forEach { index ->
forEachIndexTag(content) { index ->
try {
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == "p") {
@@ -58,8 +94,7 @@ fun findIndexTagsWithEventsOrAddresses(
tags: TagArray,
output: MutableSet<String> = mutableSetOf<String>(),
): Set<String> {
val matcher = tagSearch.findAll(content)
matcher.forEach { index ->
forEachIndexTag(content) { index ->
try {
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == "e") {
@@ -133,39 +133,103 @@ object Nip19Parser {
fun hasAny(content: String): Boolean = nip19regex.matches(content)
/**
* True when one of the NIP-19 entity prefixes starts at [i].
*
* Every prefix begins with `n`, and English prose is ~7% `n`, so testing all
* eight prefixes at each `n` is the bulk of the scan's cost. Dispatching on the
* second character first narrows it to at most two candidates — most `n`s are
* rejected by a single char compare.
*/
private fun isCandidateAt(
content: String,
i: Int,
): Boolean {
val c = content[i]
if (c != 'n' && c != 'N') return false
if (i + 1 >= content.length) return false
return when (content[i + 1]) {
'p', 'P' ->
content.regionMatches(i, "npub1", 0, 5, ignoreCase = true) ||
content.regionMatches(i, "nprofile1", 0, 9, ignoreCase = true)
's', 'S' -> content.regionMatches(i, "nsec1", 0, 5, ignoreCase = true)
'o', 'O' -> content.regionMatches(i, "note1", 0, 5, ignoreCase = true)
'e', 'E' ->
content.regionMatches(i, "nevent1", 0, 7, ignoreCase = true) ||
content.regionMatches(i, "nembed1", 0, 7, ignoreCase = true)
'a', 'A' -> content.regionMatches(i, "naddr1", 0, 6, ignoreCase = true)
'r', 'R' -> content.regionMatches(i, "nrelay1", 0, 7, ignoreCase = true)
'c', 'C' -> content.regionMatches(i, "ncryptsec1", 0, 10, ignoreCase = true)
else -> false
}
}
/**
* Applies [regex] anchored at every NIP-19 candidate position in [content].
*
* [isCandidateAt] covers the union of the prefixes across the three NIP-19
* regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does
* not accept — still far cheaper than `findAll` restarting the engine at
* every position in the string.
*/
private inline fun forEachNip19Match(
content: String,
regex: Regex,
action: (MatchResult) -> Unit,
) {
var i = 0
val len = content.length
while (i < len) {
if (isCandidateAt(content, i)) {
val match = regex.matchAt(content, i)
if (match != null) {
action(match)
i = match.range.last + 1
continue
}
}
i++
}
}
/**
* Scans [content] for NIP-19 entities.
*
* Walks to each candidate position with cheap char compares and applies
* [nip19regex] **anchored** there, instead of letting `findAll` drive the
* regex engine from every position in the string. `(nostr:)?@?` are optional,
* so anchoring at the entity's own `n` matches the same entities and captures
* the same type/key/trailing groups this function reads.
*
* Measured 9–23x faster across the production content distribution (median
* 529 B, tail to 767 KB): ~19 MB/s -> 170–445 MB/s. Equivalence and speed are
* guarded by `RegexContentBenchmark` in `commons`. Motivation: on an SM-T220
* heap dump, 2,541 of 4,573 live Matchers were running this regex.
*/
fun parseAll(content: String): List<Entity> {
val matchSequence = nip19regex.findAll(content)
val returningList = mutableListOf<Entity>()
matchSequence.forEach { matcher ->
forEachNip19Match(content, nip19regex) { matcher ->
val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1
val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32
val additionalChars = matcher.groups[7]?.value // additional chars
if (type != null) {
val parsed = parseComponents(type, key, additionalChars)?.entity
if (parsed != null) {
returningList.add(parsed)
}
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
}
}
return returningList
}
/** Same scan as [parseAll], restricted to the event-ish entities. */
fun parseAllEvents(content: String): List<Entity> {
val matchSequence = nip19regexEvents.findAll(content)
val returningList = mutableListOf<Entity>()
matchSequence.forEach { matcher ->
val type = matcher.groups[2]?.value // npub1
forEachNip19Match(content, nip19regexEvents) { matcher ->
val type = matcher.groups[2]?.value // nevent1
val key = matcher.groups[3]?.value // bech32
val additionalChars = matcher.groups[4]?.value // additional chars
if (type != null) {
val parsed = parseComponents(type, key, additionalChars)?.entity
if (parsed != null) {
returningList.add(parsed)
}
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
}
}
return returningList
@@ -0,0 +1,216 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip10Notes.content
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertTrue
/**
* Behavioural coverage for the two content scans that run on every ingested note.
*
* Both scans jump between `#` positions with `indexOf` and apply their regex
* anchored there, so the cases below deliberately cover what decides a match:
* what precedes the `#`, where it sits in the string, and what terminates the tag.
*
* In `commonTest` on purpose — these are `commonMain` parsers and the scans use
* `matchAt`, so they must behave identically on JVM, Android, Apple and native.
*/
class ContentScanTest {
// ---------- findHashtags ----------
@Test
fun hashtagAtStartOfContent() {
assertEquals(listOf("bitcoin"), findHashtags("#bitcoin"))
}
@Test
fun hashtagAfterEachKindOfWhitespace() {
assertEquals(listOf("a"), findHashtags("x #a"))
assertEquals(listOf("b"), findHashtags("x\n#b"))
assertEquals(listOf("c"), findHashtags("x\t#c"))
assertEquals(listOf("d"), findHashtags("x\r\n#d"))
}
@Test
fun hashtagMustBePrecededByWhitespaceOrStart() {
// glued to a word: not a hashtag
assertEquals(emptyList(), findHashtags("word#nottag"))
assertEquals(emptyList(), findHashtags("a#b"))
// a non-breaking space is not \s, so it does not open a hashtag either
assertEquals(emptyList(), findHashtags("x\u00a0#nope"))
}
@Test
fun hashWithNoTagBodyIsNotAHashtag() {
assertEquals(emptyList(), findHashtags("#"))
assertEquals(emptyList(), findHashtags(" #"))
assertEquals(emptyList(), findHashtags("text # more"))
// the second '#' is an excluded char, so it cannot open the tag body
assertEquals(emptyList(), findHashtags("##tag"))
}
@Test
fun punctuationTerminatesTheTag() {
assertEquals(listOf("tag"), findHashtags("#tag,"))
assertEquals(listOf("tag"), findHashtags("#tag."))
assertEquals(listOf("tag"), findHashtags("#tag!"))
assertEquals(listOf("tag"), findHashtags("#tag)"))
assertEquals(listOf("tag"), findHashtags("#tag;"))
}
@Test
fun tagsKeepCharactersOutsideTheExcludedSet() {
assertEquals(listOf("caf\u00e9"), findHashtags("#caf\u00e9"))
assertEquals(listOf("123"), findHashtags("#123"))
assertEquals(listOf("a-b_c"), findHashtags("#a-b_c"))
assertEquals(listOf("\u65e5\u672c"), findHashtags("#\u65e5\u672c"))
}
@Test
fun multipleHashtagsAndDeduplication() {
assertEquals(listOf("a", "b", "c"), findHashtags("#a #b #c").sorted())
assertEquals(listOf("dup"), findHashtags("#dup #dup #dup"))
}
@Test
fun hashtagAtVeryEndOfContent() {
assertEquals(listOf("end"), findHashtags("something #end"))
}
@Test
fun blankContentReturnsEmpty() {
assertEquals(emptyList(), findHashtags(""))
assertEquals(emptyList(), findHashtags(" "))
assertEquals(emptyList(), findHashtags("\n\t "))
}
@Test
fun callerSuppliedOutputSetIsReusedAndAccumulates() {
val shared = mutableSetOf<String>()
findHashtags("#one", shared)
findHashtags("#two", shared)
assertEquals(listOf("one", "two"), shared.toList().sorted())
}
@Test
fun scanDoesNotStopAtTheFirstNonMatchingHash() {
// a glued '#' must not hide a later real hashtag
assertEquals(listOf("real"), findHashtags("word#nottag #real"))
}
// ---------- IndexedTags ----------
private val tags =
arrayOf(
arrayOf("p", "pubkey0"),
arrayOf("e", "event1"),
arrayOf("a", "30023:author:slug"),
arrayOf("p", "pubkey3"),
arrayOf("t", "topic"),
arrayOf("p"),
)
@Test
fun indexTagResolvesPeople() {
assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#[0]", tags))
assertEquals(listOf("pubkey3"), findIndexTagsWithPeople("hi #[3]", tags))
}
@Test
fun indexTagResolvesEventsAndAddresses() {
assertEquals(setOf("event1"), findIndexTagsWithEventsOrAddresses("#[1]", tags))
assertEquals(setOf("30023:author:slug"), findIndexTagsWithEventsOrAddresses("#[2]", tags))
}
@Test
fun indexTagIgnoresWrongTagKinds() {
// "t" is neither p, e nor a
assertEquals(emptyList(), findIndexTagsWithPeople("#[4]", tags))
assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[4]", tags))
// a "p" tag with no value must not blow up or emit
assertEquals(emptyList(), findIndexTagsWithPeople("#[5]", tags))
}
@Test
fun indexTagOutOfRangeIsIgnored() {
assertEquals(emptyList(), findIndexTagsWithPeople("#[99]", tags))
assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[99]", tags))
}
@Test
fun indexTagWithHugeNumberDoesNotThrow() {
// does not fit in an Int — must be swallowed, not propagated
assertEquals(emptyList(), findIndexTagsWithPeople("#[99999999999999999999]", tags))
}
@Test
fun malformedIndexRefsAreIgnored() {
assertEquals(emptyList(), findIndexTagsWithPeople("#[abc]", tags))
assertEquals(emptyList(), findIndexTagsWithPeople("#[]", tags))
assertEquals(emptyList(), findIndexTagsWithPeople("#[0", tags))
assertEquals(emptyList(), findIndexTagsWithPeople("[0]", tags))
}
@Test
fun indexRefMustBePrecededByWhitespaceOrStart() {
assertEquals(emptyList(), findIndexTagsWithPeople("word#[0]", tags))
assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("word #[0]", tags))
}
@Test
fun multipleIndexRefsAndDeduplication() {
assertEquals(listOf("pubkey0", "pubkey3"), findIndexTagsWithPeople("#[0] #[3]", tags).sorted())
assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#[0] #[0]", tags))
assertEquals(
setOf("30023:author:slug", "event1"),
findIndexTagsWithEventsOrAddresses("#[1] #[2]", tags),
)
}
@Test
fun indexScanDoesNotStopAtTheFirstNonMatchingHash() {
assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("word#[9] #[0]", tags))
assertEquals(listOf("pubkey0"), findIndexTagsWithPeople("#hashtag #[0]", tags))
}
@Test
fun emptyContentAndEmptyTagArray() {
assertEquals(emptyList(), findIndexTagsWithPeople("", tags))
assertEquals(emptyList(), findIndexTagsWithPeople("#[0]", arrayOf()))
assertEquals(emptySet(), findIndexTagsWithEventsOrAddresses("#[0]", arrayOf()))
}
@Test
fun hashtagAndIndexRefsCoexistInOneNote() {
val content = "#intro see #[0] and #[1] about #nostr"
assertEquals(listOf("intro", "nostr"), findHashtags(content).sorted())
assertEquals(listOf("pubkey0"), findIndexTagsWithPeople(content, tags))
assertEquals(setOf("event1"), findIndexTagsWithEventsOrAddresses(content, tags))
}
@Test
fun longContentWithNoMarkersTerminates() {
val prose = "the quick brown fox jumps over the lazy dog ".repeat(2_000)
assertTrue(findHashtags(prose).isEmpty())
assertTrue(findIndexTagsWithPeople(prose, tags).isEmpty())
}
}
@@ -0,0 +1,167 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip19Bech32
import com.vitorpamplona.quartz.nip01Core.relay.normalizer.RelayUrlNormalizer
import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32
import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress
import com.vitorpamplona.quartz.nip19Bech32.entities.NEmbed
import com.vitorpamplona.quartz.nip19Bech32.entities.NProfile
import com.vitorpamplona.quartz.nip19Bech32.entities.NRelay
import com.vitorpamplona.quartz.nip19Bech32.tlv.TlvBuilder
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertNotNull
import kotlin.test.assertTrue
/**
* Coverage for the TLV-backed entities reached *through the content scan*, and for
* [Nip19Parser.tryParseAndClean].
*
* `NIP19ParserTest` decodes these through `uriToRoute`. This asserts the same
* entities survive being found inside arbitrary text, which is the path every
* ingested note takes, and pins the relay hints and identifiers they carry.
*
* `nembed1` had no `commonTest` coverage at all (its only fixtures live in
* `androidDeviceTest`/`iosTest`), and `nrelay1` had no *positive* case anywhere —
* the codebase can parse it but has no encoder for it, so one is built here from
* TLV directly.
*/
class Nip19DecodeTest {
companion object {
const val PUBKEY = "460c25e682fda7832b52d1f22d3d22b3176d972f60dcdc3212ed8c92ef85065c"
const val NADDR =
"naddr1qqyxzmt9w358jum5qyt8wumn8ghj7un9d3shjtnwdaehgu3wvfskueqzypd7v3r24z33cydnk3fmlrd0exe5dlej3506zxs05q4puerp765mzqcyqqq8scsq6mk7u"
const val NEMBED = "nembed1r79ssq9446hkwqhl642ukmku8qg0c92pu7w3j0jyfte8tc7tvg85vmrys8x3sqgle5vjy7jpjswqhphl0kd6yf4sz0n3peyjq5rp3zkat4w6c6j3f7um0724jmfu5456xxgg2yxkn8dp23j64xsn9npcggzafyh2effyntqrqxzja8dp52kpcvc9zqxlj86e8mx05vevzxkeprjkfs4wmppxm3p96vj6yvu2mqgf5l4v99492r2qsggquxuv93uzx244652h2kkj8xseg9xkq0afpygknjtty9j4ju5v0nm9mezux9wyl6s5wr7lzce7cj397mnu0u04ha7aq3w7exelrhe3zs3l3urwa9sp36u80npllrs0hmsxqdn0fsuyav3nv0azjs5suzuurg2uymncjxez8p9xksc2j6gw992enjflgrdd7n5uq2xrpvfrd3rckw624ey0elvm6grr27tyzlf4vaswgm5vc3hdyczsl983g2j8e67r6z5zt30lat84ma4wclkwwxxrcflvdsuwd7346h7zqav4vdwe3gkt9lr87sfk4aqd2aey03tt4eyspldrqcmkx9pqe2pn63rv7grwwalr86akuldnvjm6m87wrw9sdwns8wq0rnsmj57vqwtc3g7hkwum3vl2dda78dwkycgfzw6qna3ufhpatcvq5a4hm4ehl45an8umwt0clf7rn77ctke475qglwu86hhfwhn7dkca4pkfpyc4y75rll6nvr5qc8nlhf8mk22celn5mecvyuzxd830drhdck9tcdpcafymk8wajwu2w8ha8gatggjfvq0a4jlf2sdamzj0ysqks9dk8me3q7a0qpmf6vykurkrcls4pug3u4pn4u26ezx3h8e482n07x2nsmu80dpufxqc0ttcyzhnppguxma4d8aumdawnlsyy7yzcuxl7lw5y9p4nv5h8fn6u8anpm2tsze3p6mgxy9j9uuqfxg2jvlmtjpakna5m4hln0msmw804hnun96h66fh62270yhhljnmmdl7jln07ll5vft7e870hemcld34a09n943ed6629fgtctsftma9q6tf4jfm2p0ukd2j2n2dpz53fqrkk4ctdcy2j5jar095g5jntf6u807ggkzauzt6uqkwk4tg5w7w55kskspc9663zx5dzzzfwpg3q546g2ve4kukr70n0a46eyce2crsqqq247ql5"
const val NCRYPTSEC = "ncryptsec1qgg9947rlpvqu76pj5ecreduf9jxhselq2nae2kghhvd5g7dgjtcxfqtd67p9m0w57lspw8gsq6yphnm8623nsl8xn9j4jdzz84zm3frztj3z7s35vpzmqf6ksu8r89qk5z2zxfmu5gv8th8wclt0h4p"
}
// ---------- nprofile: pubkey + relay hints ----------
@Test
fun nprofileWithRelayHintsSurvivesTheContentScan() {
val relay = RelayUrlNormalizer.normalize("wss://vitor.nostr1.com/")
val encoded = NProfile.create(PUBKEY, listOf(relay))
val found = Nip19Parser.parseAll("please follow nostr:$encoded today")
assertEquals(1, found.size)
val profile = found[0] as NProfile
assertEquals(PUBKEY, profile.hex)
assertEquals(listOf(relay), profile.relay)
}
@Test
fun nprofileWithoutRelayHintsStillDecodes() {
val encoded = NProfile.create(PUBKEY, emptyList())
val found = Nip19Parser.parseAll(encoded)
assertEquals(1, found.size)
assertEquals(PUBKEY, (found[0] as NProfile).hex)
assertTrue((found[0] as NProfile).relay.isEmpty())
}
@Test
fun nprofileWithSeveralRelayHintsKeepsAllOfThem() {
val a = RelayUrlNormalizer.normalize("wss://relay.one/")
val b = RelayUrlNormalizer.normalize("wss://relay.two/")
val encoded = NProfile.create(PUBKEY, listOf(a, b))
val profile = Nip19Parser.parseAll("x $encoded").single() as NProfile
assertEquals(listOf(a, b), profile.relay)
}
// ---------- naddr: kind:author:dTag ----------
@Test
fun naddrSurvivesTheContentScan() {
val found = Nip19Parser.parseAll("read nostr:$NADDR now")
assertEquals(1, found.size)
assertEquals(
"30818:5be6446aa8a31c11b3b453bf8dafc9b346ff328d1fa11a0fa02a1e6461f6a9b1:amethyst",
(found[0] as NAddress).aTag(),
)
}
// ---------- nrelay: no encoder in the codebase, so build one from TLV ----------
@Test
fun nrelayDecodesFromTheContentScan() {
val url = "wss://relay.example.com/"
val encoded =
TlvBuilder()
.apply { addString(TlvTypes.SPECIAL.id, url) }
.build()
.let { Bech32.encodeBytes(hrp = "nrelay", it, Bech32.Encoding.Bech32) }
val found = Nip19Parser.parseAll("join nostr:$encoded please")
assertEquals(1, found.size)
assertEquals(listOf(url), (found[0] as NRelay).relay)
}
// ---------- nembed: gzipped event, first commonTest coverage ----------
@Test
fun nembedDecodesFromTheContentScan() {
// fixture mirrored from NIP19EmbedTests (androidDeviceTest/iosTest), which
// never ran on JVM — the decode goes through GZip + Event parsing
val found = Nip19Parser.parseAll("look at nostr:$NEMBED")
assertEquals(1, found.size)
val embed = found[0] as NEmbed
assertNotNull(embed.event)
}
@Test
fun truncatedNembedIsRejectedNotThrown() {
val broken = NEMBED.take(NEMBED.length / 2)
assertEquals(emptyList(), Nip19Parser.parseAll(broken))
}
// ---------- tryParseAndClean ----------
@Test
fun tryParseAndCleanStripsSchemeAndTrailingCharacters() {
val npub = Nip19ScanTest.NPUB
assertEquals(npub, Nip19Parser.tryParseAndClean("nostr:$npub"))
assertEquals(npub, Nip19Parser.tryParseAndClean(npub))
assertEquals(npub, Nip19Parser.tryParseAndClean("@$npub"))
}
@Test
fun tryParseAndCleanReturnsNullForNonEntities() {
assertEquals(null, Nip19Parser.tryParseAndClean(null))
assertEquals(null, Nip19Parser.tryParseAndClean(""))
assertEquals(null, Nip19Parser.tryParseAndClean("just some words"))
assertEquals(null, Nip19Parser.tryParseAndClean("npub1tooshort"))
}
@Test
fun tryParseAndCleanNeverReturnsALiteralNullSuffix() {
// `type!! + key` would render "npub1null" if the key group were ever absent
listOf("nostr:${Nip19ScanTest.NPUB}", Nip19ScanTest.NOTE, "nostr:${Nip19ScanTest.NEVENT}")
.forEach { assertTrue(Nip19Parser.tryParseAndClean(it)?.endsWith("null") != true, "for $it") }
}
@Test
fun tryParseAndCleanAcceptsNcryptsecWhichTheContentScanDoesNot() {
// nip19PlusNip46regex includes ncryptsec1; nip19regex (used by parseAll) does not
val ncryptsec = NCRYPTSEC
assertEquals(ncryptsec, Nip19Parser.tryParseAndClean(ncryptsec))
assertEquals(emptyList(), Nip19Parser.parseAll(ncryptsec))
}
}
@@ -0,0 +1,214 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip19Bech32
import com.vitorpamplona.quartz.nip19Bech32.entities.NEvent
import com.vitorpamplona.quartz.nip19Bech32.entities.NNote
import com.vitorpamplona.quartz.nip19Bech32.entities.NPub
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertTrue
/**
* Behavioural coverage for [Nip19Parser.parseAll] — the whole-content scan run on
* every ingested note.
*
* `NIP19ParserTest` covers [Nip19Parser.uriToRoute] (one entity, already isolated).
* This covers the other half: finding entities *inside* arbitrary text — where they
* may sit, what may precede them, and what must NOT be mistaken for one.
*
* In `commonTest` on purpose: [Nip19Parser] is `commonMain` and the scan uses
* `matchAt`/`regionMatches`, so it must behave identically on every target.
*/
class Nip19ScanTest {
companion object {
const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y"
const val NPUB_HEX = "bb3d6543d4a4f45f402b674ecb5cb2155ff12456fcdd2d66d9727660d3037365"
const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv"
const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy"
}
@Test
fun findsBareEntity() {
val found = Nip19Parser.parseAll(NPUB)
assertEquals(1, found.size)
assertEquals(NPUB_HEX, (found[0] as NPub).hex)
}
@Test
fun findsEntityWithNostrScheme() {
val found = Nip19Parser.parseAll("hello nostr:$NPUB world")
assertEquals(1, found.size)
assertEquals(NPUB_HEX, (found[0] as NPub).hex)
}
@Test
fun findsEntityWithAtPrefix() {
val found = Nip19Parser.parseAll("cc @$NPUB thanks")
assertEquals(1, found.size)
assertEquals(NPUB_HEX, (found[0] as NPub).hex)
}
@Test
fun findsEntityAtStartMiddleAndEndOfContent() {
assertEquals(1, Nip19Parser.parseAll("$NPUB trailing text").size)
assertEquals(1, Nip19Parser.parseAll("leading $NPUB trailing").size)
assertEquals(1, Nip19Parser.parseAll("leading text $NPUB").size)
}
@Test
fun findsDifferentEntityTypes() {
assertTrue(Nip19Parser.parseAll("see $NOTE")[0] is NNote)
assertTrue(Nip19Parser.parseAll("see $NEVENT")[0] is NEvent)
}
@Test
fun findsSeveralEntitiesInOneNote() {
val found = Nip19Parser.parseAll("$NPUB then $NOTE and nostr:$NEVENT")
assertEquals(3, found.size)
}
@Test
fun entityGluedToAPrecedingWordIsStillFound() {
// the regex has no leading whitespace requirement
assertEquals(1, Nip19Parser.parseAll("x$NPUB").size)
}
@Test
fun uppercaseEntityIsFound() {
// the regex is IGNORE_CASE; bech32 decode is case-insensitive too
assertEquals(1, Nip19Parser.parseAll("NOSTR:${NPUB.uppercase()}").size)
}
@Test
fun invalidChecksumIsNotReturned() {
val broken = NPUB.dropLast(6) + "qqqqqq"
assertEquals(emptyList(), Nip19Parser.parseAll(broken))
}
@Test
fun tooShortNpubIsNotMatched() {
assertEquals(emptyList(), Nip19Parser.parseAll("npub1tooshort"))
assertEquals(emptyList(), Nip19Parser.parseAll("npub1"))
}
@Test
fun proseWithoutEntitiesReturnsNothing() {
assertEquals(emptyList(), Nip19Parser.parseAll("no entities in this note at all"))
// words that start like a prefix but are not one
assertEquals(emptyList(), Nip19Parser.parseAll("nostrich nope never nan note nsec nprofile"))
}
@Test
fun contentEndingInNDoesNotReadPastTheEnd() {
// the candidate scan dispatches on the character AFTER 'n'
assertEquals(emptyList(), Nip19Parser.parseAll("n"))
assertEquals(emptyList(), Nip19Parser.parseAll("N"))
assertEquals(emptyList(), Nip19Parser.parseAll("ends with n"))
assertEquals(emptyList(), Nip19Parser.parseAll("trailing N"))
listOf("np", "ne", "na", "nr", "ns", "no", "nn").forEach {
assertEquals(emptyList(), Nip19Parser.parseAll(it), "for '$it'")
assertEquals(emptyList(), Nip19Parser.parseAll("text $it"), "for 'text $it'")
}
}
@Test
fun emptyContentReturnsNothing() {
assertEquals(emptyList(), Nip19Parser.parseAll(""))
assertEquals(emptyList(), Nip19Parser.parseAll(" "))
}
@Test
fun adjacentEntitiesWithNoSeparatorYieldOne() {
// the trailing ([\S]*) group is greedy, so it swallows the second entity —
// documented behaviour, pinned here so the scan rewrite cannot change it
assertEquals(1, Nip19Parser.parseAll(NPUB + NOTE).size)
// separated by a space, both are found
assertEquals(2, Nip19Parser.parseAll("$NPUB $NOTE").size)
}
@Test
fun trailingPunctuationDoesNotBreakTheEntity() {
assertEquals(1, Nip19Parser.parseAll("thanks nostr:$NPUB!").size)
assertEquals(1, Nip19Parser.parseAll("(nostr:$NPUB)").size)
}
@Test
fun entityInsideMultilineContent() {
assertEquals(1, Nip19Parser.parseAll("line one\nnostr:$NPUB\nline three").size)
}
@Test
fun scanDoesNotStopAtTheFirstNonEntityCandidate() {
// "nostrich" and "nevermind" are 'n' candidates that fail; a real entity follows
assertEquals(1, Nip19Parser.parseAll("nostrich nevermind nap $NPUB").size)
}
// ---------- parseAllEvents: same scan, event-ish entities only ----------
@Test
fun parseAllEventsFindsEventEntities() {
assertEquals(1, Nip19Parser.parseAllEvents("see nostr:$NEVENT").size)
assertEquals(1, Nip19Parser.parseAllEvents("see $NOTE").size)
}
@Test
fun parseAllEventsIgnoresProfileEntities() {
// npub1/nsec1/nprofile1 are not in nip19regexEvents
assertEquals(emptyList(), Nip19Parser.parseAllEvents(NPUB))
assertEquals(emptyList(), Nip19Parser.parseAllEvents("hello nostr:$NPUB there"))
}
@Test
fun parseAllEventsSkipsProfilesButStillFindsLaterEvents() {
// an npub is a candidate position that must not stop the scan
val found = Nip19Parser.parseAllEvents("$NPUB then nostr:$NEVENT")
assertEquals(1, found.size)
assertTrue(found[0] is NEvent)
}
@Test
fun parseAllEventsHandlesEdgesLikeParseAll() {
assertEquals(emptyList(), Nip19Parser.parseAllEvents(""))
assertEquals(emptyList(), Nip19Parser.parseAllEvents("n"))
assertEquals(emptyList(), Nip19Parser.parseAllEvents("ends with n"))
assertEquals(emptyList(), Nip19Parser.parseAllEvents("no entities here"))
assertEquals(2, Nip19Parser.parseAllEvents("$NOTE $NEVENT").size)
}
// ---------- uriToRoute: unchanged, pinned against the shared candidate scan ----------
@Test
fun uriToRouteStillResolvesEachForm() {
assertTrue(Nip19Parser.uriToRoute("nostr:$NPUB")?.entity is NPub)
assertTrue(Nip19Parser.uriToRoute(NPUB)?.entity is NPub)
assertTrue(Nip19Parser.uriToRoute("nostr:$NEVENT")?.entity is NEvent)
assertEquals(null, Nip19Parser.uriToRoute(null))
assertEquals(null, Nip19Parser.uriToRoute("not an entity"))
assertEquals(null, Nip19Parser.uriToRoute(""))
}
@Test
fun longProseWithNoEntitiesTerminates() {
val prose = "the nimble nocturnal nightingale never naps near noon ".repeat(2_000)
assertEquals(emptyList(), Nip19Parser.parseAll(prose))
}
}