Merge remote-tracking branch 'origin/main' into claude/concord-nip29-invitations-vx2wgj

This commit is contained in:
Claude
2026-08-18 01:37:48 +00:00
8 changed files with 724 additions and 77 deletions
@@ -22,19 +22,43 @@ package com.vitorpamplona.quartz.nip10Notes.content
val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)")
/**
* Characters that end a hashtag: the punctuation class spelled out in [hashtagSearch], plus ASCII
* whitespace. Everything else continues the tag — including every non-ASCII character, since the
* regex's class is ASCII-only, so accented letters, CJK and emoji are all valid tag content.
*/
private val HASHTAG_TERMINATORS =
BooleanArray(128).apply {
for (c in 0x09..0x0D) this[c] = true
this[' '.code] = true
for (c in "!@#\u0024%^&*()=+./,[{]};:'\"?><") this[c.code] = true
}
/**
* True while [c] can still be part of a hashtag.
*
* Non-ASCII always continues the tag: [hashtagSearch]'s excluded set is entirely ASCII and its
* `\s` is ASCII-only, so nothing above 0x7F was ever excluded.
*/
private fun isHashtagChar(c: Char): Boolean = c.code >= 128 || !HASHTAG_TERMINATORS[c.code]
/**
* Collects the hashtags in [content].
*
* [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match
* starts either at position 0 or at a whitespace. That lets the scan jump between
* `#` occurrences with `indexOf` — an intrinsified char search — and apply the
* regex **anchored** at each, instead of letting `findAll` drive the regex engine
* from every position in the string.
* Jumps between `#` occurrences with `indexOf` — an intrinsified char search — and then matches
* `#<tag>` directly, character by character, rather than anchoring a regex there.
*
* Measured on the production content distribution (median 529 B, tail to 767 KB):
* ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the
* content has no `#` at all (up to 300x). Equivalence with the previous `findAll`
* implementation is guarded by `RegexContentBenchmark` in `commons`.
* **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` ->
* `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory on every
* call. `Regex.matchAt` builds a fresh Matcher per call, so anchoring one at each candidate cost a
* full native UTF-16 copy of the note's content **per `#`** — the same defect that drove the app's
* native heap to ~1.9GB on a cold start via the NIP-19 scanner. This one is worse: measured over
* 2588 real notes it minted 43,626 Matchers copying 9.6GB in total, with a single 119KB note
* costing 279MB, because a whitespace-preceded `#` is far more common in prose than a NIP-19
* prefix. The Java `Matcher` object is tiny, so Java-heap-driven GC had no reason to reclaim them
* promptly while each pinned native memory.
*
* [hashtagSearch] is kept as the specification the scan is tested against, not used here.
*/
fun findHashtags(
content: String,
@@ -44,19 +68,17 @@ fun findHashtags(
var h = content.indexOf('#')
while (h >= 0) {
if (h == 0 || content[h - 1].isWhitespace()) {
val match =
try {
hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1)
} catch (e: Exception) {
null
}
if (match != null) {
val tag = match.groups[1]?.value
if (tag != null && tag.isNotBlank()) {
output.add(tag)
}
h = content.indexOf('#', match.range.last + 1)
// `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character.
if (h == 0 || isAsciiRegexSpace(content[h - 1])) {
var end = h + 1
while (end < content.length && isHashtagChar(content[end])) end++
// The tag group is `+`, so it needs at least one character.
if (end > h + 1) {
val tag = content.substring(h + 1, end)
// Non-ASCII whitespace (U+00A0 and friends) is valid tag content to the regex but
// still blank to Kotlin, and the old code dropped those too.
if (tag.isNotBlank()) output.add(tag)
h = content.indexOf('#', end)
continue
}
}
@@ -0,0 +1,31 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip10Notes.content
/**
* `\s` as `java.util.regex` applies it *without* `UNICODE_CHARACTER_CLASS`: space plus the five
* control characters `\t \n \x0B \f \r`, which are contiguous at 0x09..0x0D — and nothing else.
*
* The scanners in this package hand-roll grammars that used to be regexes, so they must use this
* rather than [Char.isWhitespace], which is Unicode-aware and would accept U+00A0, U+2003 and
* friends that the regexes rejected.
*/
internal fun isAsciiRegexSpace(c: Char): Boolean = c == ' ' || c.code in 0x09..0x0D
@@ -29,36 +29,36 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray
val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]")
/**
* Walks every `#[n]` reference in [content].
* Walks every `#[n]` reference in [content], handing each callback the digits between the brackets.
*
* [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match
* starts at position 0 or at a whitespace. That lets the scan jump between `#`
* occurrences with `indexOf` — an intrinsified char search — and apply the regex
* **anchored** at each, instead of letting `findAll` drive the regex engine from
* every position in the string.
* Jumps between `#` occurrences with `indexOf` — an intrinsified char search — then matches
* `#[<digits>]` directly rather than anchoring a regex there.
*
* Measured on the production content distribution (median 529 B, tail to 767 KB):
* ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense
* text. Equivalence with the previous `findAll` implementation (both callers) is
* guarded by `RegexContentBenchmark` in `commons`.
* **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` ->
* `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory per call,
* so anchoring a fresh Matcher at each candidate cost a full native UTF-16 copy of the content per
* `#`. See [findHashtags] for the measurements; this scanner shares the defect but never fires on
* real data, since `#[0]` is the legacy citation form that no current client emits.
*
* [tagSearch] is kept as the specification the scan is tested against, not used here.
*/
private inline fun forEachIndexTag(
content: String,
action: (MatchResult) -> Unit,
action: (digits: String) -> Unit,
) {
var h = content.indexOf('#')
while (h >= 0) {
if (h == 0 || content[h - 1].isWhitespace()) {
val match =
try {
tagSearch.matchAt(content, if (h == 0) 0 else h - 1)
} catch (e: Exception) {
null
// `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character.
if (h == 0 || isAsciiRegexSpace(content[h - 1])) {
if (h + 1 < content.length && content[h + 1] == '[') {
var d = h + 2
while (d < content.length && content[d] in '0'..'9') d++
// `([0-9]+)` needs a digit, and the `]` must actually be there.
if (d > h + 2 && d < content.length && content[d] == ']') {
action(content.substring(h + 2, d))
h = content.indexOf('#', d + 1)
continue
}
if (match != null) {
action(match)
h = content.indexOf('#', match.range.last + 1)
continue
}
}
h = content.indexOf('#', h + 1)
@@ -73,10 +73,11 @@ fun findIndexTagsWithPeople(
tags: TagArray,
output: MutableSet<String> = mutableSetOf<String>(),
): List<String> {
forEachIndexTag(content) { index ->
forEachIndexTag(content) { digits ->
try {
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == "p") {
// Out-of-range indexes and non-numeric digits land in the catch below.
val tag = tags[digits.toInt()]
if (tag.size > 1 && tag[0] == "p") {
output.add(tag[1])
}
} catch (e: Exception) {
@@ -94,13 +95,14 @@ fun findIndexTagsWithEventsOrAddresses(
tags: TagArray,
output: MutableSet<String> = mutableSetOf<String>(),
): Set<String> {
forEachIndexTag(content) { index ->
forEachIndexTag(content) { digits ->
try {
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == "e") {
// Out-of-range indexes and non-numeric digits land in the catch below.
val tag = tags[digits.toInt()]
if (tag.size > 1 && tag[0] == "e") {
output.add(tag[1])
}
if (tag != null && tag.size > 1 && tag[0] == "a") {
if (tag.size > 1 && tag[0] == "a") {
output.add(tag[1])
}
} catch (e: Exception) {
@@ -24,6 +24,7 @@ import androidx.compose.runtime.Immutable
import com.vitorpamplona.quartz.nip01Core.core.HexKey
import com.vitorpamplona.quartz.nip01Core.core.hexToByteArray
import com.vitorpamplona.quartz.nip01Core.core.toHexKey
import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32
import com.vitorpamplona.quartz.nip19Bech32.bech32.bechToBytes
import com.vitorpamplona.quartz.nip19Bech32.entities.Entity
import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress
@@ -133,6 +134,67 @@ object Nip19Parser {
fun hasAny(content: String): Boolean = nip19regex.matches(content)
// ---------------------------------------------------------------------------------------
// ICU-free scanner for the ingest hot path.
//
// `java.util.regex` on Android is backed by ICU, and `Matcher.region()` -> `reset()` ->
// `MatcherNative.setInput()` copies the ENTIRE input into native memory on every call.
// `Regex.matchAt` builds a fresh Matcher per call, and [forEachNip19Match] calls it once per
// candidate position — so scanning one note allocated a full native UTF-16 copy of its content
// *per candidate*, with content running to 767KB in the tail. Because the Java Matcher object
// is tiny, Java-heap-driven GC had no reason to reclaim them promptly, so the native heap grew
// unbounded: measured 45MB -> 1372MB over a cold start, ending in an lmkd kill at ~1.9GB RSS.
// Disabling this one scan made the native heap plateau at ~257MB instead.
//
// The grammar is small enough to match directly, so nothing here touches ICU. Case folding is
// done ASCII-only on purpose: `RegexOption.IGNORE_CASE` maps to `Pattern.CASE_INSENSITIVE`,
// which is ASCII-only unless `UNICODE_CASE` is also set. Using Kotlin's `ignoreCase = true`
// would be Unicode-aware and would accept inputs the regex rejected (U+212A KELVIN SIGN folding
// to `k`, say).
// ---------------------------------------------------------------------------------------
/** Entities whose bech32 payload the regexes pin to exactly 58 chars. */
private val FIXED_58_PREFIXES = arrayOf("nsec1", "npub1", "note1")
/** Entities whose bech32 payload is `+` (one or more). */
private val VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "nprofile1", "nrelay1", "nembed1")
/** [nip19regexEvents] has no fixed-58 branch — there `note1` takes a variable payload. */
private val EVENT_FIXED_58_PREFIXES = emptyArray<String>()
private val EVENT_VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "note1", "nrelay1", "nembed1")
/**
* `\S` in the trailing group. Java's `\s` is the six ASCII whitespace chars unless
* `UNICODE_CHARACTER_CLASS` is set, so U+00A0 and friends count as NON-space here.
*/
private fun isRegexSpace(c: Char): Boolean = c == ' ' || c == '\t' || c == '\n' || c == '\u000B' || c == '\u000C' || c == '\r'
/** ASCII-only case-insensitive prefix compare; [prefix] must be lowercase ASCII. */
private fun matchesPrefixAt(
content: String,
offset: Int,
prefix: String,
): Boolean {
if (offset + prefix.length > content.length) return false
for (k in prefix.indices) {
val c = content[offset + k]
val folded = if (c in 'A'..'Z') c + 32 else c
if (folded != prefix[k]) return false
}
return true
}
/** End (exclusive) of the `[\S]*` run starting at [from]. */
private fun endOfTrailing(
content: String,
from: Int,
): Int {
var j = from
while (j < content.length && !isRegexSpace(content[j])) j++
return j
}
/**
* True when one of the NIP-19 entity prefixes starts at [i].
*
@@ -165,26 +227,84 @@ object Nip19Parser {
}
/**
* Applies [regex] anchored at every NIP-19 candidate position in [content].
* Matches `<prefix><bech32 payload><[\S]*>` at every NIP-19 candidate position in [content],
* without ICU — see the block comment above [FIXED_58_PREFIXES] for why that matters.
*
* [isCandidateAt] covers the union of the prefixes across the three NIP-19
* regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does
* not accept — still far cheaper than `findAll` restarting the engine at
* every position in the string.
* [isCandidateAt] covers the union of the prefixes across the three NIP-19 regexes, so a
* caller passing a narrower prefix set simply finds no match at a prefix it does not accept.
*
* The prefixes are mutually exclusive (none is a prefix of another), so the first one that
* matches decides the branch, exactly as the regex alternation did. A fixed-58 entity with
* fewer than 58 payload chars fails outright rather than falling through to the variable
* branch, again matching the regex: no variable prefix can equal a fixed-58 one.
*/
private inline fun forEachNip19Match(
content: String,
regex: Regex,
action: (MatchResult) -> Unit,
fixed58Prefixes: Array<String>,
variablePrefixes: Array<String>,
action: (type: String, key: String, additionalChars: String) -> Unit,
) {
var i = 0
val len = content.length
// Jump between 'n'/'N' with indexOf — an intrinsified, vectorised char search — instead of
// testing every character. Both cases are tracked separately so each is only re-searched
// once consumed, amortising to about one indexOf per candidate. Measured ~2x faster on
// content with no entity at all, which is 94% of real notes (2588-note production sample),
// and ~26% faster on the 767KB mention-heavy tail. Small mention-dense content (a ~700B
// note with 2 mentions) is a few microseconds slower; that trade is deliberate.
var nextLower = content.indexOf('n')
var nextUpper = content.indexOf('N')
while (i < len) {
if (nextLower in 0..<i) nextLower = content.indexOf('n', i)
if (nextUpper in 0..<i) nextUpper = content.indexOf('N', i)
i =
when {
nextLower < 0 && nextUpper < 0 -> return
nextLower < 0 -> nextUpper
nextUpper < 0 -> nextLower
else -> if (nextLower < nextUpper) nextLower else nextUpper
}
if (i >= len) return
if (isCandidateAt(content, i)) {
val match = regex.matchAt(content, i)
if (match != null) {
action(match)
i = match.range.last + 1
var end = -1
for (prefix in fixed58Prefixes) {
if (!matchesPrefixAt(content, i, prefix)) continue
val dataStart = i + prefix.length
val dataEnd = dataStart + 58
if (dataEnd <= len && allBech32(content, dataStart, dataEnd)) {
end = endOfTrailing(content, dataEnd)
action(
content.substring(i, dataStart),
content.substring(dataStart, dataEnd),
content.substring(dataEnd, end),
)
}
break
}
if (end < 0) {
for (prefix in variablePrefixes) {
if (!matchesPrefixAt(content, i, prefix)) continue
val dataStart = i + prefix.length
var dataEnd = dataStart
while (dataEnd < len && Bech32.isDataChar(content[dataEnd])) dataEnd++
// `+` needs at least one payload char.
if (dataEnd > dataStart) {
end = endOfTrailing(content, dataEnd)
action(
content.substring(i, dataStart),
content.substring(dataStart, dataEnd),
content.substring(dataEnd, end),
)
}
break
}
}
// `end` is exclusive and always past `i`, so the scan still advances.
if (end >= 0) {
i = end
continue
}
}
@@ -192,6 +312,17 @@ object Nip19Parser {
}
}
private fun allBech32(
content: String,
from: Int,
to: Int,
): Boolean {
for (k in from until to) {
if (!Bech32.isDataChar(content[k])) return false
}
return true
}
/**
* Scans [content] for NIP-19 entities.
*
@@ -208,14 +339,8 @@ object Nip19Parser {
*/
fun parseAll(content: String): List<Entity> {
val returningList = mutableListOf<Entity>()
forEachNip19Match(content, nip19regex) { matcher ->
val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1
val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32
val additionalChars = matcher.groups[7]?.value // additional chars
if (type != null) {
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
}
forEachNip19Match(content, FIXED_58_PREFIXES, VARIABLE_PREFIXES) { type, key, additionalChars ->
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
}
return returningList
}
@@ -223,14 +348,8 @@ object Nip19Parser {
/** Same scan as [parseAll], restricted to the event-ish entities. */
fun parseAllEvents(content: String): List<Entity> {
val returningList = mutableListOf<Entity>()
forEachNip19Match(content, nip19regexEvents) { matcher ->
val type = matcher.groups[2]?.value // nevent1
val key = matcher.groups[3]?.value // bech32
val additionalChars = matcher.groups[4]?.value // additional chars
if (type != null) {
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
}
forEachNip19Match(content, EVENT_FIXED_58_PREFIXES, EVENT_VARIABLE_PREFIXES) { type, key, additionalChars ->
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
}
return returningList
}
@@ -20,6 +20,8 @@
*/
package com.vitorpamplona.quartz.nip19Bech32.bech32
import kotlin.jvm.JvmField
/*
* Copyright 2020 ACINQ SAS
*
@@ -81,15 +83,50 @@ object Bech32 {
// char -> 5 bits value
private val map = Array<Int5>(255) { -1 }
@PublishedApi
internal const val DATA_CHARS_SIZE = 128
/**
* Membership table for [isDataChar], kept separate from [map] because [map] is an
* `Array<Byte>` — a boxed `java.lang.Byte[]` — and [isDataChar] runs per character over whole
* note contents (up to ~767KB), where unboxing on every char would show up.
*
* `@JvmField` so callers of the inline [isDataChar] compile to a direct `getstatic` instead of
* a property-getter `invokevirtual` per character (the same reasoning as `PointTypes`).
* `@PublishedApi internal` because a public inline function cannot touch a private member.
*/
@PublishedApi
@JvmField
internal val DATA_CHARS = BooleanArray(DATA_CHARS_SIZE)
init {
for (i in 0..ALPHABET.lastIndex) {
map[ALPHABET[i].code] = i.toByte()
DATA_CHARS[ALPHABET[i].code] = true
}
for (i in 0..ALPHABET_UPPERCASE.lastIndex) {
map[ALPHABET_UPPERCASE[i].code] = i.toByte()
DATA_CHARS[ALPHABET_UPPERCASE[i].code] = true
}
}
/**
* True when [c] is part of the bech32 data alphabet, in either case — i.e. everything except
* `1`, `b`, `i` and `o`, which BIP-173 excludes as visually ambiguous.
*
* Exposed so scanners can find where an encoded payload *ends* without decoding it. The NIP-19
* content scan needs exactly that on every ingested event, and cannot use a regex to do it:
* Android's `java.util.regex` is ICU-backed and `Matcher.region()` copies the entire input into
* native memory per call, which drove the app's native heap to ~1.9GB on a cold start.
*
* `inline` because it is called per character over whole note contents (up to ~767KB). As a
* normal member it compiled to an `invokevirtual` per char, which measured ~1-2% slower across
* the scan benchmark than the equivalent private lookup it replaced; inlining puts the constant
* compare and the `baload` straight into the caller's loop and closes that gap.
*/
@Suppress("NOTHING_TO_INLINE")
inline fun isDataChar(c: Char): Boolean = c.code < DATA_CHARS_SIZE && DATA_CHARS[c.code]
fun expand(hrp: String): Array<Int5> {
val half = hrp.length + 1
val size = half + hrp.length
@@ -0,0 +1,167 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip10Notes.content
import kotlin.test.Test
import kotlin.test.assertEquals
/**
* Pins the ICU-free content scanners to the regexes they replaced.
*
* [findHashtags] and the `#[n]` walker used to anchor a fresh `Regex.matchAt` at every candidate.
* On Android that goes through ICU, where `Matcher.region()` copies the whole input into native
* memory per call — over 2588 real notes that was 43,626 Matchers copying 9.6GB, worst single note
* 279MB. The scanners now match the grammars directly, so [hashtagSearch] and [tagSearch] survive
* only as the specification, and this asserts the two agree.
*
* The corpus targets where a hand-rolled matcher is most likely to drift: the exact punctuation
* set that ends a tag, ASCII-vs-Unicode whitespace before the `#`, non-ASCII inside the tag, and
* the `+`/`[0-9]+` minimum-one-character rules.
*/
class ContentScanRegexEquivalenceTest {
private fun referenceHashtags(content: String): List<String> {
if (content.isBlank()) return emptyList()
val out = mutableSetOf<String>()
hashtagSearch.findAll(content).forEach { m ->
val tag = m.groups[1]?.value
if (tag != null && tag.isNotBlank()) out.add(tag)
}
return out.toList()
}
private fun referenceIndexTags(
content: String,
tags: Array<Array<String>>,
wanted: String,
): Set<String> {
val out = mutableSetOf<String>()
tagSearch.findAll(content).forEach { m ->
try {
val tag = m.groups[1]?.value?.let { tags[it.toInt()] }
if (tag != null && tag.size > 1 && tag[0] == wanted) out.add(tag[1])
} catch (e: Exception) {
}
}
return out
}
private val tagArray =
arrayOf(
arrayOf("p", "pubkey0"),
arrayOf("e", "event1"),
arrayOf("a", "addr2"),
arrayOf("p", "pubkey3"),
arrayOf("t", "topic4"),
)
private fun corpus(): List<String> =
buildList {
add("")
add(" ")
add("#")
add("#tag")
add("hello #tag world")
add("a#tag")
add("#tag#other")
add("#tag #other")
add("##tag")
add("#tag.")
add("#tag, and #more!")
add("#tag's")
add("#tag\"quoted\"")
add("#a")
add("#1")
add("#tag-with-dash")
add("#tag_with_underscore")
add("#tag~tilde|pipe\\back`tick")
// non-ASCII is valid tag content: the regex class and its \s are ASCII-only
add("#café")
add("#日本語")
add("#tagéè")
// ASCII vs Unicode whitespace BEFORE the # decides whether it matches at all
add("x\u00A0#tag")
add("x\u2003#tag")
add("x\t#tag")
add("x\n#tag")
add("x\r#tag")
// Unicode whitespace INSIDE the tag is valid to the regex but blank to Kotlin
add("#\u00A0")
add("#\u00A0x")
// every excluded char must terminate the tag
for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#tag${c}more")
for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#$c")
// index tags
add("#[0]")
add("#[1] and #[2]")
add("look #[3] here")
add("#[]")
add("#[abc]")
add("#[99]")
add("#[0")
add("#[0]]")
add("x#[0]")
add("x\u00A0#[0]")
add("#[0]#[1]")
add("#[00]")
// mixed
add("#tag #[0] #other #[1]")
add("lorem ipsum ".repeat(500) + "#tail")
add("#head" + " dolor sit ".repeat(500))
add("no hashes at all here ".repeat(200))
}
@Test
fun hashtagsMatchRegex() {
corpus().forEach { c ->
assertEquals(
referenceHashtags(c).sorted(),
findHashtags(c).sorted(),
"findHashtags diverged on: ${c.take(80)}",
)
}
}
@Test
fun indexTagsMatchRegex() {
corpus().forEach { c ->
assertEquals(
referenceIndexTags(c, tagArray, "p").sorted(),
findIndexTagsWithPeople(c, tagArray).sorted(),
"findIndexTagsWithPeople diverged on: ${c.take(80)}",
)
val refEv = referenceIndexTags(c, tagArray, "e") + referenceIndexTags(c, tagArray, "a")
assertEquals(
refEv.sorted(),
findIndexTagsWithEventsOrAddresses(c, tagArray).sorted(),
"findIndexTagsWithEventsOrAddresses diverged on: ${c.take(80)}",
)
}
}
@Test
fun hashtagsMatchRegexAtEveryOffset() {
val filler = "a b\tc\nd "
for (i in 0..filler.length) {
val c = filler.substring(0, i) + "#tag" + filler.substring(i)
assertEquals(referenceHashtags(c).sorted(), findHashtags(c).sorted(), "offset $i")
}
}
}
@@ -0,0 +1,202 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip19Bech32
import com.vitorpamplona.quartz.nip19Bech32.entities.Entity
import kotlin.test.Test
import kotlin.test.assertEquals
/**
* Pins the ICU-free scanner in [Nip19Parser] to the regexes it replaced.
*
* The scan used to run `Regex.matchAt` at every candidate position. On Android that goes through
* ICU, and `Matcher.region()` copies the whole input into native memory on each call — one full
* native copy of a note's content *per candidate* — which drove the native heap to ~1.9GB on a
* cold start and got the process lmkd-killed. The scanner matches the grammar directly instead.
*
* [Nip19Parser.nip19regex] and [Nip19Parser.nip19regexEvents] are still the specification, so this
* runs both over the same corpus and requires identical entity lists. The corpus deliberately
* targets the places a hand-rolled matcher is most likely to drift from the regex: the exact-58
* payload boundary, the bech32 alphabet's excluded characters, ASCII-only case folding, and which
* characters `[\S]*` is willing to swallow.
*/
class Nip19ScannerRegexEquivalenceTest {
companion object {
const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y"
const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv"
const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy"
/** 58 valid bech32 chars, so `npub1` + this is exactly the fixed-length branch. */
const val PAYLOAD58 = "qpzry9x8gf2tvdw0s3jn54khce6mua7lqpzry9x8gf2tvdw0s3jn54khce"
}
/** What `parseAll` did before: drive the regex from every position with `findAll`. */
private fun referenceParseAll(content: String): List<Entity> {
val out = mutableListOf<Entity>()
Nip19Parser.nip19regex.findAll(content).forEach { m ->
val type = m.groups[3]?.value ?: m.groups[5]?.value
val key = m.groups[4]?.value ?: m.groups[6]?.value
val additionalChars = m.groups[7]?.value
if (type != null) {
Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) }
}
}
return out
}
private fun referenceParseAllEvents(content: String): List<Entity> {
val out = mutableListOf<Entity>()
Nip19Parser.nip19regexEvents.findAll(content).forEach { m ->
val type = m.groups[2]?.value
val key = m.groups[3]?.value
val additionalChars = m.groups[4]?.value
if (type != null) {
Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) }
}
}
return out
}
private fun assertSameAsRegex(content: String) {
assertEquals(
referenceParseAll(content),
Nip19Parser.parseAll(content),
"parseAll diverged from nip19regex on: ${content.take(90)}",
)
assertEquals(
referenceParseAllEvents(content),
Nip19Parser.parseAllEvents(content),
"parseAllEvents diverged from nip19regexEvents on: ${content.take(90)}",
)
}
private fun corpus(): List<String> =
buildList {
// plain placement
add("")
add(NPUB)
add("hello $NPUB world")
add("nostr:$NPUB")
add("@$NPUB")
add("nostr:@$NPUB")
add("prefix-nostr:$NPUB-suffix")
add(NOTE)
add(NEVENT)
// adjacency and repetition — where scan-resume position matters
add(NPUB + NEVENT)
add("$NPUB $NEVENT")
add("$NPUB\n$NEVENT")
add("$NPUB,$NEVENT")
add(listOf(NPUB, NOTE, NEVENT).joinToString(" "))
add(NPUB.repeat(3))
// the exact-58 boundary for npub/nsec/note
add("npub1" + PAYLOAD58)
add("npub1" + PAYLOAD58.dropLast(1)) // 57 -> must not match
add("npub1" + PAYLOAD58 + "q") // 59 -> 58 key, trailing takes the rest
add("npub1" + PAYLOAD58 + " tail")
add("note1" + PAYLOAD58)
add("nsec1" + PAYLOAD58)
// bech32 alphabet: 1, b, i, o are excluded and must terminate the payload
add("nevent1qqs1qqs")
add("nevent1qqsbqqs")
add("nevent1qqsiqqs")
add("nevent1qqsoqqs")
// A *valid* variable-length entity butted straight against an excluded char. The
// payload has to stop there and still decode. These are the cases with teeth: a
// charset that wrongly accepted b/i/o/1 would swallow the extra char, fail the
// bech32 decode and silently drop the entity — whereas cases whose payload is
// invalid either way agree trivially and prove nothing.
for (excluded in listOf("b", "i", "o", "1")) {
add(NEVENT + excluded)
add(NEVENT + excluded + "xyz")
add("$NEVENT$excluded more text")
}
add("nevent1") // variable branch needs >= 1 payload char
add("nprofile1")
add("naddr1q")
// ASCII-only case folding
add(NPUB.uppercase())
add("NOSTR:" + NPUB.uppercase())
add(NEVENT.uppercase())
add("nPuB1" + PAYLOAD58)
// U+212A KELVIN SIGN folds to 'k' under Unicode rules but NOT under the regex's
// ASCII-only CASE_INSENSITIVE; both sides must reject it.
add("npub1" + PAYLOAD58.replaceFirst("k", "K"))
// what [\S]* may swallow: Java's \s is the six ASCII whitespace chars only,
// so U+00A0 and U+2003 are NON-space and belong to the trailing group.
add("$NPUB\u00A0more")
add("$NPUB\u2003more")
add("${NPUB}more")
add("${NPUB}1more")
add("$NPUB\tmore")
add("$NPUB\rmore")
// near-misses that must not be mistaken for entities
add("n")
add("nn")
add("np")
add("no")
add("nostr:")
add("nothing to see here")
add("a note about nothing")
add("nopqrstuvwxyz")
add("x$NPUB")
add("1$NPUB")
// long content with the entity at the far end (the 767KB-tail shape, scaled down)
add("lorem ipsum ".repeat(2000) + NPUB)
add(NPUB + " " + "dolor sit amet ".repeat(2000))
// many 'n' candidates but no entities — the scanner's rejection path
add("neither nor none never nothing ".repeat(500))
}
@Test
fun matchesRegexAcrossCorpus() {
corpus().forEach { assertSameAsRegex(it) }
}
@Test
fun matchesRegexWithEntityAtEveryOffset() {
// Slides the entity through a filler string so every start offset, including
// immediately after another candidate 'n', is exercised.
val filler = "n no non nost nostr "
for (i in 0..filler.length) {
assertSameAsRegex(filler.substring(0, i) + NPUB + filler.substring(i))
}
}
@Test
fun matchesRegexOnTruncatedPayloads() {
// Every truncation of a real entity: catches off-by-one at the 58 boundary and in `+`.
for (entity in listOf(NPUB, NOTE, NEVENT)) {
for (len in 1..entity.length) {
assertSameAsRegex(entity.substring(0, len))
assertSameAsRegex("text " + entity.substring(0, len) + " text")
}
}
}
}
@@ -0,0 +1,67 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip19Bech32.bech32
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertFalse
import kotlin.test.assertTrue
/**
* [Bech32.isDataChar] is the membership test the NIP-19 content scan uses to find where an encoded
* payload ends, so it has to agree with [Bech32.ALPHABET] exactly — a char wrongly accepted extends
* a payload past its real end and silently drops the entity when the decode then fails.
*/
class Bech32DataCharTest {
@Test
fun acceptsExactlyTheAlphabetInBothCases() {
for (c in Bech32.ALPHABET) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char")
for (c in Bech32.ALPHABET_UPPERCASE) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char")
}
@Test
fun rejectsTheAmbiguousFour() {
// BIP-173 leaves these out of the alphabet precisely because they are easy to misread.
for (c in "1bio1BIO") assertFalse(Bech32.isDataChar(c), "expected '$c' to be rejected")
}
@Test
fun agreesWithTheAlphabetAcrossEveryChar() {
// Sweeps the whole BMP so nothing outside the alphabet sneaks in — including the
// out-of-range guard for chars beyond the lookup table.
val expected = (Bech32.ALPHABET + Bech32.ALPHABET_UPPERCASE).toSet()
for (code in 0..0xFFFF) {
val c = code.toChar()
assertEquals(c in expected, Bech32.isDataChar(c), "disagreement at code $code")
}
}
@Test
fun countsMatchTheSpec() {
assertEquals(32, Bech32.ALPHABET.length)
// 32 symbols, but only the 23 letters have a distinct uppercase form — the 9 digits
// are the same char in both alphabets, so the accepted set is 23*2 + 9, not 64.
val letters = Bech32.ALPHABET.count { it.isLetter() }
val digits = Bech32.ALPHABET.count { it.isDigit() }
assertEquals(32, letters + digits)
assertEquals(letters * 2 + digits, (0..0xFFFF).count { Bech32.isDataChar(it.toChar()) })
}
}