diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt index 51a70d5331..748550ce50 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt @@ -22,19 +22,43 @@ package com.vitorpamplona.quartz.nip10Notes.content val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)") +/** + * Characters that end a hashtag: the punctuation class spelled out in [hashtagSearch], plus ASCII + * whitespace. Everything else continues the tag — including every non-ASCII character, since the + * regex's class is ASCII-only, so accented letters, CJK and emoji are all valid tag content. + */ +private val HASHTAG_TERMINATORS = + BooleanArray(128).apply { + for (c in 0x09..0x0D) this[c] = true + this[' '.code] = true + for (c in "!@#\u0024%^&*()=+./,[{]};:'\"?><") this[c.code] = true + } + +/** + * True while [c] can still be part of a hashtag. + * + * Non-ASCII always continues the tag: [hashtagSearch]'s excluded set is entirely ASCII and its + * `\s` is ASCII-only, so nothing above 0x7F was ever excluded. + */ +private fun isHashtagChar(c: Char): Boolean = c.code >= 128 || !HASHTAG_TERMINATORS[c.code] + /** * Collects the hashtags in [content]. * - * [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match - * starts either at position 0 or at a whitespace. That lets the scan jump between - * `#` occurrences with `indexOf` — an intrinsified char search — and apply the - * regex **anchored** at each, instead of letting `findAll` drive the regex engine - * from every position in the string. + * Jumps between `#` occurrences with `indexOf` — an intrinsified char search — and then matches + * `#` directly, character by character, rather than anchoring a regex there. * - * Measured on the production content distribution (median 529 B, tail to 767 KB): - * ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the - * content has no `#` at all (up to 300x). Equivalence with the previous `findAll` - * implementation is guarded by `RegexContentBenchmark` in `commons`. + * **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` -> + * `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory on every + * call. `Regex.matchAt` builds a fresh Matcher per call, so anchoring one at each candidate cost a + * full native UTF-16 copy of the note's content **per `#`** — the same defect that drove the app's + * native heap to ~1.9GB on a cold start via the NIP-19 scanner. This one is worse: measured over + * 2588 real notes it minted 43,626 Matchers copying 9.6GB in total, with a single 119KB note + * costing 279MB, because a whitespace-preceded `#` is far more common in prose than a NIP-19 + * prefix. The Java `Matcher` object is tiny, so Java-heap-driven GC had no reason to reclaim them + * promptly while each pinned native memory. + * + * [hashtagSearch] is kept as the specification the scan is tested against, not used here. */ fun findHashtags( content: String, @@ -44,19 +68,17 @@ fun findHashtags( var h = content.indexOf('#') while (h >= 0) { - if (h == 0 || content[h - 1].isWhitespace()) { - val match = - try { - hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1) - } catch (e: Exception) { - null - } - if (match != null) { - val tag = match.groups[1]?.value - if (tag != null && tag.isNotBlank()) { - output.add(tag) - } - h = content.indexOf('#', match.range.last + 1) + // `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character. + if (h == 0 || isAsciiRegexSpace(content[h - 1])) { + var end = h + 1 + while (end < content.length && isHashtagChar(content[end])) end++ + // The tag group is `+`, so it needs at least one character. + if (end > h + 1) { + val tag = content.substring(h + 1, end) + // Non-ASCII whitespace (U+00A0 and friends) is valid tag content to the regex but + // still blank to Kotlin, and the old code dropped those too. + if (tag.isNotBlank()) output.add(tag) + h = content.indexOf('#', end) continue } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt new file mode 100644 index 0000000000..dcf1138af0 --- /dev/null +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt @@ -0,0 +1,31 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip10Notes.content + +/** + * `\s` as `java.util.regex` applies it *without* `UNICODE_CHARACTER_CLASS`: space plus the five + * control characters `\t \n \x0B \f \r`, which are contiguous at 0x09..0x0D — and nothing else. + * + * The scanners in this package hand-roll grammars that used to be regexes, so they must use this + * rather than [Char.isWhitespace], which is Unicode-aware and would accept U+00A0, U+2003 and + * friends that the regexes rejected. + */ +internal fun isAsciiRegexSpace(c: Char): Boolean = c == ' ' || c.code in 0x09..0x0D diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt index cb6a2769a6..0ce6fed888 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt @@ -29,36 +29,36 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]") /** - * Walks every `#[n]` reference in [content]. + * Walks every `#[n]` reference in [content], handing each callback the digits between the brackets. * - * [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match - * starts at position 0 or at a whitespace. That lets the scan jump between `#` - * occurrences with `indexOf` — an intrinsified char search — and apply the regex - * **anchored** at each, instead of letting `findAll` drive the regex engine from - * every position in the string. + * Jumps between `#` occurrences with `indexOf` — an intrinsified char search — then matches + * `#[]` directly rather than anchoring a regex there. * - * Measured on the production content distribution (median 529 B, tail to 767 KB): - * ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense - * text. Equivalence with the previous `findAll` implementation (both callers) is - * guarded by `RegexContentBenchmark` in `commons`. + * **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` -> + * `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory per call, + * so anchoring a fresh Matcher at each candidate cost a full native UTF-16 copy of the content per + * `#`. See [findHashtags] for the measurements; this scanner shares the defect but never fires on + * real data, since `#[0]` is the legacy citation form that no current client emits. + * + * [tagSearch] is kept as the specification the scan is tested against, not used here. */ private inline fun forEachIndexTag( content: String, - action: (MatchResult) -> Unit, + action: (digits: String) -> Unit, ) { var h = content.indexOf('#') while (h >= 0) { - if (h == 0 || content[h - 1].isWhitespace()) { - val match = - try { - tagSearch.matchAt(content, if (h == 0) 0 else h - 1) - } catch (e: Exception) { - null + // `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character. + if (h == 0 || isAsciiRegexSpace(content[h - 1])) { + if (h + 1 < content.length && content[h + 1] == '[') { + var d = h + 2 + while (d < content.length && content[d] in '0'..'9') d++ + // `([0-9]+)` needs a digit, and the `]` must actually be there. + if (d > h + 2 && d < content.length && content[d] == ']') { + action(content.substring(h + 2, d)) + h = content.indexOf('#', d + 1) + continue } - if (match != null) { - action(match) - h = content.indexOf('#', match.range.last + 1) - continue } } h = content.indexOf('#', h + 1) @@ -73,10 +73,11 @@ fun findIndexTagsWithPeople( tags: TagArray, output: MutableSet = mutableSetOf(), ): List { - forEachIndexTag(content) { index -> + forEachIndexTag(content) { digits -> try { - val tag = index.groups[1]?.value?.let { tags[it.toInt()] } - if (tag != null && tag.size > 1 && tag[0] == "p") { + // Out-of-range indexes and non-numeric digits land in the catch below. + val tag = tags[digits.toInt()] + if (tag.size > 1 && tag[0] == "p") { output.add(tag[1]) } } catch (e: Exception) { @@ -94,13 +95,14 @@ fun findIndexTagsWithEventsOrAddresses( tags: TagArray, output: MutableSet = mutableSetOf(), ): Set { - forEachIndexTag(content) { index -> + forEachIndexTag(content) { digits -> try { - val tag = index.groups[1]?.value?.let { tags[it.toInt()] } - if (tag != null && tag.size > 1 && tag[0] == "e") { + // Out-of-range indexes and non-numeric digits land in the catch below. + val tag = tags[digits.toInt()] + if (tag.size > 1 && tag[0] == "e") { output.add(tag[1]) } - if (tag != null && tag.size > 1 && tag[0] == "a") { + if (tag.size > 1 && tag[0] == "a") { output.add(tag[1]) } } catch (e: Exception) { diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index 0d8fb3bb4d..c7e83091ac 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -24,6 +24,7 @@ import androidx.compose.runtime.Immutable import com.vitorpamplona.quartz.nip01Core.core.HexKey import com.vitorpamplona.quartz.nip01Core.core.hexToByteArray import com.vitorpamplona.quartz.nip01Core.core.toHexKey +import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32 import com.vitorpamplona.quartz.nip19Bech32.bech32.bechToBytes import com.vitorpamplona.quartz.nip19Bech32.entities.Entity import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress @@ -133,6 +134,67 @@ object Nip19Parser { fun hasAny(content: String): Boolean = nip19regex.matches(content) + // --------------------------------------------------------------------------------------- + // ICU-free scanner for the ingest hot path. + // + // `java.util.regex` on Android is backed by ICU, and `Matcher.region()` -> `reset()` -> + // `MatcherNative.setInput()` copies the ENTIRE input into native memory on every call. + // `Regex.matchAt` builds a fresh Matcher per call, and [forEachNip19Match] calls it once per + // candidate position — so scanning one note allocated a full native UTF-16 copy of its content + // *per candidate*, with content running to 767KB in the tail. Because the Java Matcher object + // is tiny, Java-heap-driven GC had no reason to reclaim them promptly, so the native heap grew + // unbounded: measured 45MB -> 1372MB over a cold start, ending in an lmkd kill at ~1.9GB RSS. + // Disabling this one scan made the native heap plateau at ~257MB instead. + // + // The grammar is small enough to match directly, so nothing here touches ICU. Case folding is + // done ASCII-only on purpose: `RegexOption.IGNORE_CASE` maps to `Pattern.CASE_INSENSITIVE`, + // which is ASCII-only unless `UNICODE_CASE` is also set. Using Kotlin's `ignoreCase = true` + // would be Unicode-aware and would accept inputs the regex rejected (U+212A KELVIN SIGN folding + // to `k`, say). + // --------------------------------------------------------------------------------------- + + /** Entities whose bech32 payload the regexes pin to exactly 58 chars. */ + private val FIXED_58_PREFIXES = arrayOf("nsec1", "npub1", "note1") + + /** Entities whose bech32 payload is `+` (one or more). */ + private val VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "nprofile1", "nrelay1", "nembed1") + + /** [nip19regexEvents] has no fixed-58 branch — there `note1` takes a variable payload. */ + private val EVENT_FIXED_58_PREFIXES = emptyArray() + + private val EVENT_VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "note1", "nrelay1", "nembed1") + + /** + * `\S` in the trailing group. Java's `\s` is the six ASCII whitespace chars unless + * `UNICODE_CHARACTER_CLASS` is set, so U+00A0 and friends count as NON-space here. + */ + private fun isRegexSpace(c: Char): Boolean = c == ' ' || c == '\t' || c == '\n' || c == '\u000B' || c == '\u000C' || c == '\r' + + /** ASCII-only case-insensitive prefix compare; [prefix] must be lowercase ASCII. */ + private fun matchesPrefixAt( + content: String, + offset: Int, + prefix: String, + ): Boolean { + if (offset + prefix.length > content.length) return false + for (k in prefix.indices) { + val c = content[offset + k] + val folded = if (c in 'A'..'Z') c + 32 else c + if (folded != prefix[k]) return false + } + return true + } + + /** End (exclusive) of the `[\S]*` run starting at [from]. */ + private fun endOfTrailing( + content: String, + from: Int, + ): Int { + var j = from + while (j < content.length && !isRegexSpace(content[j])) j++ + return j + } + /** * True when one of the NIP-19 entity prefixes starts at [i]. * @@ -165,26 +227,84 @@ object Nip19Parser { } /** - * Applies [regex] anchored at every NIP-19 candidate position in [content]. + * Matches `<[\S]*>` at every NIP-19 candidate position in [content], + * without ICU — see the block comment above [FIXED_58_PREFIXES] for why that matters. * - * [isCandidateAt] covers the union of the prefixes across the three NIP-19 - * regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does - * not accept — still far cheaper than `findAll` restarting the engine at - * every position in the string. + * [isCandidateAt] covers the union of the prefixes across the three NIP-19 regexes, so a + * caller passing a narrower prefix set simply finds no match at a prefix it does not accept. + * + * The prefixes are mutually exclusive (none is a prefix of another), so the first one that + * matches decides the branch, exactly as the regex alternation did. A fixed-58 entity with + * fewer than 58 payload chars fails outright rather than falling through to the variable + * branch, again matching the regex: no variable prefix can equal a fixed-58 one. */ private inline fun forEachNip19Match( content: String, - regex: Regex, - action: (MatchResult) -> Unit, + fixed58Prefixes: Array, + variablePrefixes: Array, + action: (type: String, key: String, additionalChars: String) -> Unit, ) { var i = 0 val len = content.length + // Jump between 'n'/'N' with indexOf — an intrinsified, vectorised char search — instead of + // testing every character. Both cases are tracked separately so each is only re-searched + // once consumed, amortising to about one indexOf per candidate. Measured ~2x faster on + // content with no entity at all, which is 94% of real notes (2588-note production sample), + // and ~26% faster on the 767KB mention-heavy tail. Small mention-dense content (a ~700B + // note with 2 mentions) is a few microseconds slower; that trade is deliberate. + var nextLower = content.indexOf('n') + var nextUpper = content.indexOf('N') while (i < len) { + if (nextLower in 0.. return + nextLower < 0 -> nextUpper + nextUpper < 0 -> nextLower + else -> if (nextLower < nextUpper) nextLower else nextUpper + } + if (i >= len) return if (isCandidateAt(content, i)) { - val match = regex.matchAt(content, i) - if (match != null) { - action(match) - i = match.range.last + 1 + var end = -1 + + for (prefix in fixed58Prefixes) { + if (!matchesPrefixAt(content, i, prefix)) continue + val dataStart = i + prefix.length + val dataEnd = dataStart + 58 + if (dataEnd <= len && allBech32(content, dataStart, dataEnd)) { + end = endOfTrailing(content, dataEnd) + action( + content.substring(i, dataStart), + content.substring(dataStart, dataEnd), + content.substring(dataEnd, end), + ) + } + break + } + + if (end < 0) { + for (prefix in variablePrefixes) { + if (!matchesPrefixAt(content, i, prefix)) continue + val dataStart = i + prefix.length + var dataEnd = dataStart + while (dataEnd < len && Bech32.isDataChar(content[dataEnd])) dataEnd++ + // `+` needs at least one payload char. + if (dataEnd > dataStart) { + end = endOfTrailing(content, dataEnd) + action( + content.substring(i, dataStart), + content.substring(dataStart, dataEnd), + content.substring(dataEnd, end), + ) + } + break + } + } + + // `end` is exclusive and always past `i`, so the scan still advances. + if (end >= 0) { + i = end continue } } @@ -192,6 +312,17 @@ object Nip19Parser { } } + private fun allBech32( + content: String, + from: Int, + to: Int, + ): Boolean { + for (k in from until to) { + if (!Bech32.isDataChar(content[k])) return false + } + return true + } + /** * Scans [content] for NIP-19 entities. * @@ -208,14 +339,8 @@ object Nip19Parser { */ fun parseAll(content: String): List { val returningList = mutableListOf() - forEachNip19Match(content, nip19regex) { matcher -> - val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 - val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 - val additionalChars = matcher.groups[7]?.value // additional chars - - if (type != null) { - parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } - } + forEachNip19Match(content, FIXED_58_PREFIXES, VARIABLE_PREFIXES) { type, key, additionalChars -> + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } return returningList } @@ -223,14 +348,8 @@ object Nip19Parser { /** Same scan as [parseAll], restricted to the event-ish entities. */ fun parseAllEvents(content: String): List { val returningList = mutableListOf() - forEachNip19Match(content, nip19regexEvents) { matcher -> - val type = matcher.groups[2]?.value // nevent1 - val key = matcher.groups[3]?.value // bech32 - val additionalChars = matcher.groups[4]?.value // additional chars - - if (type != null) { - parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } - } + forEachNip19Match(content, EVENT_FIXED_58_PREFIXES, EVENT_VARIABLE_PREFIXES) { type, key, additionalChars -> + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } return returningList } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt index 2cc561e561..7d2a2b6541 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt @@ -20,6 +20,8 @@ */ package com.vitorpamplona.quartz.nip19Bech32.bech32 +import kotlin.jvm.JvmField + /* * Copyright 2020 ACINQ SAS * @@ -81,15 +83,50 @@ object Bech32 { // char -> 5 bits value private val map = Array(255) { -1 } + @PublishedApi + internal const val DATA_CHARS_SIZE = 128 + + /** + * Membership table for [isDataChar], kept separate from [map] because [map] is an + * `Array` — a boxed `java.lang.Byte[]` — and [isDataChar] runs per character over whole + * note contents (up to ~767KB), where unboxing on every char would show up. + * + * `@JvmField` so callers of the inline [isDataChar] compile to a direct `getstatic` instead of + * a property-getter `invokevirtual` per character (the same reasoning as `PointTypes`). + * `@PublishedApi internal` because a public inline function cannot touch a private member. + */ + @PublishedApi + @JvmField + internal val DATA_CHARS = BooleanArray(DATA_CHARS_SIZE) + init { for (i in 0..ALPHABET.lastIndex) { map[ALPHABET[i].code] = i.toByte() + DATA_CHARS[ALPHABET[i].code] = true } for (i in 0..ALPHABET_UPPERCASE.lastIndex) { map[ALPHABET_UPPERCASE[i].code] = i.toByte() + DATA_CHARS[ALPHABET_UPPERCASE[i].code] = true } } + /** + * True when [c] is part of the bech32 data alphabet, in either case — i.e. everything except + * `1`, `b`, `i` and `o`, which BIP-173 excludes as visually ambiguous. + * + * Exposed so scanners can find where an encoded payload *ends* without decoding it. The NIP-19 + * content scan needs exactly that on every ingested event, and cannot use a regex to do it: + * Android's `java.util.regex` is ICU-backed and `Matcher.region()` copies the entire input into + * native memory per call, which drove the app's native heap to ~1.9GB on a cold start. + * + * `inline` because it is called per character over whole note contents (up to ~767KB). As a + * normal member it compiled to an `invokevirtual` per char, which measured ~1-2% slower across + * the scan benchmark than the equivalent private lookup it replaced; inlining puts the constant + * compare and the `baload` straight into the caller's loop and closes that gap. + */ + @Suppress("NOTHING_TO_INLINE") + inline fun isDataChar(c: Char): Boolean = c.code < DATA_CHARS_SIZE && DATA_CHARS[c.code] + fun expand(hrp: String): Array { val half = hrp.length + 1 val size = half + hrp.length diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt new file mode 100644 index 0000000000..4bbf310bf6 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt @@ -0,0 +1,167 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip10Notes.content + +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * Pins the ICU-free content scanners to the regexes they replaced. + * + * [findHashtags] and the `#[n]` walker used to anchor a fresh `Regex.matchAt` at every candidate. + * On Android that goes through ICU, where `Matcher.region()` copies the whole input into native + * memory per call — over 2588 real notes that was 43,626 Matchers copying 9.6GB, worst single note + * 279MB. The scanners now match the grammars directly, so [hashtagSearch] and [tagSearch] survive + * only as the specification, and this asserts the two agree. + * + * The corpus targets where a hand-rolled matcher is most likely to drift: the exact punctuation + * set that ends a tag, ASCII-vs-Unicode whitespace before the `#`, non-ASCII inside the tag, and + * the `+`/`[0-9]+` minimum-one-character rules. + */ +class ContentScanRegexEquivalenceTest { + private fun referenceHashtags(content: String): List { + if (content.isBlank()) return emptyList() + val out = mutableSetOf() + hashtagSearch.findAll(content).forEach { m -> + val tag = m.groups[1]?.value + if (tag != null && tag.isNotBlank()) out.add(tag) + } + return out.toList() + } + + private fun referenceIndexTags( + content: String, + tags: Array>, + wanted: String, + ): Set { + val out = mutableSetOf() + tagSearch.findAll(content).forEach { m -> + try { + val tag = m.groups[1]?.value?.let { tags[it.toInt()] } + if (tag != null && tag.size > 1 && tag[0] == wanted) out.add(tag[1]) + } catch (e: Exception) { + } + } + return out + } + + private val tagArray = + arrayOf( + arrayOf("p", "pubkey0"), + arrayOf("e", "event1"), + arrayOf("a", "addr2"), + arrayOf("p", "pubkey3"), + arrayOf("t", "topic4"), + ) + + private fun corpus(): List = + buildList { + add("") + add(" ") + add("#") + add("#tag") + add("hello #tag world") + add("a#tag") + add("#tag#other") + add("#tag #other") + add("##tag") + add("#tag.") + add("#tag, and #more!") + add("#tag's") + add("#tag\"quoted\"") + add("#a") + add("#1") + add("#tag-with-dash") + add("#tag_with_underscore") + add("#tag~tilde|pipe\\back`tick") + // non-ASCII is valid tag content: the regex class and its \s are ASCII-only + add("#café") + add("#日本語") + add("#tagéè") + // ASCII vs Unicode whitespace BEFORE the # decides whether it matches at all + add("x\u00A0#tag") + add("x\u2003#tag") + add("x\t#tag") + add("x\n#tag") + add("x\r#tag") + // Unicode whitespace INSIDE the tag is valid to the regex but blank to Kotlin + add("#\u00A0") + add("#\u00A0x") + // every excluded char must terminate the tag + for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#tag${c}more") + for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#$c") + // index tags + add("#[0]") + add("#[1] and #[2]") + add("look #[3] here") + add("#[]") + add("#[abc]") + add("#[99]") + add("#[0") + add("#[0]]") + add("x#[0]") + add("x\u00A0#[0]") + add("#[0]#[1]") + add("#[00]") + // mixed + add("#tag #[0] #other #[1]") + add("lorem ipsum ".repeat(500) + "#tail") + add("#head" + " dolor sit ".repeat(500)) + add("no hashes at all here ".repeat(200)) + } + + @Test + fun hashtagsMatchRegex() { + corpus().forEach { c -> + assertEquals( + referenceHashtags(c).sorted(), + findHashtags(c).sorted(), + "findHashtags diverged on: ${c.take(80)}", + ) + } + } + + @Test + fun indexTagsMatchRegex() { + corpus().forEach { c -> + assertEquals( + referenceIndexTags(c, tagArray, "p").sorted(), + findIndexTagsWithPeople(c, tagArray).sorted(), + "findIndexTagsWithPeople diverged on: ${c.take(80)}", + ) + val refEv = referenceIndexTags(c, tagArray, "e") + referenceIndexTags(c, tagArray, "a") + assertEquals( + refEv.sorted(), + findIndexTagsWithEventsOrAddresses(c, tagArray).sorted(), + "findIndexTagsWithEventsOrAddresses diverged on: ${c.take(80)}", + ) + } + } + + @Test + fun hashtagsMatchRegexAtEveryOffset() { + val filler = "a b\tc\nd " + for (i in 0..filler.length) { + val c = filler.substring(0, i) + "#tag" + filler.substring(i) + assertEquals(referenceHashtags(c).sorted(), findHashtags(c).sorted(), "offset $i") + } + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt new file mode 100644 index 0000000000..2799195579 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt @@ -0,0 +1,202 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32 + +import com.vitorpamplona.quartz.nip19Bech32.entities.Entity +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * Pins the ICU-free scanner in [Nip19Parser] to the regexes it replaced. + * + * The scan used to run `Regex.matchAt` at every candidate position. On Android that goes through + * ICU, and `Matcher.region()` copies the whole input into native memory on each call — one full + * native copy of a note's content *per candidate* — which drove the native heap to ~1.9GB on a + * cold start and got the process lmkd-killed. The scanner matches the grammar directly instead. + * + * [Nip19Parser.nip19regex] and [Nip19Parser.nip19regexEvents] are still the specification, so this + * runs both over the same corpus and requires identical entity lists. The corpus deliberately + * targets the places a hand-rolled matcher is most likely to drift from the regex: the exact-58 + * payload boundary, the bech32 alphabet's excluded characters, ASCII-only case folding, and which + * characters `[\S]*` is willing to swallow. + */ +class Nip19ScannerRegexEquivalenceTest { + companion object { + const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y" + const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv" + const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy" + + /** 58 valid bech32 chars, so `npub1` + this is exactly the fixed-length branch. */ + const val PAYLOAD58 = "qpzry9x8gf2tvdw0s3jn54khce6mua7lqpzry9x8gf2tvdw0s3jn54khce" + } + + /** What `parseAll` did before: drive the regex from every position with `findAll`. */ + private fun referenceParseAll(content: String): List { + val out = mutableListOf() + Nip19Parser.nip19regex.findAll(content).forEach { m -> + val type = m.groups[3]?.value ?: m.groups[5]?.value + val key = m.groups[4]?.value ?: m.groups[6]?.value + val additionalChars = m.groups[7]?.value + if (type != null) { + Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) } + } + } + return out + } + + private fun referenceParseAllEvents(content: String): List { + val out = mutableListOf() + Nip19Parser.nip19regexEvents.findAll(content).forEach { m -> + val type = m.groups[2]?.value + val key = m.groups[3]?.value + val additionalChars = m.groups[4]?.value + if (type != null) { + Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) } + } + } + return out + } + + private fun assertSameAsRegex(content: String) { + assertEquals( + referenceParseAll(content), + Nip19Parser.parseAll(content), + "parseAll diverged from nip19regex on: ${content.take(90)}", + ) + assertEquals( + referenceParseAllEvents(content), + Nip19Parser.parseAllEvents(content), + "parseAllEvents diverged from nip19regexEvents on: ${content.take(90)}", + ) + } + + private fun corpus(): List = + buildList { + // plain placement + add("") + add(NPUB) + add("hello $NPUB world") + add("nostr:$NPUB") + add("@$NPUB") + add("nostr:@$NPUB") + add("prefix-nostr:$NPUB-suffix") + add(NOTE) + add(NEVENT) + + // adjacency and repetition — where scan-resume position matters + add(NPUB + NEVENT) + add("$NPUB $NEVENT") + add("$NPUB\n$NEVENT") + add("$NPUB,$NEVENT") + add(listOf(NPUB, NOTE, NEVENT).joinToString(" ")) + add(NPUB.repeat(3)) + + // the exact-58 boundary for npub/nsec/note + add("npub1" + PAYLOAD58) + add("npub1" + PAYLOAD58.dropLast(1)) // 57 -> must not match + add("npub1" + PAYLOAD58 + "q") // 59 -> 58 key, trailing takes the rest + add("npub1" + PAYLOAD58 + " tail") + add("note1" + PAYLOAD58) + add("nsec1" + PAYLOAD58) + + // bech32 alphabet: 1, b, i, o are excluded and must terminate the payload + add("nevent1qqs1qqs") + add("nevent1qqsbqqs") + add("nevent1qqsiqqs") + add("nevent1qqsoqqs") + + // A *valid* variable-length entity butted straight against an excluded char. The + // payload has to stop there and still decode. These are the cases with teeth: a + // charset that wrongly accepted b/i/o/1 would swallow the extra char, fail the + // bech32 decode and silently drop the entity — whereas cases whose payload is + // invalid either way agree trivially and prove nothing. + for (excluded in listOf("b", "i", "o", "1")) { + add(NEVENT + excluded) + add(NEVENT + excluded + "xyz") + add("$NEVENT$excluded more text") + } + add("nevent1") // variable branch needs >= 1 payload char + add("nprofile1") + add("naddr1q") + + // ASCII-only case folding + add(NPUB.uppercase()) + add("NOSTR:" + NPUB.uppercase()) + add(NEVENT.uppercase()) + add("nPuB1" + PAYLOAD58) + // U+212A KELVIN SIGN folds to 'k' under Unicode rules but NOT under the regex's + // ASCII-only CASE_INSENSITIVE; both sides must reject it. + add("npub1" + PAYLOAD58.replaceFirst("k", "K")) + + // what [\S]* may swallow: Java's \s is the six ASCII whitespace chars only, + // so U+00A0 and U+2003 are NON-space and belong to the trailing group. + add("$NPUB\u00A0more") + add("$NPUB\u2003more") + add("${NPUB}more") + add("${NPUB}1more") + add("$NPUB\tmore") + add("$NPUB\rmore") + + // near-misses that must not be mistaken for entities + add("n") + add("nn") + add("np") + add("no") + add("nostr:") + add("nothing to see here") + add("a note about nothing") + add("nopqrstuvwxyz") + add("x$NPUB") + add("1$NPUB") + + // long content with the entity at the far end (the 767KB-tail shape, scaled down) + add("lorem ipsum ".repeat(2000) + NPUB) + add(NPUB + " " + "dolor sit amet ".repeat(2000)) + // many 'n' candidates but no entities — the scanner's rejection path + add("neither nor none never nothing ".repeat(500)) + } + + @Test + fun matchesRegexAcrossCorpus() { + corpus().forEach { assertSameAsRegex(it) } + } + + @Test + fun matchesRegexWithEntityAtEveryOffset() { + // Slides the entity through a filler string so every start offset, including + // immediately after another candidate 'n', is exercised. + val filler = "n no non nost nostr " + for (i in 0..filler.length) { + assertSameAsRegex(filler.substring(0, i) + NPUB + filler.substring(i)) + } + } + + @Test + fun matchesRegexOnTruncatedPayloads() { + // Every truncation of a real entity: catches off-by-one at the 58 boundary and in `+`. + for (entity in listOf(NPUB, NOTE, NEVENT)) { + for (len in 1..entity.length) { + assertSameAsRegex(entity.substring(0, len)) + assertSameAsRegex("text " + entity.substring(0, len) + " text") + } + } + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt new file mode 100644 index 0000000000..dd2cd46816 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt @@ -0,0 +1,67 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32.bech32 + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertFalse +import kotlin.test.assertTrue + +/** + * [Bech32.isDataChar] is the membership test the NIP-19 content scan uses to find where an encoded + * payload ends, so it has to agree with [Bech32.ALPHABET] exactly — a char wrongly accepted extends + * a payload past its real end and silently drops the entity when the decode then fails. + */ +class Bech32DataCharTest { + @Test + fun acceptsExactlyTheAlphabetInBothCases() { + for (c in Bech32.ALPHABET) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char") + for (c in Bech32.ALPHABET_UPPERCASE) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char") + } + + @Test + fun rejectsTheAmbiguousFour() { + // BIP-173 leaves these out of the alphabet precisely because they are easy to misread. + for (c in "1bio1BIO") assertFalse(Bech32.isDataChar(c), "expected '$c' to be rejected") + } + + @Test + fun agreesWithTheAlphabetAcrossEveryChar() { + // Sweeps the whole BMP so nothing outside the alphabet sneaks in — including the + // out-of-range guard for chars beyond the lookup table. + val expected = (Bech32.ALPHABET + Bech32.ALPHABET_UPPERCASE).toSet() + for (code in 0..0xFFFF) { + val c = code.toChar() + assertEquals(c in expected, Bech32.isDataChar(c), "disagreement at code $code") + } + } + + @Test + fun countsMatchTheSpec() { + assertEquals(32, Bech32.ALPHABET.length) + // 32 symbols, but only the 23 letters have a distinct uppercase form — the 9 digits + // are the same char in both alphabets, so the accepted set is 23*2 + 9, not 64. + val letters = Bech32.ALPHABET.count { it.isLetter() } + val digits = Bech32.ALPHABET.count { it.isDigit() } + assertEquals(32, letters + digits) + assertEquals(letters * 2 + digits, (0..0xFFFF).count { Bech32.isDataChar(it.toChar()) }) + } +}