From d13a54d8320015fdbe47903c43ac6a3e1759da3d Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Mon, 17 Aug 2026 17:30:43 -0400 Subject: [PATCH 1/4] perf: scan NIP-19 entities without ICU to stop the cold-start native-heap OOM On a 2.9GB SM-T220 the release build grew to ~1.9GB RSS during a cold start and was lmkd-killed ~32s in ("to free 1871628kB rss, 375492kB swap"). The growth was entirely in the NATIVE heap -- the Java heap plateaued at its 512MB largeHeap ceiling and GCed back down, while native ran 45MB -> 1372MB. Cause: `forEachNip19Match` called `Regex.matchAt` once per candidate position. On Android `java.util.regex` is ICU-backed, and `Regex.matchAt` builds a fresh Matcher whose `region()` -> `reset()` -> `MatcherNative.setInput()` copies the ENTIRE input into native memory. So scanning one note allocated a full native UTF-16 copy of its content *per candidate `n`*, and note content reaches 767KB in the tail. The Java Matcher object is tiny, so Java-heap-driven GC had no reason to reclaim them promptly and native memory grew unbounded. heapprofd's top malloc stack was exactly this path, under LocalCache.justConsume -> updateHintIndexes. The file's own KDoc had already recorded the symptom from an earlier pass -- "2,541 of 4,573 live Matchers were running this regex" -- but that pass optimized speed (the 9-23x anchoring win) and left the native retention in place. The grammar is prefix + bech32 payload + trailing non-space, so it is matched directly with char compares instead. Case folding is deliberately ASCII-only: RegexOption.IGNORE_CASE maps to Pattern.CASE_INSENSITIVE, which is ASCII-only unless UNICODE_CASE is set, so Kotlin's Unicode-aware `ignoreCase = true` would have accepted inputs the regex rejected (U+212A folding to 'k'). Reusing a single Matcher would NOT have fixed this: region() re-copies the input on every call. Only the ingest hot path changes. `uriToRoute`/`tryParseAndClean`/`hasAny` still use the regexes -- they run on short user input, not per ingested event. Measured on device (release codegen, 3 runs), native heap RSS: t~9s t~14s t~22s t~28s t~43s before 45M 497M 626M 1372M (killed at 31.9s) after 48M 127M 170M 175M 172M Native now plateaus at ~170MB, total RSS falls back to ~500MB instead of climbing to 1.88GB, and the process survives past 45s with zero lmkd kills. Equivalence is pinned by a new test that runs both original regexes over the same corpus and requires identical entity lists, targeting the exact-58 boundary, the excluded bech32 chars, ASCII-only case folding and what `[\S]*` swallows. Both mutations tried against it (58 -> 57, and admitting 'b' into the alphabet) fail the test. The pre-existing `Nip19ScanTest` (23 tests) and commons' `nip19MatchesReferenceScan` guard also still pass; full quartz suite 4288/0. Co-Authored-By: Claude Opus 5 (1M context) --- .../quartz/nip19Bech32/Nip19Parser.kt | 165 +++++++++++--- .../Nip19ScannerRegexEquivalenceTest.kt | 202 ++++++++++++++++++ 2 files changed, 340 insertions(+), 27 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index 0d8fb3bb4d..b78ddfaf4b 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -133,6 +133,78 @@ object Nip19Parser { fun hasAny(content: String): Boolean = nip19regex.matches(content) + // --------------------------------------------------------------------------------------- + // ICU-free scanner for the ingest hot path. + // + // `java.util.regex` on Android is backed by ICU, and `Matcher.region()` -> `reset()` -> + // `MatcherNative.setInput()` copies the ENTIRE input into native memory on every call. + // `Regex.matchAt` builds a fresh Matcher per call, and [forEachNip19Match] calls it once per + // candidate position — so scanning one note allocated a full native UTF-16 copy of its content + // *per candidate*, with content running to 767KB in the tail. Because the Java Matcher object + // is tiny, Java-heap-driven GC had no reason to reclaim them promptly, so the native heap grew + // unbounded: measured 45MB -> 1372MB over a cold start, ending in an lmkd kill at ~1.9GB RSS. + // Disabling this one scan made the native heap plateau at ~257MB instead. + // + // The grammar is small enough to match directly, so nothing here touches ICU. Case folding is + // done ASCII-only on purpose: `RegexOption.IGNORE_CASE` maps to `Pattern.CASE_INSENSITIVE`, + // which is ASCII-only unless `UNICODE_CASE` is also set. Using Kotlin's `ignoreCase = true` + // would be Unicode-aware and would accept inputs the regex rejected (U+212A KELVIN SIGN folding + // to `k`, say). + // --------------------------------------------------------------------------------------- + + /** Entities whose bech32 payload the regexes pin to exactly 58 chars. */ + private val FIXED_58_PREFIXES = arrayOf("nsec1", "npub1", "note1") + + /** Entities whose bech32 payload is `+` (one or more). */ + private val VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "nprofile1", "nrelay1", "nembed1") + + /** [nip19regexEvents] has no fixed-58 branch — there `note1` takes a variable payload. */ + private val EVENT_FIXED_58_PREFIXES = emptyArray() + + private val EVENT_VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "note1", "nrelay1", "nembed1") + + /** bech32: digits except `1`, letters except `b`, `i`, `o`. ASCII-only, both cases. */ + private val BECH32_CHARS = + BooleanArray(128).apply { + for (c in "qpzry9x8gf2tvdw0s3jn54khce6mua7l") { + this[c.code] = true + if (c in 'a'..'z') this[c.code - 32] = true + } + } + + private fun isBech32(c: Char): Boolean = c.code < 128 && BECH32_CHARS[c.code] + + /** + * `\S` in the trailing group. Java's `\s` is the six ASCII whitespace chars unless + * `UNICODE_CHARACTER_CLASS` is set, so U+00A0 and friends count as NON-space here. + */ + private fun isRegexSpace(c: Char): Boolean = c == ' ' || c == '\t' || c == '\n' || c == '\u000B' || c == '\u000C' || c == '\r' + + /** ASCII-only case-insensitive prefix compare; [prefix] must be lowercase ASCII. */ + private fun matchesPrefixAt( + content: String, + offset: Int, + prefix: String, + ): Boolean { + if (offset + prefix.length > content.length) return false + for (k in prefix.indices) { + val c = content[offset + k] + val folded = if (c in 'A'..'Z') c + 32 else c + if (folded != prefix[k]) return false + } + return true + } + + /** End (exclusive) of the `[\S]*` run starting at [from]. */ + private fun endOfTrailing( + content: String, + from: Int, + ): Int { + var j = from + while (j < content.length && !isRegexSpace(content[j])) j++ + return j + } + /** * True when one of the NIP-19 entity prefixes starts at [i]. * @@ -165,26 +237,66 @@ object Nip19Parser { } /** - * Applies [regex] anchored at every NIP-19 candidate position in [content]. + * Matches `<[\S]*>` at every NIP-19 candidate position in [content], + * without ICU — see the block comment above [FIXED_58_PREFIXES] for why that matters. * - * [isCandidateAt] covers the union of the prefixes across the three NIP-19 - * regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does - * not accept — still far cheaper than `findAll` restarting the engine at - * every position in the string. + * [isCandidateAt] covers the union of the prefixes across the three NIP-19 regexes, so a + * caller passing a narrower prefix set simply finds no match at a prefix it does not accept. + * + * The prefixes are mutually exclusive (none is a prefix of another), so the first one that + * matches decides the branch, exactly as the regex alternation did. A fixed-58 entity with + * fewer than 58 payload chars fails outright rather than falling through to the variable + * branch, again matching the regex: no variable prefix can equal a fixed-58 one. */ private inline fun forEachNip19Match( content: String, - regex: Regex, - action: (MatchResult) -> Unit, + fixed58Prefixes: Array, + variablePrefixes: Array, + action: (type: String, key: String, additionalChars: String) -> Unit, ) { var i = 0 val len = content.length while (i < len) { if (isCandidateAt(content, i)) { - val match = regex.matchAt(content, i) - if (match != null) { - action(match) - i = match.range.last + 1 + var end = -1 + + for (prefix in fixed58Prefixes) { + if (!matchesPrefixAt(content, i, prefix)) continue + val dataStart = i + prefix.length + val dataEnd = dataStart + 58 + if (dataEnd <= len && allBech32(content, dataStart, dataEnd)) { + end = endOfTrailing(content, dataEnd) + action( + content.substring(i, dataStart), + content.substring(dataStart, dataEnd), + content.substring(dataEnd, end), + ) + } + break + } + + if (end < 0) { + for (prefix in variablePrefixes) { + if (!matchesPrefixAt(content, i, prefix)) continue + val dataStart = i + prefix.length + var dataEnd = dataStart + while (dataEnd < len && isBech32(content[dataEnd])) dataEnd++ + // `+` needs at least one payload char. + if (dataEnd > dataStart) { + end = endOfTrailing(content, dataEnd) + action( + content.substring(i, dataStart), + content.substring(dataStart, dataEnd), + content.substring(dataEnd, end), + ) + } + break + } + } + + // `end` is exclusive and always past `i`, so the scan still advances. + if (end >= 0) { + i = end continue } } @@ -192,6 +304,17 @@ object Nip19Parser { } } + private fun allBech32( + content: String, + from: Int, + to: Int, + ): Boolean { + for (k in from until to) { + if (!isBech32(content[k])) return false + } + return true + } + /** * Scans [content] for NIP-19 entities. * @@ -208,14 +331,8 @@ object Nip19Parser { */ fun parseAll(content: String): List { val returningList = mutableListOf() - forEachNip19Match(content, nip19regex) { matcher -> - val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1 - val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32 - val additionalChars = matcher.groups[7]?.value // additional chars - - if (type != null) { - parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } - } + forEachNip19Match(content, FIXED_58_PREFIXES, VARIABLE_PREFIXES) { type, key, additionalChars -> + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } return returningList } @@ -223,14 +340,8 @@ object Nip19Parser { /** Same scan as [parseAll], restricted to the event-ish entities. */ fun parseAllEvents(content: String): List { val returningList = mutableListOf() - forEachNip19Match(content, nip19regexEvents) { matcher -> - val type = matcher.groups[2]?.value // nevent1 - val key = matcher.groups[3]?.value // bech32 - val additionalChars = matcher.groups[4]?.value // additional chars - - if (type != null) { - parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } - } + forEachNip19Match(content, EVENT_FIXED_58_PREFIXES, EVENT_VARIABLE_PREFIXES) { type, key, additionalChars -> + parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) } } return returningList } diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt new file mode 100644 index 0000000000..2799195579 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19ScannerRegexEquivalenceTest.kt @@ -0,0 +1,202 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32 + +import com.vitorpamplona.quartz.nip19Bech32.entities.Entity +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * Pins the ICU-free scanner in [Nip19Parser] to the regexes it replaced. + * + * The scan used to run `Regex.matchAt` at every candidate position. On Android that goes through + * ICU, and `Matcher.region()` copies the whole input into native memory on each call — one full + * native copy of a note's content *per candidate* — which drove the native heap to ~1.9GB on a + * cold start and got the process lmkd-killed. The scanner matches the grammar directly instead. + * + * [Nip19Parser.nip19regex] and [Nip19Parser.nip19regexEvents] are still the specification, so this + * runs both over the same corpus and requires identical entity lists. The corpus deliberately + * targets the places a hand-rolled matcher is most likely to drift from the regex: the exact-58 + * payload boundary, the bech32 alphabet's excluded characters, ASCII-only case folding, and which + * characters `[\S]*` is willing to swallow. + */ +class Nip19ScannerRegexEquivalenceTest { + companion object { + const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y" + const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv" + const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy" + + /** 58 valid bech32 chars, so `npub1` + this is exactly the fixed-length branch. */ + const val PAYLOAD58 = "qpzry9x8gf2tvdw0s3jn54khce6mua7lqpzry9x8gf2tvdw0s3jn54khce" + } + + /** What `parseAll` did before: drive the regex from every position with `findAll`. */ + private fun referenceParseAll(content: String): List { + val out = mutableListOf() + Nip19Parser.nip19regex.findAll(content).forEach { m -> + val type = m.groups[3]?.value ?: m.groups[5]?.value + val key = m.groups[4]?.value ?: m.groups[6]?.value + val additionalChars = m.groups[7]?.value + if (type != null) { + Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) } + } + } + return out + } + + private fun referenceParseAllEvents(content: String): List { + val out = mutableListOf() + Nip19Parser.nip19regexEvents.findAll(content).forEach { m -> + val type = m.groups[2]?.value + val key = m.groups[3]?.value + val additionalChars = m.groups[4]?.value + if (type != null) { + Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) } + } + } + return out + } + + private fun assertSameAsRegex(content: String) { + assertEquals( + referenceParseAll(content), + Nip19Parser.parseAll(content), + "parseAll diverged from nip19regex on: ${content.take(90)}", + ) + assertEquals( + referenceParseAllEvents(content), + Nip19Parser.parseAllEvents(content), + "parseAllEvents diverged from nip19regexEvents on: ${content.take(90)}", + ) + } + + private fun corpus(): List = + buildList { + // plain placement + add("") + add(NPUB) + add("hello $NPUB world") + add("nostr:$NPUB") + add("@$NPUB") + add("nostr:@$NPUB") + add("prefix-nostr:$NPUB-suffix") + add(NOTE) + add(NEVENT) + + // adjacency and repetition — where scan-resume position matters + add(NPUB + NEVENT) + add("$NPUB $NEVENT") + add("$NPUB\n$NEVENT") + add("$NPUB,$NEVENT") + add(listOf(NPUB, NOTE, NEVENT).joinToString(" ")) + add(NPUB.repeat(3)) + + // the exact-58 boundary for npub/nsec/note + add("npub1" + PAYLOAD58) + add("npub1" + PAYLOAD58.dropLast(1)) // 57 -> must not match + add("npub1" + PAYLOAD58 + "q") // 59 -> 58 key, trailing takes the rest + add("npub1" + PAYLOAD58 + " tail") + add("note1" + PAYLOAD58) + add("nsec1" + PAYLOAD58) + + // bech32 alphabet: 1, b, i, o are excluded and must terminate the payload + add("nevent1qqs1qqs") + add("nevent1qqsbqqs") + add("nevent1qqsiqqs") + add("nevent1qqsoqqs") + + // A *valid* variable-length entity butted straight against an excluded char. The + // payload has to stop there and still decode. These are the cases with teeth: a + // charset that wrongly accepted b/i/o/1 would swallow the extra char, fail the + // bech32 decode and silently drop the entity — whereas cases whose payload is + // invalid either way agree trivially and prove nothing. + for (excluded in listOf("b", "i", "o", "1")) { + add(NEVENT + excluded) + add(NEVENT + excluded + "xyz") + add("$NEVENT$excluded more text") + } + add("nevent1") // variable branch needs >= 1 payload char + add("nprofile1") + add("naddr1q") + + // ASCII-only case folding + add(NPUB.uppercase()) + add("NOSTR:" + NPUB.uppercase()) + add(NEVENT.uppercase()) + add("nPuB1" + PAYLOAD58) + // U+212A KELVIN SIGN folds to 'k' under Unicode rules but NOT under the regex's + // ASCII-only CASE_INSENSITIVE; both sides must reject it. + add("npub1" + PAYLOAD58.replaceFirst("k", "K")) + + // what [\S]* may swallow: Java's \s is the six ASCII whitespace chars only, + // so U+00A0 and U+2003 are NON-space and belong to the trailing group. + add("$NPUB\u00A0more") + add("$NPUB\u2003more") + add("${NPUB}more") + add("${NPUB}1more") + add("$NPUB\tmore") + add("$NPUB\rmore") + + // near-misses that must not be mistaken for entities + add("n") + add("nn") + add("np") + add("no") + add("nostr:") + add("nothing to see here") + add("a note about nothing") + add("nopqrstuvwxyz") + add("x$NPUB") + add("1$NPUB") + + // long content with the entity at the far end (the 767KB-tail shape, scaled down) + add("lorem ipsum ".repeat(2000) + NPUB) + add(NPUB + " " + "dolor sit amet ".repeat(2000)) + // many 'n' candidates but no entities — the scanner's rejection path + add("neither nor none never nothing ".repeat(500)) + } + + @Test + fun matchesRegexAcrossCorpus() { + corpus().forEach { assertSameAsRegex(it) } + } + + @Test + fun matchesRegexWithEntityAtEveryOffset() { + // Slides the entity through a filler string so every start offset, including + // immediately after another candidate 'n', is exercised. + val filler = "n no non nost nostr " + for (i in 0..filler.length) { + assertSameAsRegex(filler.substring(0, i) + NPUB + filler.substring(i)) + } + } + + @Test + fun matchesRegexOnTruncatedPayloads() { + // Every truncation of a real entity: catches off-by-one at the 58 boundary and in `+`. + for (entity in listOf(NPUB, NOTE, NEVENT)) { + for (len in 1..entity.length) { + assertSameAsRegex(entity.substring(0, len)) + assertSameAsRegex("text " + entity.substring(0, len) + " text") + } + } + } +} From c5a6aa2090e049363a47561d76d079900819596f Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Mon, 17 Aug 2026 18:13:13 -0400 Subject: [PATCH 2/4] refactor: move the bech32 alphabet test into Bech32 as isDataChar MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The scanner added in the previous commit carried its own copy of the bech32 alphabet, duplicating `Bech32.ALPHABET`. "Is this a bech32 data character" is the codec's own question, so it belongs on `Bech32` next to the alphabet it derives from -- the same way `Hex` owns its parsing helpers. `Bech32.map` could not be reused for it: it is an `Array`, i.e. a boxed `java.lang.Byte[]`, and this runs per character over whole note contents (up to ~767KB), so it would unbox on every char. `isDataChar` gets its own primitive `BooleanArray` built from the same ALPHABET constants in the existing init block, so there is still one source of truth. Kept at parity with the private lookup it replaced, checked in the bytecode: - as a plain member it compiled to an `invokevirtual` per character and measured ~1-2% slower across the scan benchmark, consistently signed across 8 of 10 cases - `inline` removed that call, but property access to the table then compiled to a `getDATA_CHARS()` getter `invokevirtual` per character instead - `@JvmField` on the table makes the call site `getstatic; iload; baload` -- the same three instructions the private array produced Benchmark, medians of 3 runs of RegexContentBenchmark (nanoseconds, lower better): bytes before inlined 0 mentions 4104 4473 4518 0 mentions 68096 73398 74167 0 mentions 767144 836400 835772 m=120 767072 1029977 1030072 TOTAL 2102097 2104899 (+0.1%) The 767KB cases -- the tail that caused the OOM -- overlap run to run (before [855512, 828137, 836400] vs inlined [833456, 839112, 835772]). The two sub-microsecond cases swing ±80% between repeats of the *same* build, so they carry no signal. On-device native heap is unchanged from the previous commit: plateaus at ~169-182MB over two cold starts, zero lmkd kills. Adds Bech32DataCharTest, which sweeps the whole BMP and requires isDataChar to agree with ALPHABET exactly. Mutating the shared ALPHABET (adding 'b') now fails both it and the NIP-19 equivalence test. Co-Authored-By: Claude Opus 5 (1M context) --- .../quartz/nip19Bech32/Nip19Parser.kt | 16 +---- .../quartz/nip19Bech32/bech32/Bech32Util.kt | 37 ++++++++++ .../nip19Bech32/bech32/Bech32DataCharTest.kt | 67 +++++++++++++++++++ 3 files changed, 107 insertions(+), 13 deletions(-) create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index b78ddfaf4b..25190ecc9d 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -24,6 +24,7 @@ import androidx.compose.runtime.Immutable import com.vitorpamplona.quartz.nip01Core.core.HexKey import com.vitorpamplona.quartz.nip01Core.core.hexToByteArray import com.vitorpamplona.quartz.nip01Core.core.toHexKey +import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32 import com.vitorpamplona.quartz.nip19Bech32.bech32.bechToBytes import com.vitorpamplona.quartz.nip19Bech32.entities.Entity import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress @@ -163,17 +164,6 @@ object Nip19Parser { private val EVENT_VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "note1", "nrelay1", "nembed1") - /** bech32: digits except `1`, letters except `b`, `i`, `o`. ASCII-only, both cases. */ - private val BECH32_CHARS = - BooleanArray(128).apply { - for (c in "qpzry9x8gf2tvdw0s3jn54khce6mua7l") { - this[c.code] = true - if (c in 'a'..'z') this[c.code - 32] = true - } - } - - private fun isBech32(c: Char): Boolean = c.code < 128 && BECH32_CHARS[c.code] - /** * `\S` in the trailing group. Java's `\s` is the six ASCII whitespace chars unless * `UNICODE_CHARACTER_CLASS` is set, so U+00A0 and friends count as NON-space here. @@ -280,7 +270,7 @@ object Nip19Parser { if (!matchesPrefixAt(content, i, prefix)) continue val dataStart = i + prefix.length var dataEnd = dataStart - while (dataEnd < len && isBech32(content[dataEnd])) dataEnd++ + while (dataEnd < len && Bech32.isDataChar(content[dataEnd])) dataEnd++ // `+` needs at least one payload char. if (dataEnd > dataStart) { end = endOfTrailing(content, dataEnd) @@ -310,7 +300,7 @@ object Nip19Parser { to: Int, ): Boolean { for (k in from until to) { - if (!isBech32(content[k])) return false + if (!Bech32.isDataChar(content[k])) return false } return true } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt index 2cc561e561..7d2a2b6541 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32Util.kt @@ -20,6 +20,8 @@ */ package com.vitorpamplona.quartz.nip19Bech32.bech32 +import kotlin.jvm.JvmField + /* * Copyright 2020 ACINQ SAS * @@ -81,15 +83,50 @@ object Bech32 { // char -> 5 bits value private val map = Array(255) { -1 } + @PublishedApi + internal const val DATA_CHARS_SIZE = 128 + + /** + * Membership table for [isDataChar], kept separate from [map] because [map] is an + * `Array` — a boxed `java.lang.Byte[]` — and [isDataChar] runs per character over whole + * note contents (up to ~767KB), where unboxing on every char would show up. + * + * `@JvmField` so callers of the inline [isDataChar] compile to a direct `getstatic` instead of + * a property-getter `invokevirtual` per character (the same reasoning as `PointTypes`). + * `@PublishedApi internal` because a public inline function cannot touch a private member. + */ + @PublishedApi + @JvmField + internal val DATA_CHARS = BooleanArray(DATA_CHARS_SIZE) + init { for (i in 0..ALPHABET.lastIndex) { map[ALPHABET[i].code] = i.toByte() + DATA_CHARS[ALPHABET[i].code] = true } for (i in 0..ALPHABET_UPPERCASE.lastIndex) { map[ALPHABET_UPPERCASE[i].code] = i.toByte() + DATA_CHARS[ALPHABET_UPPERCASE[i].code] = true } } + /** + * True when [c] is part of the bech32 data alphabet, in either case — i.e. everything except + * `1`, `b`, `i` and `o`, which BIP-173 excludes as visually ambiguous. + * + * Exposed so scanners can find where an encoded payload *ends* without decoding it. The NIP-19 + * content scan needs exactly that on every ingested event, and cannot use a regex to do it: + * Android's `java.util.regex` is ICU-backed and `Matcher.region()` copies the entire input into + * native memory per call, which drove the app's native heap to ~1.9GB on a cold start. + * + * `inline` because it is called per character over whole note contents (up to ~767KB). As a + * normal member it compiled to an `invokevirtual` per char, which measured ~1-2% slower across + * the scan benchmark than the equivalent private lookup it replaced; inlining puts the constant + * compare and the `baload` straight into the caller's loop and closes that gap. + */ + @Suppress("NOTHING_TO_INLINE") + inline fun isDataChar(c: Char): Boolean = c.code < DATA_CHARS_SIZE && DATA_CHARS[c.code] + fun expand(hrp: String): Array { val half = hrp.length + 1 val size = half + hrp.length diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt new file mode 100644 index 0000000000..dd2cd46816 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip19Bech32/bech32/Bech32DataCharTest.kt @@ -0,0 +1,67 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip19Bech32.bech32 + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertFalse +import kotlin.test.assertTrue + +/** + * [Bech32.isDataChar] is the membership test the NIP-19 content scan uses to find where an encoded + * payload ends, so it has to agree with [Bech32.ALPHABET] exactly — a char wrongly accepted extends + * a payload past its real end and silently drops the entity when the decode then fails. + */ +class Bech32DataCharTest { + @Test + fun acceptsExactlyTheAlphabetInBothCases() { + for (c in Bech32.ALPHABET) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char") + for (c in Bech32.ALPHABET_UPPERCASE) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char") + } + + @Test + fun rejectsTheAmbiguousFour() { + // BIP-173 leaves these out of the alphabet precisely because they are easy to misread. + for (c in "1bio1BIO") assertFalse(Bech32.isDataChar(c), "expected '$c' to be rejected") + } + + @Test + fun agreesWithTheAlphabetAcrossEveryChar() { + // Sweeps the whole BMP so nothing outside the alphabet sneaks in — including the + // out-of-range guard for chars beyond the lookup table. + val expected = (Bech32.ALPHABET + Bech32.ALPHABET_UPPERCASE).toSet() + for (code in 0..0xFFFF) { + val c = code.toChar() + assertEquals(c in expected, Bech32.isDataChar(c), "disagreement at code $code") + } + } + + @Test + fun countsMatchTheSpec() { + assertEquals(32, Bech32.ALPHABET.length) + // 32 symbols, but only the 23 letters have a distinct uppercase form — the 9 digits + // are the same char in both alphabets, so the accepted set is 23*2 + 9, not 64. + val letters = Bech32.ALPHABET.count { it.isLetter() } + val digits = Bech32.ALPHABET.count { it.isDigit() } + assertEquals(32, letters + digits) + assertEquals(letters * 2 + digits, (0..0xFFFF).count { Bech32.isDataChar(it.toChar()) }) + } +} From 86a9d3b7803d89af72f2e7e40e49006a78a956bc Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Mon, 17 Aug 2026 19:32:37 -0400 Subject: [PATCH 3/4] perf: jump NIP-19 candidates with indexOf instead of testing every char The scanner walked the content one character at a time looking for a candidate prefix. `findHashtags` already showed the better shape for this: jump between candidate positions with `indexOf`, which is an intrinsified, vectorised char search, and only do real work where one lands. 'n'/'N' are two separate searches, so both are tracked and each is only re-searched once consumed, amortising to about one indexOf per candidate. Medians of 6 RegexContentBenchmark runs per arm (ns/op, lower better): case bytes char-loop indexOf delta overlap 0 mentions 152 774.5 419.5 -45.8% no 0 mentions 608 1048.0 422.5 -59.7% no 0 mentions 4104 4704.5 2443.5 -48.1% no 0 mentions 68096 75678.0 38936.5 -48.5% no 0 mentions 767144 867870.5 434141.0 -50.0% no m=120 767072 1097200.5 808300.5 -26.3% no m=40 68072 150933.5 125125.0 -17.1% yes m=5 4050 12317.0 10835.5 -12.0% yes m=1 222 3185.5 3367.0 +5.7% yes m=2 698 4453.5 7211.0 +61.9% yes TOTAL 2218165.5 1431202.0 -35.5% Every no-match case is ~2x faster with non-overlapping ranges, and that is the path that matters: of 2588 real notes sampled off production relays, only 160 contain a NIP-19 prefix at all, so 94% never leave the scan loop. The 767KB mention-heavy tail -- the shape behind the OOM -- is 26% faster too. The one arguable regression is a ~700B note with 2 mentions, +2.7us in absolute terms with overlapping ranges across six runs. Accepted deliberately: it is microseconds on the rarest shape, against halving the case that runs on nearly every event. Still 0 mismatches against both original regexes over the 2588-note production corpus (7742 entities parsed), plus the synthetic equivalence corpus and Nip19ScanTest. Co-Authored-By: Claude Opus 5 (1M context) --- .../quartz/nip19Bech32/Nip19Parser.kt | 18 ++++++++++++++++++ 1 file changed, 18 insertions(+) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt index 25190ecc9d..c7e83091ac 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip19Bech32/Nip19Parser.kt @@ -246,7 +246,25 @@ object Nip19Parser { ) { var i = 0 val len = content.length + // Jump between 'n'/'N' with indexOf — an intrinsified, vectorised char search — instead of + // testing every character. Both cases are tracked separately so each is only re-searched + // once consumed, amortising to about one indexOf per candidate. Measured ~2x faster on + // content with no entity at all, which is 94% of real notes (2588-note production sample), + // and ~26% faster on the 767KB mention-heavy tail. Small mention-dense content (a ~700B + // note with 2 mentions) is a few microseconds slower; that trade is deliberate. + var nextLower = content.indexOf('n') + var nextUpper = content.indexOf('N') while (i < len) { + if (nextLower in 0.. return + nextLower < 0 -> nextUpper + nextUpper < 0 -> nextLower + else -> if (nextLower < nextUpper) nextLower else nextUpper + } + if (i >= len) return if (isCandidateAt(content, i)) { var end = -1 From 6d57f30307bd43350fe36c6a1cc5fea6a41f01b0 Mon Sep 17 00:00:00 2001 From: Vitor Pamplona Date: Mon, 17 Aug 2026 20:22:04 -0400 Subject: [PATCH 4/4] perf: scan hashtags and #[n] references without ICU MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `findHashtags` and `forEachIndexTag` jumped between `#` candidates with indexOf and then anchored `Regex.matchAt` at each. On Android `java.util.regex` is ICU-backed, and `Matcher.region()` -> `reset()` -> `MatcherNative.setInput()` copies the ENTIRE input into native memory per call, so every candidate cost a full native UTF-16 copy of the note's content. This is the same defect fixed for the NIP-19 scanner in the OOM work, and measured over 2588 notes pulled off production relays it is considerably worse, because a whitespace-preceded `#` is far more common in prose than a NIP-19 prefix: scanner Matchers native bytes copied worst single note nip19 7,871 1,752 MB 62.8 MB findHashtags 43,626 9,639 MB 279.6 MB (a 119KB note) findIndexTags 0 0 - Both grammars are small, so they are matched directly instead. `hashtagSearch` and `tagSearch` stay as the specification the scan is tested against. Case handling is deliberate: `(?:\s|\A)` is Java's `\s`, which without UNICODE_CHARACTER_CLASS is space plus 0x09..0x0D and nothing else, so the new `isAsciiRegexSpace` is used rather than Char.isWhitespace() — the latter is Unicode-aware and would accept U+00A0 before a `#`, which the regex rejected. The same asymmetry runs the other way inside a tag: the excluded punctuation class is entirely ASCII, so non-ASCII always continues a tag, and a tag made only of U+00A0 is non-empty to the regex but still dropped by `isNotBlank()`. Speed, medians of 5 RegexContentBenchmark runs (ns/op, lower better): case bytes regex ICU-free delta ranges overlap hashtags m=5 4050 3267 1035 -68.3% yes hashtags m=40 68072 56956 16033 -71.9% yes hashtags m=120 767072 656535 185675 -71.7% no TOTAL 717550 203400 -71.7% idxTags TOTAL 112932.5 99937.5 -11.5% Equivalence is pinned by a new test running both original regexes over a corpus covering the punctuation class, ASCII-vs-Unicode whitespace either side of the `#`, non-ASCII tag content and the minimum-one-character rules. Two mutations (dropping `.` from the terminators, and swapping in Char.isWhitespace) fail both it and the pre-existing ContentScanTest. Against 2588 real production notes the scanners agree with the regexes on every one, 32,075 hashtags parsed. `findIndexTags` shares the defect but never fires on real data — `#[0]` is the legacy citation form no current client emits — so it is fixed for consistency rather than impact. Co-Authored-By: Claude Opus 5 (1M context) --- .../nip10Notes/content/ContentHashTags.kt | 66 ++++--- .../nip10Notes/content/ContentScanChars.kt | 31 ++++ .../quartz/nip10Notes/content/IndexedTags.kt | 58 +++--- .../ContentScanRegexEquivalenceTest.kt | 167 ++++++++++++++++++ 4 files changed, 272 insertions(+), 50 deletions(-) create mode 100644 quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt create mode 100644 quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt index 51a70d5331..748550ce50 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentHashTags.kt @@ -22,19 +22,43 @@ package com.vitorpamplona.quartz.nip10Notes.content val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)") +/** + * Characters that end a hashtag: the punctuation class spelled out in [hashtagSearch], plus ASCII + * whitespace. Everything else continues the tag — including every non-ASCII character, since the + * regex's class is ASCII-only, so accented letters, CJK and emoji are all valid tag content. + */ +private val HASHTAG_TERMINATORS = + BooleanArray(128).apply { + for (c in 0x09..0x0D) this[c] = true + this[' '.code] = true + for (c in "!@#\u0024%^&*()=+./,[{]};:'\"?><") this[c.code] = true + } + +/** + * True while [c] can still be part of a hashtag. + * + * Non-ASCII always continues the tag: [hashtagSearch]'s excluded set is entirely ASCII and its + * `\s` is ASCII-only, so nothing above 0x7F was ever excluded. + */ +private fun isHashtagChar(c: Char): Boolean = c.code >= 128 || !HASHTAG_TERMINATORS[c.code] + /** * Collects the hashtags in [content]. * - * [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match - * starts either at position 0 or at a whitespace. That lets the scan jump between - * `#` occurrences with `indexOf` — an intrinsified char search — and apply the - * regex **anchored** at each, instead of letting `findAll` drive the regex engine - * from every position in the string. + * Jumps between `#` occurrences with `indexOf` — an intrinsified char search — and then matches + * `#` directly, character by character, rather than anchoring a regex there. * - * Measured on the production content distribution (median 529 B, tail to 767 KB): - * ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the - * content has no `#` at all (up to 300x). Equivalence with the previous `findAll` - * implementation is guarded by `RegexContentBenchmark` in `commons`. + * **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` -> + * `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory on every + * call. `Regex.matchAt` builds a fresh Matcher per call, so anchoring one at each candidate cost a + * full native UTF-16 copy of the note's content **per `#`** — the same defect that drove the app's + * native heap to ~1.9GB on a cold start via the NIP-19 scanner. This one is worse: measured over + * 2588 real notes it minted 43,626 Matchers copying 9.6GB in total, with a single 119KB note + * costing 279MB, because a whitespace-preceded `#` is far more common in prose than a NIP-19 + * prefix. The Java `Matcher` object is tiny, so Java-heap-driven GC had no reason to reclaim them + * promptly while each pinned native memory. + * + * [hashtagSearch] is kept as the specification the scan is tested against, not used here. */ fun findHashtags( content: String, @@ -44,19 +68,17 @@ fun findHashtags( var h = content.indexOf('#') while (h >= 0) { - if (h == 0 || content[h - 1].isWhitespace()) { - val match = - try { - hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1) - } catch (e: Exception) { - null - } - if (match != null) { - val tag = match.groups[1]?.value - if (tag != null && tag.isNotBlank()) { - output.add(tag) - } - h = content.indexOf('#', match.range.last + 1) + // `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character. + if (h == 0 || isAsciiRegexSpace(content[h - 1])) { + var end = h + 1 + while (end < content.length && isHashtagChar(content[end])) end++ + // The tag group is `+`, so it needs at least one character. + if (end > h + 1) { + val tag = content.substring(h + 1, end) + // Non-ASCII whitespace (U+00A0 and friends) is valid tag content to the regex but + // still blank to Kotlin, and the old code dropped those too. + if (tag.isNotBlank()) output.add(tag) + h = content.indexOf('#', end) continue } } diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt new file mode 100644 index 0000000000..dcf1138af0 --- /dev/null +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanChars.kt @@ -0,0 +1,31 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip10Notes.content + +/** + * `\s` as `java.util.regex` applies it *without* `UNICODE_CHARACTER_CLASS`: space plus the five + * control characters `\t \n \x0B \f \r`, which are contiguous at 0x09..0x0D — and nothing else. + * + * The scanners in this package hand-roll grammars that used to be regexes, so they must use this + * rather than [Char.isWhitespace], which is Unicode-aware and would accept U+00A0, U+2003 and + * friends that the regexes rejected. + */ +internal fun isAsciiRegexSpace(c: Char): Boolean = c == ' ' || c.code in 0x09..0x0D diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt index cb6a2769a6..0ce6fed888 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip10Notes/content/IndexedTags.kt @@ -29,36 +29,36 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]") /** - * Walks every `#[n]` reference in [content]. + * Walks every `#[n]` reference in [content], handing each callback the digits between the brackets. * - * [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match - * starts at position 0 or at a whitespace. That lets the scan jump between `#` - * occurrences with `indexOf` — an intrinsified char search — and apply the regex - * **anchored** at each, instead of letting `findAll` drive the regex engine from - * every position in the string. + * Jumps between `#` occurrences with `indexOf` — an intrinsified char search — then matches + * `#[]` directly rather than anchoring a regex there. * - * Measured on the production content distribution (median 529 B, tail to 767 KB): - * ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense - * text. Equivalence with the previous `findAll` implementation (both callers) is - * guarded by `RegexContentBenchmark` in `commons`. + * **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` -> + * `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory per call, + * so anchoring a fresh Matcher at each candidate cost a full native UTF-16 copy of the content per + * `#`. See [findHashtags] for the measurements; this scanner shares the defect but never fires on + * real data, since `#[0]` is the legacy citation form that no current client emits. + * + * [tagSearch] is kept as the specification the scan is tested against, not used here. */ private inline fun forEachIndexTag( content: String, - action: (MatchResult) -> Unit, + action: (digits: String) -> Unit, ) { var h = content.indexOf('#') while (h >= 0) { - if (h == 0 || content[h - 1].isWhitespace()) { - val match = - try { - tagSearch.matchAt(content, if (h == 0) 0 else h - 1) - } catch (e: Exception) { - null + // `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character. + if (h == 0 || isAsciiRegexSpace(content[h - 1])) { + if (h + 1 < content.length && content[h + 1] == '[') { + var d = h + 2 + while (d < content.length && content[d] in '0'..'9') d++ + // `([0-9]+)` needs a digit, and the `]` must actually be there. + if (d > h + 2 && d < content.length && content[d] == ']') { + action(content.substring(h + 2, d)) + h = content.indexOf('#', d + 1) + continue } - if (match != null) { - action(match) - h = content.indexOf('#', match.range.last + 1) - continue } } h = content.indexOf('#', h + 1) @@ -73,10 +73,11 @@ fun findIndexTagsWithPeople( tags: TagArray, output: MutableSet = mutableSetOf(), ): List { - forEachIndexTag(content) { index -> + forEachIndexTag(content) { digits -> try { - val tag = index.groups[1]?.value?.let { tags[it.toInt()] } - if (tag != null && tag.size > 1 && tag[0] == "p") { + // Out-of-range indexes and non-numeric digits land in the catch below. + val tag = tags[digits.toInt()] + if (tag.size > 1 && tag[0] == "p") { output.add(tag[1]) } } catch (e: Exception) { @@ -94,13 +95,14 @@ fun findIndexTagsWithEventsOrAddresses( tags: TagArray, output: MutableSet = mutableSetOf(), ): Set { - forEachIndexTag(content) { index -> + forEachIndexTag(content) { digits -> try { - val tag = index.groups[1]?.value?.let { tags[it.toInt()] } - if (tag != null && tag.size > 1 && tag[0] == "e") { + // Out-of-range indexes and non-numeric digits land in the catch below. + val tag = tags[digits.toInt()] + if (tag.size > 1 && tag[0] == "e") { output.add(tag[1]) } - if (tag != null && tag.size > 1 && tag[0] == "a") { + if (tag.size > 1 && tag[0] == "a") { output.add(tag[1]) } } catch (e: Exception) { diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt new file mode 100644 index 0000000000..4bbf310bf6 --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip10Notes/content/ContentScanRegexEquivalenceTest.kt @@ -0,0 +1,167 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip10Notes.content + +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * Pins the ICU-free content scanners to the regexes they replaced. + * + * [findHashtags] and the `#[n]` walker used to anchor a fresh `Regex.matchAt` at every candidate. + * On Android that goes through ICU, where `Matcher.region()` copies the whole input into native + * memory per call — over 2588 real notes that was 43,626 Matchers copying 9.6GB, worst single note + * 279MB. The scanners now match the grammars directly, so [hashtagSearch] and [tagSearch] survive + * only as the specification, and this asserts the two agree. + * + * The corpus targets where a hand-rolled matcher is most likely to drift: the exact punctuation + * set that ends a tag, ASCII-vs-Unicode whitespace before the `#`, non-ASCII inside the tag, and + * the `+`/`[0-9]+` minimum-one-character rules. + */ +class ContentScanRegexEquivalenceTest { + private fun referenceHashtags(content: String): List { + if (content.isBlank()) return emptyList() + val out = mutableSetOf() + hashtagSearch.findAll(content).forEach { m -> + val tag = m.groups[1]?.value + if (tag != null && tag.isNotBlank()) out.add(tag) + } + return out.toList() + } + + private fun referenceIndexTags( + content: String, + tags: Array>, + wanted: String, + ): Set { + val out = mutableSetOf() + tagSearch.findAll(content).forEach { m -> + try { + val tag = m.groups[1]?.value?.let { tags[it.toInt()] } + if (tag != null && tag.size > 1 && tag[0] == wanted) out.add(tag[1]) + } catch (e: Exception) { + } + } + return out + } + + private val tagArray = + arrayOf( + arrayOf("p", "pubkey0"), + arrayOf("e", "event1"), + arrayOf("a", "addr2"), + arrayOf("p", "pubkey3"), + arrayOf("t", "topic4"), + ) + + private fun corpus(): List = + buildList { + add("") + add(" ") + add("#") + add("#tag") + add("hello #tag world") + add("a#tag") + add("#tag#other") + add("#tag #other") + add("##tag") + add("#tag.") + add("#tag, and #more!") + add("#tag's") + add("#tag\"quoted\"") + add("#a") + add("#1") + add("#tag-with-dash") + add("#tag_with_underscore") + add("#tag~tilde|pipe\\back`tick") + // non-ASCII is valid tag content: the regex class and its \s are ASCII-only + add("#café") + add("#日本語") + add("#tagéè") + // ASCII vs Unicode whitespace BEFORE the # decides whether it matches at all + add("x\u00A0#tag") + add("x\u2003#tag") + add("x\t#tag") + add("x\n#tag") + add("x\r#tag") + // Unicode whitespace INSIDE the tag is valid to the regex but blank to Kotlin + add("#\u00A0") + add("#\u00A0x") + // every excluded char must terminate the tag + for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#tag${c}more") + for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#$c") + // index tags + add("#[0]") + add("#[1] and #[2]") + add("look #[3] here") + add("#[]") + add("#[abc]") + add("#[99]") + add("#[0") + add("#[0]]") + add("x#[0]") + add("x\u00A0#[0]") + add("#[0]#[1]") + add("#[00]") + // mixed + add("#tag #[0] #other #[1]") + add("lorem ipsum ".repeat(500) + "#tail") + add("#head" + " dolor sit ".repeat(500)) + add("no hashes at all here ".repeat(200)) + } + + @Test + fun hashtagsMatchRegex() { + corpus().forEach { c -> + assertEquals( + referenceHashtags(c).sorted(), + findHashtags(c).sorted(), + "findHashtags diverged on: ${c.take(80)}", + ) + } + } + + @Test + fun indexTagsMatchRegex() { + corpus().forEach { c -> + assertEquals( + referenceIndexTags(c, tagArray, "p").sorted(), + findIndexTagsWithPeople(c, tagArray).sorted(), + "findIndexTagsWithPeople diverged on: ${c.take(80)}", + ) + val refEv = referenceIndexTags(c, tagArray, "e") + referenceIndexTags(c, tagArray, "a") + assertEquals( + refEv.sorted(), + findIndexTagsWithEventsOrAddresses(c, tagArray).sorted(), + "findIndexTagsWithEventsOrAddresses diverged on: ${c.take(80)}", + ) + } + } + + @Test + fun hashtagsMatchRegexAtEveryOffset() { + val filler = "a b\tc\nd " + for (i in 0..filler.length) { + val c = filler.substring(0, i) + "#tag" + filler.substring(i) + assertEquals(referenceHashtags(c).sorted(), findHashtags(c).sorted(), "offset $i") + } + } +}