mirror of
https://github.com/vitorpamplona/amethyst.git
synced 2026-10-06 11:48:24 +00:00
Merge remote-tracking branch 'origin/main' into claude/concord-nip29-invitations-vx2wgj
This commit is contained in:
+44
-22
@@ -22,19 +22,43 @@ package com.vitorpamplona.quartz.nip10Notes.content
|
||||
|
||||
val hashtagSearch = Regex("(?:\\s|\\A)#([^\\s!@#\$%^&*()=+./,\\[{\\]};:'\"?><]+)")
|
||||
|
||||
/**
|
||||
* Characters that end a hashtag: the punctuation class spelled out in [hashtagSearch], plus ASCII
|
||||
* whitespace. Everything else continues the tag — including every non-ASCII character, since the
|
||||
* regex's class is ASCII-only, so accented letters, CJK and emoji are all valid tag content.
|
||||
*/
|
||||
private val HASHTAG_TERMINATORS =
|
||||
BooleanArray(128).apply {
|
||||
for (c in 0x09..0x0D) this[c] = true
|
||||
this[' '.code] = true
|
||||
for (c in "!@#\u0024%^&*()=+./,[{]};:'\"?><") this[c.code] = true
|
||||
}
|
||||
|
||||
/**
|
||||
* True while [c] can still be part of a hashtag.
|
||||
*
|
||||
* Non-ASCII always continues the tag: [hashtagSearch]'s excluded set is entirely ASCII and its
|
||||
* `\s` is ASCII-only, so nothing above 0x7F was ever excluded.
|
||||
*/
|
||||
private fun isHashtagChar(c: Char): Boolean = c.code >= 128 || !HASHTAG_TERMINATORS[c.code]
|
||||
|
||||
/**
|
||||
* Collects the hashtags in [content].
|
||||
*
|
||||
* [hashtagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match
|
||||
* starts either at position 0 or at a whitespace. That lets the scan jump between
|
||||
* `#` occurrences with `indexOf` — an intrinsified char search — and apply the
|
||||
* regex **anchored** at each, instead of letting `findAll` drive the regex engine
|
||||
* from every position in the string.
|
||||
* Jumps between `#` occurrences with `indexOf` — an intrinsified char search — and then matches
|
||||
* `#<tag>` directly, character by character, rather than anchoring a regex there.
|
||||
*
|
||||
* Measured on the production content distribution (median 529 B, tail to 767 KB):
|
||||
* ~68 MB/s -> ~1,240 MB/s on hashtag-dense text (18x) and ~19,000 MB/s when the
|
||||
* content has no `#` at all (up to 300x). Equivalence with the previous `findAll`
|
||||
* implementation is guarded by `RegexContentBenchmark` in `commons`.
|
||||
* **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` ->
|
||||
* `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory on every
|
||||
* call. `Regex.matchAt` builds a fresh Matcher per call, so anchoring one at each candidate cost a
|
||||
* full native UTF-16 copy of the note's content **per `#`** — the same defect that drove the app's
|
||||
* native heap to ~1.9GB on a cold start via the NIP-19 scanner. This one is worse: measured over
|
||||
* 2588 real notes it minted 43,626 Matchers copying 9.6GB in total, with a single 119KB note
|
||||
* costing 279MB, because a whitespace-preceded `#` is far more common in prose than a NIP-19
|
||||
* prefix. The Java `Matcher` object is tiny, so Java-heap-driven GC had no reason to reclaim them
|
||||
* promptly while each pinned native memory.
|
||||
*
|
||||
* [hashtagSearch] is kept as the specification the scan is tested against, not used here.
|
||||
*/
|
||||
fun findHashtags(
|
||||
content: String,
|
||||
@@ -44,19 +68,17 @@ fun findHashtags(
|
||||
|
||||
var h = content.indexOf('#')
|
||||
while (h >= 0) {
|
||||
if (h == 0 || content[h - 1].isWhitespace()) {
|
||||
val match =
|
||||
try {
|
||||
hashtagSearch.matchAt(content, if (h == 0) 0 else h - 1)
|
||||
} catch (e: Exception) {
|
||||
null
|
||||
}
|
||||
if (match != null) {
|
||||
val tag = match.groups[1]?.value
|
||||
if (tag != null && tag.isNotBlank()) {
|
||||
output.add(tag)
|
||||
}
|
||||
h = content.indexOf('#', match.range.last + 1)
|
||||
// `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character.
|
||||
if (h == 0 || isAsciiRegexSpace(content[h - 1])) {
|
||||
var end = h + 1
|
||||
while (end < content.length && isHashtagChar(content[end])) end++
|
||||
// The tag group is `+`, so it needs at least one character.
|
||||
if (end > h + 1) {
|
||||
val tag = content.substring(h + 1, end)
|
||||
// Non-ASCII whitespace (U+00A0 and friends) is valid tag content to the regex but
|
||||
// still blank to Kotlin, and the old code dropped those too.
|
||||
if (tag.isNotBlank()) output.add(tag)
|
||||
h = content.indexOf('#', end)
|
||||
continue
|
||||
}
|
||||
}
|
||||
|
||||
+31
@@ -0,0 +1,31 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip10Notes.content
|
||||
|
||||
/**
|
||||
* `\s` as `java.util.regex` applies it *without* `UNICODE_CHARACTER_CLASS`: space plus the five
|
||||
* control characters `\t \n \x0B \f \r`, which are contiguous at 0x09..0x0D — and nothing else.
|
||||
*
|
||||
* The scanners in this package hand-roll grammars that used to be regexes, so they must use this
|
||||
* rather than [Char.isWhitespace], which is Unicode-aware and would accept U+00A0, U+2003 and
|
||||
* friends that the regexes rejected.
|
||||
*/
|
||||
internal fun isAsciiRegexSpace(c: Char): Boolean = c == ' ' || c.code in 0x09..0x0D
|
||||
+30
-28
@@ -29,36 +29,36 @@ import com.vitorpamplona.quartz.nip01Core.core.TagArray
|
||||
val tagSearch = Regex("(?:\\s|\\A)\\#\\[([0-9]+)\\]")
|
||||
|
||||
/**
|
||||
* Walks every `#[n]` reference in [content].
|
||||
* Walks every `#[n]` reference in [content], handing each callback the digits between the brackets.
|
||||
*
|
||||
* [tagSearch] requires `(?:\s|\A)` immediately before the `#`, so every match
|
||||
* starts at position 0 or at a whitespace. That lets the scan jump between `#`
|
||||
* occurrences with `indexOf` — an intrinsified char search — and apply the regex
|
||||
* **anchored** at each, instead of letting `findAll` drive the regex engine from
|
||||
* every position in the string.
|
||||
* Jumps between `#` occurrences with `indexOf` — an intrinsified char search — then matches
|
||||
* `#[<digits>]` directly rather than anchoring a regex there.
|
||||
*
|
||||
* Measured on the production content distribution (median 529 B, tail to 767 KB):
|
||||
* ~63 MB/s -> multiple GB/s when the content has no `#`, and ~18x on reference-dense
|
||||
* text. Equivalence with the previous `findAll` implementation (both callers) is
|
||||
* guarded by `RegexContentBenchmark` in `commons`.
|
||||
* **Why not a regex.** On Android `java.util.regex` is ICU-backed, and `Matcher.region()` ->
|
||||
* `reset()` -> `MatcherNative.setInput()` copies the *entire input* into native memory per call,
|
||||
* so anchoring a fresh Matcher at each candidate cost a full native UTF-16 copy of the content per
|
||||
* `#`. See [findHashtags] for the measurements; this scanner shares the defect but never fires on
|
||||
* real data, since `#[0]` is the legacy citation form that no current client emits.
|
||||
*
|
||||
* [tagSearch] is kept as the specification the scan is tested against, not used here.
|
||||
*/
|
||||
private inline fun forEachIndexTag(
|
||||
content: String,
|
||||
action: (MatchResult) -> Unit,
|
||||
action: (digits: String) -> Unit,
|
||||
) {
|
||||
var h = content.indexOf('#')
|
||||
while (h >= 0) {
|
||||
if (h == 0 || content[h - 1].isWhitespace()) {
|
||||
val match =
|
||||
try {
|
||||
tagSearch.matchAt(content, if (h == 0) 0 else h - 1)
|
||||
} catch (e: Exception) {
|
||||
null
|
||||
// `(?:\s|\A)` — the `#` must open the string or follow one ASCII space character.
|
||||
if (h == 0 || isAsciiRegexSpace(content[h - 1])) {
|
||||
if (h + 1 < content.length && content[h + 1] == '[') {
|
||||
var d = h + 2
|
||||
while (d < content.length && content[d] in '0'..'9') d++
|
||||
// `([0-9]+)` needs a digit, and the `]` must actually be there.
|
||||
if (d > h + 2 && d < content.length && content[d] == ']') {
|
||||
action(content.substring(h + 2, d))
|
||||
h = content.indexOf('#', d + 1)
|
||||
continue
|
||||
}
|
||||
if (match != null) {
|
||||
action(match)
|
||||
h = content.indexOf('#', match.range.last + 1)
|
||||
continue
|
||||
}
|
||||
}
|
||||
h = content.indexOf('#', h + 1)
|
||||
@@ -73,10 +73,11 @@ fun findIndexTagsWithPeople(
|
||||
tags: TagArray,
|
||||
output: MutableSet<String> = mutableSetOf<String>(),
|
||||
): List<String> {
|
||||
forEachIndexTag(content) { index ->
|
||||
forEachIndexTag(content) { digits ->
|
||||
try {
|
||||
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
|
||||
if (tag != null && tag.size > 1 && tag[0] == "p") {
|
||||
// Out-of-range indexes and non-numeric digits land in the catch below.
|
||||
val tag = tags[digits.toInt()]
|
||||
if (tag.size > 1 && tag[0] == "p") {
|
||||
output.add(tag[1])
|
||||
}
|
||||
} catch (e: Exception) {
|
||||
@@ -94,13 +95,14 @@ fun findIndexTagsWithEventsOrAddresses(
|
||||
tags: TagArray,
|
||||
output: MutableSet<String> = mutableSetOf<String>(),
|
||||
): Set<String> {
|
||||
forEachIndexTag(content) { index ->
|
||||
forEachIndexTag(content) { digits ->
|
||||
try {
|
||||
val tag = index.groups[1]?.value?.let { tags[it.toInt()] }
|
||||
if (tag != null && tag.size > 1 && tag[0] == "e") {
|
||||
// Out-of-range indexes and non-numeric digits land in the catch below.
|
||||
val tag = tags[digits.toInt()]
|
||||
if (tag.size > 1 && tag[0] == "e") {
|
||||
output.add(tag[1])
|
||||
}
|
||||
if (tag != null && tag.size > 1 && tag[0] == "a") {
|
||||
if (tag.size > 1 && tag[0] == "a") {
|
||||
output.add(tag[1])
|
||||
}
|
||||
} catch (e: Exception) {
|
||||
|
||||
@@ -24,6 +24,7 @@ import androidx.compose.runtime.Immutable
|
||||
import com.vitorpamplona.quartz.nip01Core.core.HexKey
|
||||
import com.vitorpamplona.quartz.nip01Core.core.hexToByteArray
|
||||
import com.vitorpamplona.quartz.nip01Core.core.toHexKey
|
||||
import com.vitorpamplona.quartz.nip19Bech32.bech32.Bech32
|
||||
import com.vitorpamplona.quartz.nip19Bech32.bech32.bechToBytes
|
||||
import com.vitorpamplona.quartz.nip19Bech32.entities.Entity
|
||||
import com.vitorpamplona.quartz.nip19Bech32.entities.NAddress
|
||||
@@ -133,6 +134,67 @@ object Nip19Parser {
|
||||
|
||||
fun hasAny(content: String): Boolean = nip19regex.matches(content)
|
||||
|
||||
// ---------------------------------------------------------------------------------------
|
||||
// ICU-free scanner for the ingest hot path.
|
||||
//
|
||||
// `java.util.regex` on Android is backed by ICU, and `Matcher.region()` -> `reset()` ->
|
||||
// `MatcherNative.setInput()` copies the ENTIRE input into native memory on every call.
|
||||
// `Regex.matchAt` builds a fresh Matcher per call, and [forEachNip19Match] calls it once per
|
||||
// candidate position — so scanning one note allocated a full native UTF-16 copy of its content
|
||||
// *per candidate*, with content running to 767KB in the tail. Because the Java Matcher object
|
||||
// is tiny, Java-heap-driven GC had no reason to reclaim them promptly, so the native heap grew
|
||||
// unbounded: measured 45MB -> 1372MB over a cold start, ending in an lmkd kill at ~1.9GB RSS.
|
||||
// Disabling this one scan made the native heap plateau at ~257MB instead.
|
||||
//
|
||||
// The grammar is small enough to match directly, so nothing here touches ICU. Case folding is
|
||||
// done ASCII-only on purpose: `RegexOption.IGNORE_CASE` maps to `Pattern.CASE_INSENSITIVE`,
|
||||
// which is ASCII-only unless `UNICODE_CASE` is also set. Using Kotlin's `ignoreCase = true`
|
||||
// would be Unicode-aware and would accept inputs the regex rejected (U+212A KELVIN SIGN folding
|
||||
// to `k`, say).
|
||||
// ---------------------------------------------------------------------------------------
|
||||
|
||||
/** Entities whose bech32 payload the regexes pin to exactly 58 chars. */
|
||||
private val FIXED_58_PREFIXES = arrayOf("nsec1", "npub1", "note1")
|
||||
|
||||
/** Entities whose bech32 payload is `+` (one or more). */
|
||||
private val VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "nprofile1", "nrelay1", "nembed1")
|
||||
|
||||
/** [nip19regexEvents] has no fixed-58 branch — there `note1` takes a variable payload. */
|
||||
private val EVENT_FIXED_58_PREFIXES = emptyArray<String>()
|
||||
|
||||
private val EVENT_VARIABLE_PREFIXES = arrayOf("nevent1", "naddr1", "note1", "nrelay1", "nembed1")
|
||||
|
||||
/**
|
||||
* `\S` in the trailing group. Java's `\s` is the six ASCII whitespace chars unless
|
||||
* `UNICODE_CHARACTER_CLASS` is set, so U+00A0 and friends count as NON-space here.
|
||||
*/
|
||||
private fun isRegexSpace(c: Char): Boolean = c == ' ' || c == '\t' || c == '\n' || c == '\u000B' || c == '\u000C' || c == '\r'
|
||||
|
||||
/** ASCII-only case-insensitive prefix compare; [prefix] must be lowercase ASCII. */
|
||||
private fun matchesPrefixAt(
|
||||
content: String,
|
||||
offset: Int,
|
||||
prefix: String,
|
||||
): Boolean {
|
||||
if (offset + prefix.length > content.length) return false
|
||||
for (k in prefix.indices) {
|
||||
val c = content[offset + k]
|
||||
val folded = if (c in 'A'..'Z') c + 32 else c
|
||||
if (folded != prefix[k]) return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
/** End (exclusive) of the `[\S]*` run starting at [from]. */
|
||||
private fun endOfTrailing(
|
||||
content: String,
|
||||
from: Int,
|
||||
): Int {
|
||||
var j = from
|
||||
while (j < content.length && !isRegexSpace(content[j])) j++
|
||||
return j
|
||||
}
|
||||
|
||||
/**
|
||||
* True when one of the NIP-19 entity prefixes starts at [i].
|
||||
*
|
||||
@@ -165,26 +227,84 @@ object Nip19Parser {
|
||||
}
|
||||
|
||||
/**
|
||||
* Applies [regex] anchored at every NIP-19 candidate position in [content].
|
||||
* Matches `<prefix><bech32 payload><[\S]*>` at every NIP-19 candidate position in [content],
|
||||
* without ICU — see the block comment above [FIXED_58_PREFIXES] for why that matters.
|
||||
*
|
||||
* [isCandidateAt] covers the union of the prefixes across the three NIP-19
|
||||
* regexes, so a narrower [regex] simply fails `matchAt` on a prefix it does
|
||||
* not accept — still far cheaper than `findAll` restarting the engine at
|
||||
* every position in the string.
|
||||
* [isCandidateAt] covers the union of the prefixes across the three NIP-19 regexes, so a
|
||||
* caller passing a narrower prefix set simply finds no match at a prefix it does not accept.
|
||||
*
|
||||
* The prefixes are mutually exclusive (none is a prefix of another), so the first one that
|
||||
* matches decides the branch, exactly as the regex alternation did. A fixed-58 entity with
|
||||
* fewer than 58 payload chars fails outright rather than falling through to the variable
|
||||
* branch, again matching the regex: no variable prefix can equal a fixed-58 one.
|
||||
*/
|
||||
private inline fun forEachNip19Match(
|
||||
content: String,
|
||||
regex: Regex,
|
||||
action: (MatchResult) -> Unit,
|
||||
fixed58Prefixes: Array<String>,
|
||||
variablePrefixes: Array<String>,
|
||||
action: (type: String, key: String, additionalChars: String) -> Unit,
|
||||
) {
|
||||
var i = 0
|
||||
val len = content.length
|
||||
// Jump between 'n'/'N' with indexOf — an intrinsified, vectorised char search — instead of
|
||||
// testing every character. Both cases are tracked separately so each is only re-searched
|
||||
// once consumed, amortising to about one indexOf per candidate. Measured ~2x faster on
|
||||
// content with no entity at all, which is 94% of real notes (2588-note production sample),
|
||||
// and ~26% faster on the 767KB mention-heavy tail. Small mention-dense content (a ~700B
|
||||
// note with 2 mentions) is a few microseconds slower; that trade is deliberate.
|
||||
var nextLower = content.indexOf('n')
|
||||
var nextUpper = content.indexOf('N')
|
||||
while (i < len) {
|
||||
if (nextLower in 0..<i) nextLower = content.indexOf('n', i)
|
||||
if (nextUpper in 0..<i) nextUpper = content.indexOf('N', i)
|
||||
i =
|
||||
when {
|
||||
nextLower < 0 && nextUpper < 0 -> return
|
||||
nextLower < 0 -> nextUpper
|
||||
nextUpper < 0 -> nextLower
|
||||
else -> if (nextLower < nextUpper) nextLower else nextUpper
|
||||
}
|
||||
if (i >= len) return
|
||||
if (isCandidateAt(content, i)) {
|
||||
val match = regex.matchAt(content, i)
|
||||
if (match != null) {
|
||||
action(match)
|
||||
i = match.range.last + 1
|
||||
var end = -1
|
||||
|
||||
for (prefix in fixed58Prefixes) {
|
||||
if (!matchesPrefixAt(content, i, prefix)) continue
|
||||
val dataStart = i + prefix.length
|
||||
val dataEnd = dataStart + 58
|
||||
if (dataEnd <= len && allBech32(content, dataStart, dataEnd)) {
|
||||
end = endOfTrailing(content, dataEnd)
|
||||
action(
|
||||
content.substring(i, dataStart),
|
||||
content.substring(dataStart, dataEnd),
|
||||
content.substring(dataEnd, end),
|
||||
)
|
||||
}
|
||||
break
|
||||
}
|
||||
|
||||
if (end < 0) {
|
||||
for (prefix in variablePrefixes) {
|
||||
if (!matchesPrefixAt(content, i, prefix)) continue
|
||||
val dataStart = i + prefix.length
|
||||
var dataEnd = dataStart
|
||||
while (dataEnd < len && Bech32.isDataChar(content[dataEnd])) dataEnd++
|
||||
// `+` needs at least one payload char.
|
||||
if (dataEnd > dataStart) {
|
||||
end = endOfTrailing(content, dataEnd)
|
||||
action(
|
||||
content.substring(i, dataStart),
|
||||
content.substring(dataStart, dataEnd),
|
||||
content.substring(dataEnd, end),
|
||||
)
|
||||
}
|
||||
break
|
||||
}
|
||||
}
|
||||
|
||||
// `end` is exclusive and always past `i`, so the scan still advances.
|
||||
if (end >= 0) {
|
||||
i = end
|
||||
continue
|
||||
}
|
||||
}
|
||||
@@ -192,6 +312,17 @@ object Nip19Parser {
|
||||
}
|
||||
}
|
||||
|
||||
private fun allBech32(
|
||||
content: String,
|
||||
from: Int,
|
||||
to: Int,
|
||||
): Boolean {
|
||||
for (k in from until to) {
|
||||
if (!Bech32.isDataChar(content[k])) return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
/**
|
||||
* Scans [content] for NIP-19 entities.
|
||||
*
|
||||
@@ -208,14 +339,8 @@ object Nip19Parser {
|
||||
*/
|
||||
fun parseAll(content: String): List<Entity> {
|
||||
val returningList = mutableListOf<Entity>()
|
||||
forEachNip19Match(content, nip19regex) { matcher ->
|
||||
val type = matcher.groups[3]?.value ?: matcher.groups[5]?.value // npub1
|
||||
val key = matcher.groups[4]?.value ?: matcher.groups[6]?.value // bech32
|
||||
val additionalChars = matcher.groups[7]?.value // additional chars
|
||||
|
||||
if (type != null) {
|
||||
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
|
||||
}
|
||||
forEachNip19Match(content, FIXED_58_PREFIXES, VARIABLE_PREFIXES) { type, key, additionalChars ->
|
||||
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
|
||||
}
|
||||
return returningList
|
||||
}
|
||||
@@ -223,14 +348,8 @@ object Nip19Parser {
|
||||
/** Same scan as [parseAll], restricted to the event-ish entities. */
|
||||
fun parseAllEvents(content: String): List<Entity> {
|
||||
val returningList = mutableListOf<Entity>()
|
||||
forEachNip19Match(content, nip19regexEvents) { matcher ->
|
||||
val type = matcher.groups[2]?.value // nevent1
|
||||
val key = matcher.groups[3]?.value // bech32
|
||||
val additionalChars = matcher.groups[4]?.value // additional chars
|
||||
|
||||
if (type != null) {
|
||||
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
|
||||
}
|
||||
forEachNip19Match(content, EVENT_FIXED_58_PREFIXES, EVENT_VARIABLE_PREFIXES) { type, key, additionalChars ->
|
||||
parseComponents(type, key, additionalChars)?.entity?.let { returningList.add(it) }
|
||||
}
|
||||
return returningList
|
||||
}
|
||||
|
||||
+37
@@ -20,6 +20,8 @@
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip19Bech32.bech32
|
||||
|
||||
import kotlin.jvm.JvmField
|
||||
|
||||
/*
|
||||
* Copyright 2020 ACINQ SAS
|
||||
*
|
||||
@@ -81,15 +83,50 @@ object Bech32 {
|
||||
// char -> 5 bits value
|
||||
private val map = Array<Int5>(255) { -1 }
|
||||
|
||||
@PublishedApi
|
||||
internal const val DATA_CHARS_SIZE = 128
|
||||
|
||||
/**
|
||||
* Membership table for [isDataChar], kept separate from [map] because [map] is an
|
||||
* `Array<Byte>` — a boxed `java.lang.Byte[]` — and [isDataChar] runs per character over whole
|
||||
* note contents (up to ~767KB), where unboxing on every char would show up.
|
||||
*
|
||||
* `@JvmField` so callers of the inline [isDataChar] compile to a direct `getstatic` instead of
|
||||
* a property-getter `invokevirtual` per character (the same reasoning as `PointTypes`).
|
||||
* `@PublishedApi internal` because a public inline function cannot touch a private member.
|
||||
*/
|
||||
@PublishedApi
|
||||
@JvmField
|
||||
internal val DATA_CHARS = BooleanArray(DATA_CHARS_SIZE)
|
||||
|
||||
init {
|
||||
for (i in 0..ALPHABET.lastIndex) {
|
||||
map[ALPHABET[i].code] = i.toByte()
|
||||
DATA_CHARS[ALPHABET[i].code] = true
|
||||
}
|
||||
for (i in 0..ALPHABET_UPPERCASE.lastIndex) {
|
||||
map[ALPHABET_UPPERCASE[i].code] = i.toByte()
|
||||
DATA_CHARS[ALPHABET_UPPERCASE[i].code] = true
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* True when [c] is part of the bech32 data alphabet, in either case — i.e. everything except
|
||||
* `1`, `b`, `i` and `o`, which BIP-173 excludes as visually ambiguous.
|
||||
*
|
||||
* Exposed so scanners can find where an encoded payload *ends* without decoding it. The NIP-19
|
||||
* content scan needs exactly that on every ingested event, and cannot use a regex to do it:
|
||||
* Android's `java.util.regex` is ICU-backed and `Matcher.region()` copies the entire input into
|
||||
* native memory per call, which drove the app's native heap to ~1.9GB on a cold start.
|
||||
*
|
||||
* `inline` because it is called per character over whole note contents (up to ~767KB). As a
|
||||
* normal member it compiled to an `invokevirtual` per char, which measured ~1-2% slower across
|
||||
* the scan benchmark than the equivalent private lookup it replaced; inlining puts the constant
|
||||
* compare and the `baload` straight into the caller's loop and closes that gap.
|
||||
*/
|
||||
@Suppress("NOTHING_TO_INLINE")
|
||||
inline fun isDataChar(c: Char): Boolean = c.code < DATA_CHARS_SIZE && DATA_CHARS[c.code]
|
||||
|
||||
fun expand(hrp: String): Array<Int5> {
|
||||
val half = hrp.length + 1
|
||||
val size = half + hrp.length
|
||||
|
||||
+167
@@ -0,0 +1,167 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip10Notes.content
|
||||
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
|
||||
/**
|
||||
* Pins the ICU-free content scanners to the regexes they replaced.
|
||||
*
|
||||
* [findHashtags] and the `#[n]` walker used to anchor a fresh `Regex.matchAt` at every candidate.
|
||||
* On Android that goes through ICU, where `Matcher.region()` copies the whole input into native
|
||||
* memory per call — over 2588 real notes that was 43,626 Matchers copying 9.6GB, worst single note
|
||||
* 279MB. The scanners now match the grammars directly, so [hashtagSearch] and [tagSearch] survive
|
||||
* only as the specification, and this asserts the two agree.
|
||||
*
|
||||
* The corpus targets where a hand-rolled matcher is most likely to drift: the exact punctuation
|
||||
* set that ends a tag, ASCII-vs-Unicode whitespace before the `#`, non-ASCII inside the tag, and
|
||||
* the `+`/`[0-9]+` minimum-one-character rules.
|
||||
*/
|
||||
class ContentScanRegexEquivalenceTest {
|
||||
private fun referenceHashtags(content: String): List<String> {
|
||||
if (content.isBlank()) return emptyList()
|
||||
val out = mutableSetOf<String>()
|
||||
hashtagSearch.findAll(content).forEach { m ->
|
||||
val tag = m.groups[1]?.value
|
||||
if (tag != null && tag.isNotBlank()) out.add(tag)
|
||||
}
|
||||
return out.toList()
|
||||
}
|
||||
|
||||
private fun referenceIndexTags(
|
||||
content: String,
|
||||
tags: Array<Array<String>>,
|
||||
wanted: String,
|
||||
): Set<String> {
|
||||
val out = mutableSetOf<String>()
|
||||
tagSearch.findAll(content).forEach { m ->
|
||||
try {
|
||||
val tag = m.groups[1]?.value?.let { tags[it.toInt()] }
|
||||
if (tag != null && tag.size > 1 && tag[0] == wanted) out.add(tag[1])
|
||||
} catch (e: Exception) {
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
private val tagArray =
|
||||
arrayOf(
|
||||
arrayOf("p", "pubkey0"),
|
||||
arrayOf("e", "event1"),
|
||||
arrayOf("a", "addr2"),
|
||||
arrayOf("p", "pubkey3"),
|
||||
arrayOf("t", "topic4"),
|
||||
)
|
||||
|
||||
private fun corpus(): List<String> =
|
||||
buildList {
|
||||
add("")
|
||||
add(" ")
|
||||
add("#")
|
||||
add("#tag")
|
||||
add("hello #tag world")
|
||||
add("a#tag")
|
||||
add("#tag#other")
|
||||
add("#tag #other")
|
||||
add("##tag")
|
||||
add("#tag.")
|
||||
add("#tag, and #more!")
|
||||
add("#tag's")
|
||||
add("#tag\"quoted\"")
|
||||
add("#a")
|
||||
add("#1")
|
||||
add("#tag-with-dash")
|
||||
add("#tag_with_underscore")
|
||||
add("#tag~tilde|pipe\\back`tick")
|
||||
// non-ASCII is valid tag content: the regex class and its \s are ASCII-only
|
||||
add("#café")
|
||||
add("#日本語")
|
||||
add("#tagéè")
|
||||
// ASCII vs Unicode whitespace BEFORE the # decides whether it matches at all
|
||||
add("x\u00A0#tag")
|
||||
add("x\u2003#tag")
|
||||
add("x\t#tag")
|
||||
add("x\n#tag")
|
||||
add("x\r#tag")
|
||||
// Unicode whitespace INSIDE the tag is valid to the regex but blank to Kotlin
|
||||
add("#\u00A0")
|
||||
add("#\u00A0x")
|
||||
// every excluded char must terminate the tag
|
||||
for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#tag${c}more")
|
||||
for (c in "!@#$%^&*()=+./,[{]};:'\"?><") add("#$c")
|
||||
// index tags
|
||||
add("#[0]")
|
||||
add("#[1] and #[2]")
|
||||
add("look #[3] here")
|
||||
add("#[]")
|
||||
add("#[abc]")
|
||||
add("#[99]")
|
||||
add("#[0")
|
||||
add("#[0]]")
|
||||
add("x#[0]")
|
||||
add("x\u00A0#[0]")
|
||||
add("#[0]#[1]")
|
||||
add("#[00]")
|
||||
// mixed
|
||||
add("#tag #[0] #other #[1]")
|
||||
add("lorem ipsum ".repeat(500) + "#tail")
|
||||
add("#head" + " dolor sit ".repeat(500))
|
||||
add("no hashes at all here ".repeat(200))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun hashtagsMatchRegex() {
|
||||
corpus().forEach { c ->
|
||||
assertEquals(
|
||||
referenceHashtags(c).sorted(),
|
||||
findHashtags(c).sorted(),
|
||||
"findHashtags diverged on: ${c.take(80)}",
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun indexTagsMatchRegex() {
|
||||
corpus().forEach { c ->
|
||||
assertEquals(
|
||||
referenceIndexTags(c, tagArray, "p").sorted(),
|
||||
findIndexTagsWithPeople(c, tagArray).sorted(),
|
||||
"findIndexTagsWithPeople diverged on: ${c.take(80)}",
|
||||
)
|
||||
val refEv = referenceIndexTags(c, tagArray, "e") + referenceIndexTags(c, tagArray, "a")
|
||||
assertEquals(
|
||||
refEv.sorted(),
|
||||
findIndexTagsWithEventsOrAddresses(c, tagArray).sorted(),
|
||||
"findIndexTagsWithEventsOrAddresses diverged on: ${c.take(80)}",
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun hashtagsMatchRegexAtEveryOffset() {
|
||||
val filler = "a b\tc\nd "
|
||||
for (i in 0..filler.length) {
|
||||
val c = filler.substring(0, i) + "#tag" + filler.substring(i)
|
||||
assertEquals(referenceHashtags(c).sorted(), findHashtags(c).sorted(), "offset $i")
|
||||
}
|
||||
}
|
||||
}
|
||||
+202
@@ -0,0 +1,202 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip19Bech32
|
||||
|
||||
import com.vitorpamplona.quartz.nip19Bech32.entities.Entity
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
|
||||
/**
|
||||
* Pins the ICU-free scanner in [Nip19Parser] to the regexes it replaced.
|
||||
*
|
||||
* The scan used to run `Regex.matchAt` at every candidate position. On Android that goes through
|
||||
* ICU, and `Matcher.region()` copies the whole input into native memory on each call — one full
|
||||
* native copy of a note's content *per candidate* — which drove the native heap to ~1.9GB on a
|
||||
* cold start and got the process lmkd-killed. The scanner matches the grammar directly instead.
|
||||
*
|
||||
* [Nip19Parser.nip19regex] and [Nip19Parser.nip19regexEvents] are still the specification, so this
|
||||
* runs both over the same corpus and requires identical entity lists. The corpus deliberately
|
||||
* targets the places a hand-rolled matcher is most likely to drift from the regex: the exact-58
|
||||
* payload boundary, the bech32 alphabet's excluded characters, ASCII-only case folding, and which
|
||||
* characters `[\S]*` is willing to swallow.
|
||||
*/
|
||||
class Nip19ScannerRegexEquivalenceTest {
|
||||
companion object {
|
||||
const val NPUB = "npub1hv7k2s755n697sptva8vkh9jz40lzfzklnwj6ekewfmxp5crwdjs27007y"
|
||||
const val NOTE = "note1stqea6wmwezg9x6yyr6qkukw95ewtdukyaztycws65l8wppjmtpscawevv"
|
||||
const val NEVENT = "nevent1qqs0tsw8hjacs4fppgdg7f5yhgwwfkyua4xcs3re9wwkpkk2qeu6mhql22rcy"
|
||||
|
||||
/** 58 valid bech32 chars, so `npub1` + this is exactly the fixed-length branch. */
|
||||
const val PAYLOAD58 = "qpzry9x8gf2tvdw0s3jn54khce6mua7lqpzry9x8gf2tvdw0s3jn54khce"
|
||||
}
|
||||
|
||||
/** What `parseAll` did before: drive the regex from every position with `findAll`. */
|
||||
private fun referenceParseAll(content: String): List<Entity> {
|
||||
val out = mutableListOf<Entity>()
|
||||
Nip19Parser.nip19regex.findAll(content).forEach { m ->
|
||||
val type = m.groups[3]?.value ?: m.groups[5]?.value
|
||||
val key = m.groups[4]?.value ?: m.groups[6]?.value
|
||||
val additionalChars = m.groups[7]?.value
|
||||
if (type != null) {
|
||||
Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) }
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
private fun referenceParseAllEvents(content: String): List<Entity> {
|
||||
val out = mutableListOf<Entity>()
|
||||
Nip19Parser.nip19regexEvents.findAll(content).forEach { m ->
|
||||
val type = m.groups[2]?.value
|
||||
val key = m.groups[3]?.value
|
||||
val additionalChars = m.groups[4]?.value
|
||||
if (type != null) {
|
||||
Nip19Parser.parseComponents(type, key, additionalChars)?.entity?.let { out.add(it) }
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
private fun assertSameAsRegex(content: String) {
|
||||
assertEquals(
|
||||
referenceParseAll(content),
|
||||
Nip19Parser.parseAll(content),
|
||||
"parseAll diverged from nip19regex on: ${content.take(90)}",
|
||||
)
|
||||
assertEquals(
|
||||
referenceParseAllEvents(content),
|
||||
Nip19Parser.parseAllEvents(content),
|
||||
"parseAllEvents diverged from nip19regexEvents on: ${content.take(90)}",
|
||||
)
|
||||
}
|
||||
|
||||
private fun corpus(): List<String> =
|
||||
buildList {
|
||||
// plain placement
|
||||
add("")
|
||||
add(NPUB)
|
||||
add("hello $NPUB world")
|
||||
add("nostr:$NPUB")
|
||||
add("@$NPUB")
|
||||
add("nostr:@$NPUB")
|
||||
add("prefix-nostr:$NPUB-suffix")
|
||||
add(NOTE)
|
||||
add(NEVENT)
|
||||
|
||||
// adjacency and repetition — where scan-resume position matters
|
||||
add(NPUB + NEVENT)
|
||||
add("$NPUB $NEVENT")
|
||||
add("$NPUB\n$NEVENT")
|
||||
add("$NPUB,$NEVENT")
|
||||
add(listOf(NPUB, NOTE, NEVENT).joinToString(" "))
|
||||
add(NPUB.repeat(3))
|
||||
|
||||
// the exact-58 boundary for npub/nsec/note
|
||||
add("npub1" + PAYLOAD58)
|
||||
add("npub1" + PAYLOAD58.dropLast(1)) // 57 -> must not match
|
||||
add("npub1" + PAYLOAD58 + "q") // 59 -> 58 key, trailing takes the rest
|
||||
add("npub1" + PAYLOAD58 + " tail")
|
||||
add("note1" + PAYLOAD58)
|
||||
add("nsec1" + PAYLOAD58)
|
||||
|
||||
// bech32 alphabet: 1, b, i, o are excluded and must terminate the payload
|
||||
add("nevent1qqs1qqs")
|
||||
add("nevent1qqsbqqs")
|
||||
add("nevent1qqsiqqs")
|
||||
add("nevent1qqsoqqs")
|
||||
|
||||
// A *valid* variable-length entity butted straight against an excluded char. The
|
||||
// payload has to stop there and still decode. These are the cases with teeth: a
|
||||
// charset that wrongly accepted b/i/o/1 would swallow the extra char, fail the
|
||||
// bech32 decode and silently drop the entity — whereas cases whose payload is
|
||||
// invalid either way agree trivially and prove nothing.
|
||||
for (excluded in listOf("b", "i", "o", "1")) {
|
||||
add(NEVENT + excluded)
|
||||
add(NEVENT + excluded + "xyz")
|
||||
add("$NEVENT$excluded more text")
|
||||
}
|
||||
add("nevent1") // variable branch needs >= 1 payload char
|
||||
add("nprofile1")
|
||||
add("naddr1q")
|
||||
|
||||
// ASCII-only case folding
|
||||
add(NPUB.uppercase())
|
||||
add("NOSTR:" + NPUB.uppercase())
|
||||
add(NEVENT.uppercase())
|
||||
add("nPuB1" + PAYLOAD58)
|
||||
// U+212A KELVIN SIGN folds to 'k' under Unicode rules but NOT under the regex's
|
||||
// ASCII-only CASE_INSENSITIVE; both sides must reject it.
|
||||
add("npub1" + PAYLOAD58.replaceFirst("k", "K"))
|
||||
|
||||
// what [\S]* may swallow: Java's \s is the six ASCII whitespace chars only,
|
||||
// so U+00A0 and U+2003 are NON-space and belong to the trailing group.
|
||||
add("$NPUB\u00A0more")
|
||||
add("$NPUB\u2003more")
|
||||
add("${NPUB}more")
|
||||
add("${NPUB}1more")
|
||||
add("$NPUB\tmore")
|
||||
add("$NPUB\rmore")
|
||||
|
||||
// near-misses that must not be mistaken for entities
|
||||
add("n")
|
||||
add("nn")
|
||||
add("np")
|
||||
add("no")
|
||||
add("nostr:")
|
||||
add("nothing to see here")
|
||||
add("a note about nothing")
|
||||
add("nopqrstuvwxyz")
|
||||
add("x$NPUB")
|
||||
add("1$NPUB")
|
||||
|
||||
// long content with the entity at the far end (the 767KB-tail shape, scaled down)
|
||||
add("lorem ipsum ".repeat(2000) + NPUB)
|
||||
add(NPUB + " " + "dolor sit amet ".repeat(2000))
|
||||
// many 'n' candidates but no entities — the scanner's rejection path
|
||||
add("neither nor none never nothing ".repeat(500))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun matchesRegexAcrossCorpus() {
|
||||
corpus().forEach { assertSameAsRegex(it) }
|
||||
}
|
||||
|
||||
@Test
|
||||
fun matchesRegexWithEntityAtEveryOffset() {
|
||||
// Slides the entity through a filler string so every start offset, including
|
||||
// immediately after another candidate 'n', is exercised.
|
||||
val filler = "n no non nost nostr "
|
||||
for (i in 0..filler.length) {
|
||||
assertSameAsRegex(filler.substring(0, i) + NPUB + filler.substring(i))
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun matchesRegexOnTruncatedPayloads() {
|
||||
// Every truncation of a real entity: catches off-by-one at the 58 boundary and in `+`.
|
||||
for (entity in listOf(NPUB, NOTE, NEVENT)) {
|
||||
for (len in 1..entity.length) {
|
||||
assertSameAsRegex(entity.substring(0, len))
|
||||
assertSameAsRegex("text " + entity.substring(0, len) + " text")
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
+67
@@ -0,0 +1,67 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.nip19Bech32.bech32
|
||||
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertFalse
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* [Bech32.isDataChar] is the membership test the NIP-19 content scan uses to find where an encoded
|
||||
* payload ends, so it has to agree with [Bech32.ALPHABET] exactly — a char wrongly accepted extends
|
||||
* a payload past its real end and silently drops the entity when the decode then fails.
|
||||
*/
|
||||
class Bech32DataCharTest {
|
||||
@Test
|
||||
fun acceptsExactlyTheAlphabetInBothCases() {
|
||||
for (c in Bech32.ALPHABET) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char")
|
||||
for (c in Bech32.ALPHABET_UPPERCASE) assertTrue(Bech32.isDataChar(c), "expected '$c' to be a data char")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun rejectsTheAmbiguousFour() {
|
||||
// BIP-173 leaves these out of the alphabet precisely because they are easy to misread.
|
||||
for (c in "1bio1BIO") assertFalse(Bech32.isDataChar(c), "expected '$c' to be rejected")
|
||||
}
|
||||
|
||||
@Test
|
||||
fun agreesWithTheAlphabetAcrossEveryChar() {
|
||||
// Sweeps the whole BMP so nothing outside the alphabet sneaks in — including the
|
||||
// out-of-range guard for chars beyond the lookup table.
|
||||
val expected = (Bech32.ALPHABET + Bech32.ALPHABET_UPPERCASE).toSet()
|
||||
for (code in 0..0xFFFF) {
|
||||
val c = code.toChar()
|
||||
assertEquals(c in expected, Bech32.isDataChar(c), "disagreement at code $code")
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun countsMatchTheSpec() {
|
||||
assertEquals(32, Bech32.ALPHABET.length)
|
||||
// 32 symbols, but only the 23 letters have a distinct uppercase form — the 9 digits
|
||||
// are the same char in both alphabets, so the accepted set is 23*2 + 9, not 64.
|
||||
val letters = Bech32.ALPHABET.count { it.isLetter() }
|
||||
val digits = Bech32.ALPHABET.count { it.isDigit() }
|
||||
assertEquals(32, letters + digits)
|
||||
assertEquals(letters * 2 + digits, (0..0xFFFF).count { Bech32.isDataChar(it.toChar()) })
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user