From 66fb5d6f856413e8c0cd5480838541094b2bbf62 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 28 Feb 2026 00:59:57 +0000 Subject: [PATCH] perf(commons/richtext): eliminate string splits in RichTextParser Replace content.split('\n') and paragraph.split(' ') with index-based scanning so no intermediate List or line-substring allocations are created during paragraph/word tokenization. Key changes to findTextSegments: - Scan for '\n' via indexOf to determine line boundaries without allocating a List or any line substrings. - Compute trimEnd boundary in-place (no trimEnd() String copy). - For each line, do a first pass with isRegularByIndex() that checks every word using character-level inspection without creating substrings. If all words are plain text the entire paragraph is represented as a single RegularTextSegment(content.substring(lineStart, actualEnd)), avoiding M per-word String allocations and the later joinToString. - Only when a paragraph contains special tokens (URLs, hashtags, mentions, emoji, etc.) fall back to the existing wordIdentifier() path, which produces individual Segment objects as before. - Short-circuit the final map pass for single-segment paragraphs (already optimal) to skip an unnecessary joinToString + object creation. New helpers: - isRegularByIndex(): character-level fast path that detects all token types (http/https, lnbc, cashu, NIP-19, emoji, email, phone, schemeless URL, EmojiCoder variation selectors) without creating a substring. Returns true only when the word is definitely plain text. - isArabicInRange(): RTL detection directly on the source string range, replacing the substring-based isArabic(). Behaviour is identical: the same Segment types are produced in the same order for every input, as validated by the existing test suite. https://claude.ai/code/session_01NXsow7yLBd6ModmGSsqR9g --- .../commons/richtext/RichTextParser.kt | 189 ++++++++++++++++-- 1 file changed, 171 insertions(+), 18 deletions(-) diff --git a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt index 615d64ce89..33c302ef82 100644 --- a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt +++ b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/richtext/RichTextParser.kt @@ -169,36 +169,191 @@ class RichTextParser { emojis: Map, tags: ImmutableListOfLists, ): ImmutableList { - val lines = content.split('\n') - val paragraphSegments = ArrayList(lines.size) + val paragraphSegments = ArrayList() + val contentLength = content.length + var lineStart = 0 - lines.forEach { paragraph -> - val isRTL = isArabic(paragraph) + // Scan for newlines manually to avoid allocating a List from split('\n') + // and avoid allocating each paragraph as a substring. + while (lineStart <= contentLength) { + val nlIdx = content.indexOf('\n', lineStart) + val lineEnd = if (nlIdx < 0) contentLength else nlIdx - val wordList = paragraph.trimEnd().split(' ') - val segments = ArrayList(wordList.size) - wordList.forEach { word -> - segments.add(wordIdentifier(word, images, videos, urls, emojis, tags)) + val isRTL = isArabicInRange(content, lineStart, lineEnd) + + // Compute trimEnd boundary without creating a substring (matches trimEnd().split(' ')). + var actualEnd = lineEnd + while (actualEnd > lineStart && content[actualEnd - 1].isWhitespace()) actualEnd-- + + val paragraphState: ParagraphState + if (actualEnd <= lineStart) { + // Empty / all-whitespace line: mirror "".split(' ') == [""] from the original. + paragraphState = ParagraphState(persistentListOf(RegularTextSegment("")), isRTL) + } else { + // First pass: check every word without creating substrings. + // For the common all-plain-text paragraph this avoids all per-word allocations. + var allRegular = true + var tempWordStart = lineStart + while (tempWordStart < actualEnd) { + val spIdx = content.indexOf(' ', tempWordStart) + val tempWordEnd = if (spIdx < 0 || spIdx >= actualEnd) actualEnd else spIdx + if (!isRegularByIndex(content, tempWordStart, tempWordEnd, emojis)) { + allRegular = false + break + } + tempWordStart = tempWordEnd + 1 + } + + paragraphState = + if (allRegular) { + // Entire paragraph is plain text: one substring for the whole trimmed line, + // skipping per-word allocations and the joinToString done later. + ParagraphState( + persistentListOf(RegularTextSegment(content.substring(lineStart, actualEnd))), + isRTL, + ) + } else { + // Mixed paragraph: classify each word with the full wordIdentifier. + val segments = ArrayList() + var wordStart = lineStart + while (wordStart < actualEnd) { + val spIdx = content.indexOf(' ', wordStart) + val wordEnd = if (spIdx < 0 || spIdx >= actualEnd) actualEnd else spIdx + segments.add(wordIdentifier(content.substring(wordStart, wordEnd), images, videos, urls, emojis, tags)) + wordStart = wordEnd + 1 + } + ParagraphState(segments.toPersistentList(), isRTL) + } } - paragraphSegments.add(ParagraphState(segments.toPersistentList(), isRTL)) + paragraphSegments.add(paragraphState) + lineStart = lineEnd + 1 } val segmentsWithGalleries = GalleryParser().processParagraphs(paragraphSegments) return segmentsWithGalleries .map { paragraph -> - if (paragraph.words.isEmpty() || paragraph.words.any { it !is RegularTextSegment }) { - paragraph - } else { - ParagraphState( - persistentListOf(RegularTextSegment(paragraph.words.joinToString(" ") { it.segmentText })), - paragraph.isRTL, - ) + when { + paragraph.words.isEmpty() -> paragraph + // Single segment: already optimal, no join needed. + paragraph.words.size == 1 -> paragraph + paragraph.words.any { it !is RegularTextSegment } -> paragraph + else -> + ParagraphState( + persistentListOf(RegularTextSegment(paragraph.words.joinToString(" ") { it.segmentText })), + paragraph.isRTL, + ) } }.toImmutableList() } + /** + * Returns true when the word at content[wordStart, wordEnd) is definitely plain text and + * does not need a substring to be created for classification. Conservative: a false return + * only means the caller should run the full [wordIdentifier] check; it never produces a + * wrong result. + */ + private fun isRegularByIndex( + content: String, + wordStart: Int, + wordEnd: Int, + emojis: Map, + ): Boolean { + val len = wordEnd - wordStart + if (len == 0) return true + + val c0 = content[wordStart] + + // Quick first-character reject for token types that always start with a known char. + when (c0) { + '#' -> return false // hashtag (#hashtag) or tag reference (#[n]) + '@' -> return false // @npub… NIP-19 mention or email starting with @ + 'd', 'D' -> if (len > 11 && content.startsWith("data:image/", wordStart)) return false + 'l', 'L' -> { + if (len > 4 && content.startsWith("lnbc", wordStart, ignoreCase = true)) return false + if (len > 5 && content.startsWith("lnurl", wordStart, ignoreCase = true)) return false + } + 'c', 'C' -> { + if (len > 6 && + ( + content.startsWith("cashuA", wordStart, ignoreCase = true) || + content.startsWith("cashuB", wordStart, ignoreCase = true) + ) + ) return false + } + 'n', 'N' -> { + // nostr: prefix or NIP-19 bech32 schemes (npub1, note1, naddr1, nevent1, nprofile1, nembed) + if (len >= 5) { + if (content.startsWith("nostr:", wordStart, ignoreCase = true)) return false + if (wordStart + 1 < wordEnd) { + when (content[wordStart + 1]) { + 'p', 'P' -> + if (content.startsWith("npub1", wordStart, ignoreCase = true) || + content.startsWith("nprofile1", wordStart, ignoreCase = true) + ) return false + 'o', 'O' -> if (content.startsWith("note1", wordStart, ignoreCase = true)) return false + 'a', 'A' -> if (content.startsWith("naddr1", wordStart, ignoreCase = true)) return false + 'e', 'E' -> + if (content.startsWith("nevent1", wordStart, ignoreCase = true) || + content.startsWith("nembed", wordStart, ignoreCase = true) + ) return false + } + } + } + } + 'h', 'H' -> { + // Only http(s):// words can be in the URL sets (parseValidUrls filters by HTTPRegex). + if (len >= 7 && + ( + content.startsWith("http://", wordStart, ignoreCase = true) || + content.startsWith("https://", wordStart, ignoreCase = true) + ) + ) return false + } + } + + // Single-pass character scan for special markers. + var isPotentialPhone = len in 7..14 + var hasMidPeriod = false + for (i in wordStart until wordEnd) { + val c = content[i] + val code = c.code + when { + // Custom emoji uses :name: format; fastMightContainEmoji checks for ':'. + c == ':' && emojis.isNotEmpty() -> return false + // Email address + c == '@' -> return false + // Possible schemeless URL (domain.tld): period not at first or last position + c == '.' && i > wordStart && i < wordEnd - 1 -> hasMidPeriod = true + // EmojiCoder: Unicode variation selectors BMP range U+FE00..U+FE0F + code in 0xFE00..0xFE0F -> return false + // EmojiCoder: high surrogate 0xDB40 leads a variation-selector supplement codepoint + code == 0xDB40 -> return false + // Phone number: allowed chars are digits, '-', ' ', '.' + isPotentialPhone && c !in '0'..'9' && c != '-' && c != ' ' && c != '.' -> isPotentialPhone = false + } + } + + // Let wordIdentifier confirm whether these are actually phone / schemeless-URL. + if (isPotentialPhone) return false + if (hasMidPeriod) return false + + return true + } + + private fun isArabicInRange( + content: String, + start: Int, + end: Int, + ): Boolean { + for (i in start until end) { + val c = content[i] + if (c in '\u0600'..'\u06FF' || c in '\u0750'..'\u077F') return true + } + return false + } + private fun isNumber(word: String) = numberPattern.matches(word) private fun isPhoneNumberChar(c: Char): Boolean = @@ -225,8 +380,6 @@ class RichTextParser { fun isDate(word: String): Boolean = shortDatePattern.matches(word) || longDatePattern.matches(word) - private fun isArabic(text: String): Boolean = text.any { it in '\u0600'..'\u06FF' || it in '\u0750'..'\u077F' } - private fun wordIdentifier( word: String, images: Set,