From 3f3486c7c2a2772728655a195febeba63fc05da4 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 4 Oct 2026 15:43:31 +0000 Subject: [PATCH] fix(quartz): drop spaced URLs, spaced addresses and imeta from list search text Audit of the natural-language tag indexing against 1.5M list events: - A value containing whitespace skipped every machine check, so feed URLs with unescaped spaces (url, artwork, feedUrl, t: ~1,400 values) and an address with a spaced `d` were indexed. A `scheme://` prefix and an address prefix now mark a value as machine data with or without spaces. - NIP-92 `imeta` values are space-separated `key value` pairs, so every URL, mime type and hash in them would pass; the tag is skipped. - The classifier is now a single scan for nearly every value: an ASCII single token is decided by the capitalized-word test alone. 316 -> 150 ns per value on the corpus. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_016XJSf4ekFVuhSj5m8eeNsP --- .../decentralizedLists/SearchExt.kt | 7 +- .../nip50Search/NaturalLanguageValue.kt | 88 +++++++++++++------ .../DecentralizedListsTest.kt | 2 + .../nip50Search/NaturalLanguageValueTest.kt | 9 +- 4 files changed, 78 insertions(+), 28 deletions(-) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt index 5aaefaf5f4..89d91269ec 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt @@ -40,6 +40,7 @@ import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor import com.vitorpamplona.quartz.nip50Search.isMachineValue import com.vitorpamplona.quartz.nip50Search.isNaturalLanguageValue import com.vitorpamplona.quartz.nip89AppHandlers.clientTag.ClientTag +import com.vitorpamplona.quartz.nip92IMeta.IMetaTag /** * The human-authored text of any kind in the family, in a fixed order: the header's `names` @@ -100,9 +101,10 @@ fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean * `relationshipType` and tags nobody has named yet — so this walk decides by what a value looks * like rather than by the tag it sits in. * - * Skips the tags [forEachSearchableListField] already visits, plus two NIP-defined metadata tags + * Skips the tags [forEachSearchableListField] already visits, plus the NIP-defined metadata tags * that are not the event's own text: `alt` (NIP-31 fallback text, which here restates `title` - * and `artist` behind a fixed "Song: … by …" prefix) and `client` (NIP-89 app name). + * and `artist` behind a fixed "Song: … by …" prefix), `client` (NIP-89 app name) and `imeta` + * (NIP-92 `key value` pairs, whose space would otherwise pass every URL and hash as text). * * @return false when the visitor stopped the walk. */ @@ -128,6 +130,7 @@ private fun isExtraFieldTagName(name: String) = HashtagTag.TAG_NAME, AltTag.TAG_NAME, ClientTag.TAG_NAME, + IMetaTag.TAG_NAME, -> false else -> true diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt index 328f09df10..8e436ee138 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt @@ -30,27 +30,39 @@ package com.vitorpamplona.quartz.nip50Search * tokens that label data rather than describe it: `music`, `eng`, `openlibrary`, `IS_A_PROPERTY_OF`, * `concept-header`, `OL19722168W`, `2026-06-01`. * - * Allocation-free: the read path calls it once per value per event per keystroke. + * Allocation-free, and a single scan for nearly every value: the read path calls it once per + * value per event per keystroke. */ fun isNaturalLanguageValue(value: String): Boolean { var start = 0 var end = value.length while (start < end && value[start].isWhitespace()) start++ while (end > start && value[end - 1].isWhitespace()) end-- - if (start == end || isMachineValue(value, start, end)) return false + if (start == end || isStructuredValue(value, start, end)) return false + var nonAscii = false for (i in start until end) { val c = value[i] - if (c.isWhitespace() || c.code > 127) return true + // The token shapes in isMachineToken cannot contain whitespace. + if (c.isWhitespace()) return true + if (c.code > 127) nonAscii = true + } + + // A single token. A capitalized letters-only word can only collide with a hex id, so the + // other machine checks are needed just for the rare non-ASCII token, such as an IRI. + return if (nonAscii) { + !isMachineToken(value, start, end) + } else { + isCapitalizedWord(value, start, end) && !isHexId(value, start, end) } - return isCapitalizedWord(value, start, end) } /** * Whether a tag value is an identifier or a structure rather than text: a JSON object or array, - * a number (including a signed one and an ISBN-10 ending in `X`), a URI of any scheme - * (`https://…`, `wss://…`, `tag:…`, `urn:…`, `isbn:…`), an event address (`kind:pubkey:d`), a - * long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine + * a URL (`https://…`, `wss://…`, even with unescaped spaces in its path), an event address + * (`kind:pubkey:d`, even with spaces in `d`), or, as a single token, a number (including a + * signed one and an ISBN-10 ending in `X`), a URI of any scheme (`tag:…`, `urn:…`, `isbn:…`), + * a long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine * values too: there is nothing in them to index. */ fun isMachineValue(value: String): Boolean { @@ -58,10 +70,13 @@ fun isMachineValue(value: String): Boolean { var end = value.length while (start < end && value[start].isWhitespace()) start++ while (end > start && value[end - 1].isWhitespace()) end-- - return start == end || isMachineValue(value, start, end) + if (start == end || isStructuredValue(value, start, end)) return true + for (i in start until end) if (value[i].isWhitespace()) return false + return isMachineToken(value, start, end) } -private fun isMachineValue( +/** The machine shapes recognizable from their first characters, whitespace or not. */ +private fun isStructuredValue( v: String, start: Int, end: Int, @@ -69,17 +84,20 @@ private fun isMachineValue( val first = v[start] val last = v[end - 1] if ((first == '{' && last == '}') || (first == '[' && last == ']')) return true + return isUrl(v, start, end) || isAddress(v, start, end) +} - // Every other shape is a single token. - for (i in start until end) if (v[i].isWhitespace()) return false - - return isNumber(v, start, end) || +/** The machine shapes that are a single token. */ +private fun isMachineToken( + v: String, + start: Int, + end: Int, +): Boolean = + isNumber(v, start, end) || isUri(v, start, end) || - isAddress(v, start, end) || isHexId(v, start, end) || isUuid(v, start, end) || isBech32(v, start) -} private fun Char.isAsciiDigit() = this in '0'..'9' @@ -108,21 +126,41 @@ private fun isNumber( return i == end } -/** An RFC 3986 scheme followed by `:` and something more, as a single token. */ +/** The index of the `:` that ends a leading RFC 3986 scheme, or -1 when there is none. */ +private fun schemeEnd( + v: String, + start: Int, + end: Int, +): Int { + if (!v[start].isAsciiLetter()) return -1 + var i = start + 1 + while (i < end) { + val c = v[i] + if (c == ':') return i + if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return -1 + i++ + } + return -1 +} + +/** A scheme followed by `:` and something more, as a single token. */ private fun isUri( v: String, start: Int, end: Int, ): Boolean { - if (!v[start].isAsciiLetter()) return false - var i = start + 1 - while (i < end) { - val c = v[i] - if (c == ':') return i + 1 < end - if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return false - i++ - } - return false + val colon = schemeEnd(v, start, end) + return colon >= 0 && colon + 1 < end +} + +/** A scheme followed by `://`. Unlike "Song: Gold", never text. */ +private fun isUrl( + v: String, + start: Int, + end: Int, +): Boolean { + val colon = schemeEnd(v, start, end) + return colon >= 0 && colon + 2 < end && v[colon + 1] == '/' && v[colon + 2] == '/' } /** `:<64-hex pubkey>` optionally followed by `:`. */ diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt index 0522964b34..7853830fa7 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt @@ -340,6 +340,8 @@ class DecentralizedListsTest { arrayOf("e", headerId, "wss://relay.example.com", "mention"), arrayOf("alt", "Book: Herfsttij der Middeleeuwen by Johan Huizinga"), arrayOf("client", "Brainstorm"), + arrayOf("imeta", "url https://example.com/cover.jpg", "m image/jpeg", "alt A book cover"), + arrayOf("artwork", "https://example.com/Cover Art.jpg"), ), ), ) diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt index 1dd4c9aa55..ee79e85d3f 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt @@ -76,6 +76,11 @@ class NaturalLanguageValueTest { "{\"word\": {\"slug\": \"ol-ol98624w\", \"name\": \"Waking with Enemies\"}}", "[\"title\", \"artist\"]", "Re:Zero", + // URLs and addresses with unescaped spaces, as feeds publish them + "https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg", + "https://headstarts.uk/msp/longy/Katherines Wheel/katherines wheel.xml", + "39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid", + "ABCDEFABCDEFABCDEFABCDEFABCDEFAB", ).forEach { assertFalse(isNaturalLanguageValue(it), it) } } @@ -97,13 +102,15 @@ class NaturalLanguageValueTest { "c65d75f4d058f4746e12e44441cacea0", "8ad7c296-67e9-5f57-ba52-2ff732e62e87", "npub1f5pre6wl6ad87vr4hr5wppqq30sh58m4p33mthnjreh03qadcajs7gwt3z", + "https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg", + "39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid", "", ).forEach { assertTrue(isMachineValue(it), it) } } @Test fun wordsAndPhrasesAreNotMachineValues() { - listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell").forEach { + listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell", "{ राधेय }Rishi Verma").forEach { assertFalse(isMachineValue(it), it) } }