fix(quartz): drop spaced URLs, spaced addresses and imeta from list search text

Audit of the natural-language tag indexing against 1.5M list events:

- A value containing whitespace skipped every machine check, so feed URLs
  with unescaped spaces (url, artwork, feedUrl, t: ~1,400 values) and an
  address with a spaced `d` were indexed. A `scheme://` prefix and an
  address prefix now mark a value as machine data with or without spaces.
- NIP-92 `imeta` values are space-separated `key value` pairs, so every URL,
  mime type and hash in them would pass; the tag is skipped.
- The classifier is now a single scan for nearly every value: an ASCII
  single token is decided by the capitalized-word test alone. 316 -> 150 ns
  per value on the corpus.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_016XJSf4ekFVuhSj5m8eeNsP
This commit is contained in:
Claude
2026-10-04 15:43:31 +00:00
parent 02adb57e21
commit 3f3486c7c2
4 changed files with 78 additions and 28 deletions
@@ -40,6 +40,7 @@ import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor
import com.vitorpamplona.quartz.nip50Search.isMachineValue
import com.vitorpamplona.quartz.nip50Search.isNaturalLanguageValue
import com.vitorpamplona.quartz.nip89AppHandlers.clientTag.ClientTag
import com.vitorpamplona.quartz.nip92IMeta.IMetaTag
/**
* The human-authored text of any kind in the family, in a fixed order: the header's `names`
@@ -100,9 +101,10 @@ fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean
* `relationshipType` and tags nobody has named yet — so this walk decides by what a value looks
* like rather than by the tag it sits in.
*
* Skips the tags [forEachSearchableListField] already visits, plus two NIP-defined metadata tags
* Skips the tags [forEachSearchableListField] already visits, plus the NIP-defined metadata tags
* that are not the event's own text: `alt` (NIP-31 fallback text, which here restates `title`
* and `artist` behind a fixed "Song: … by …" prefix) and `client` (NIP-89 app name).
* and `artist` behind a fixed "Song: … by …" prefix), `client` (NIP-89 app name) and `imeta`
* (NIP-92 `key value` pairs, whose space would otherwise pass every URL and hash as text).
*
* @return false when the visitor stopped the walk.
*/
@@ -128,6 +130,7 @@ private fun isExtraFieldTagName(name: String) =
HashtagTag.TAG_NAME,
AltTag.TAG_NAME,
ClientTag.TAG_NAME,
IMetaTag.TAG_NAME,
-> false
else -> true
@@ -30,27 +30,39 @@ package com.vitorpamplona.quartz.nip50Search
* tokens that label data rather than describe it: `music`, `eng`, `openlibrary`, `IS_A_PROPERTY_OF`,
* `concept-header`, `OL19722168W`, `2026-06-01`.
*
* Allocation-free: the read path calls it once per value per event per keystroke.
* Allocation-free, and a single scan for nearly every value: the read path calls it once per
* value per event per keystroke.
*/
fun isNaturalLanguageValue(value: String): Boolean {
var start = 0
var end = value.length
while (start < end && value[start].isWhitespace()) start++
while (end > start && value[end - 1].isWhitespace()) end--
if (start == end || isMachineValue(value, start, end)) return false
if (start == end || isStructuredValue(value, start, end)) return false
var nonAscii = false
for (i in start until end) {
val c = value[i]
if (c.isWhitespace() || c.code > 127) return true
// The token shapes in isMachineToken cannot contain whitespace.
if (c.isWhitespace()) return true
if (c.code > 127) nonAscii = true
}
// A single token. A capitalized letters-only word can only collide with a hex id, so the
// other machine checks are needed just for the rare non-ASCII token, such as an IRI.
return if (nonAscii) {
!isMachineToken(value, start, end)
} else {
isCapitalizedWord(value, start, end) && !isHexId(value, start, end)
}
return isCapitalizedWord(value, start, end)
}
/**
* Whether a tag value is an identifier or a structure rather than text: a JSON object or array,
* a number (including a signed one and an ISBN-10 ending in `X`), a URI of any scheme
* (`https://…`, `wss://…`, `tag:…`, `urn:…`, `isbn:…`), an event address (`kind:pubkey:d`), a
* long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine
* a URL (`https://…`, `wss://…`, even with unescaped spaces in its path), an event address
* (`kind:pubkey:d`, even with spaces in `d`), or, as a single token, a number (including a
* signed one and an ISBN-10 ending in `X`), a URI of any scheme (`tag:…`, `urn:…`, `isbn:…`),
* a long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine
* values too: there is nothing in them to index.
*/
fun isMachineValue(value: String): Boolean {
@@ -58,10 +70,13 @@ fun isMachineValue(value: String): Boolean {
var end = value.length
while (start < end && value[start].isWhitespace()) start++
while (end > start && value[end - 1].isWhitespace()) end--
return start == end || isMachineValue(value, start, end)
if (start == end || isStructuredValue(value, start, end)) return true
for (i in start until end) if (value[i].isWhitespace()) return false
return isMachineToken(value, start, end)
}
private fun isMachineValue(
/** The machine shapes recognizable from their first characters, whitespace or not. */
private fun isStructuredValue(
v: String,
start: Int,
end: Int,
@@ -69,17 +84,20 @@ private fun isMachineValue(
val first = v[start]
val last = v[end - 1]
if ((first == '{' && last == '}') || (first == '[' && last == ']')) return true
return isUrl(v, start, end) || isAddress(v, start, end)
}
// Every other shape is a single token.
for (i in start until end) if (v[i].isWhitespace()) return false
return isNumber(v, start, end) ||
/** The machine shapes that are a single token. */
private fun isMachineToken(
v: String,
start: Int,
end: Int,
): Boolean =
isNumber(v, start, end) ||
isUri(v, start, end) ||
isAddress(v, start, end) ||
isHexId(v, start, end) ||
isUuid(v, start, end) ||
isBech32(v, start)
}
private fun Char.isAsciiDigit() = this in '0'..'9'
@@ -108,21 +126,41 @@ private fun isNumber(
return i == end
}
/** An RFC 3986 scheme followed by `:` and something more, as a single token. */
/** The index of the `:` that ends a leading RFC 3986 scheme, or -1 when there is none. */
private fun schemeEnd(
v: String,
start: Int,
end: Int,
): Int {
if (!v[start].isAsciiLetter()) return -1
var i = start + 1
while (i < end) {
val c = v[i]
if (c == ':') return i
if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return -1
i++
}
return -1
}
/** A scheme followed by `:` and something more, as a single token. */
private fun isUri(
v: String,
start: Int,
end: Int,
): Boolean {
if (!v[start].isAsciiLetter()) return false
var i = start + 1
while (i < end) {
val c = v[i]
if (c == ':') return i + 1 < end
if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return false
i++
}
return false
val colon = schemeEnd(v, start, end)
return colon >= 0 && colon + 1 < end
}
/** A scheme followed by `://`. Unlike "Song: Gold", never text. */
private fun isUrl(
v: String,
start: Int,
end: Int,
): Boolean {
val colon = schemeEnd(v, start, end)
return colon >= 0 && colon + 2 < end && v[colon + 1] == '/' && v[colon + 2] == '/'
}
/** `<kind>:<64-hex pubkey>` optionally followed by `:<d>`. */
@@ -340,6 +340,8 @@ class DecentralizedListsTest {
arrayOf("e", headerId, "wss://relay.example.com", "mention"),
arrayOf("alt", "Book: Herfsttij der Middeleeuwen by Johan Huizinga"),
arrayOf("client", "Brainstorm"),
arrayOf("imeta", "url https://example.com/cover.jpg", "m image/jpeg", "alt A book cover"),
arrayOf("artwork", "https://example.com/Cover Art.jpg"),
),
),
)
@@ -76,6 +76,11 @@ class NaturalLanguageValueTest {
"{\"word\": {\"slug\": \"ol-ol98624w\", \"name\": \"Waking with Enemies\"}}",
"[\"title\", \"artist\"]",
"Re:Zero",
// URLs and addresses with unescaped spaces, as feeds publish them
"https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg",
"https://headstarts.uk/msp/longy/Katherines Wheel/katherines wheel.xml",
"39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid",
"ABCDEFABCDEFABCDEFABCDEFABCDEFAB",
).forEach { assertFalse(isNaturalLanguageValue(it), it) }
}
@@ -97,13 +102,15 @@ class NaturalLanguageValueTest {
"c65d75f4d058f4746e12e44441cacea0",
"8ad7c296-67e9-5f57-ba52-2ff732e62e87",
"npub1f5pre6wl6ad87vr4hr5wppqq30sh58m4p33mthnjreh03qadcajs7gwt3z",
"https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg",
"39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid",
"",
).forEach { assertTrue(isMachineValue(it), it) }
}
@Test
fun wordsAndPhrasesAreNotMachineValues() {
listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell").forEach {
listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell", "{ राधेय }Rishi Verma").forEach {
assertFalse(isMachineValue(it), it)
}
}