mirror of
https://github.com/vitorpamplona/amethyst.git
synced 2026-10-05 19:28:25 +00:00
fix(quartz): drop spaced URLs, spaced addresses and imeta from list search text
Audit of the natural-language tag indexing against 1.5M list events: - A value containing whitespace skipped every machine check, so feed URLs with unescaped spaces (url, artwork, feedUrl, t: ~1,400 values) and an address with a spaced `d` were indexed. A `scheme://` prefix and an address prefix now mark a value as machine data with or without spaces. - NIP-92 `imeta` values are space-separated `key value` pairs, so every URL, mime type and hash in them would pass; the tag is skipped. - The classifier is now a single scan for nearly every value: an ASCII single token is decided by the capitalized-word test alone. 316 -> 150 ns per value on the corpus. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_016XJSf4ekFVuhSj5m8eeNsP
This commit is contained in:
+5
-2
@@ -40,6 +40,7 @@ import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor
|
||||
import com.vitorpamplona.quartz.nip50Search.isMachineValue
|
||||
import com.vitorpamplona.quartz.nip50Search.isNaturalLanguageValue
|
||||
import com.vitorpamplona.quartz.nip89AppHandlers.clientTag.ClientTag
|
||||
import com.vitorpamplona.quartz.nip92IMeta.IMetaTag
|
||||
|
||||
/**
|
||||
* The human-authored text of any kind in the family, in a fixed order: the header's `names`
|
||||
@@ -100,9 +101,10 @@ fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean
|
||||
* `relationshipType` and tags nobody has named yet — so this walk decides by what a value looks
|
||||
* like rather than by the tag it sits in.
|
||||
*
|
||||
* Skips the tags [forEachSearchableListField] already visits, plus two NIP-defined metadata tags
|
||||
* Skips the tags [forEachSearchableListField] already visits, plus the NIP-defined metadata tags
|
||||
* that are not the event's own text: `alt` (NIP-31 fallback text, which here restates `title`
|
||||
* and `artist` behind a fixed "Song: … by …" prefix) and `client` (NIP-89 app name).
|
||||
* and `artist` behind a fixed "Song: … by …" prefix), `client` (NIP-89 app name) and `imeta`
|
||||
* (NIP-92 `key value` pairs, whose space would otherwise pass every URL and hash as text).
|
||||
*
|
||||
* @return false when the visitor stopped the walk.
|
||||
*/
|
||||
@@ -128,6 +130,7 @@ private fun isExtraFieldTagName(name: String) =
|
||||
HashtagTag.TAG_NAME,
|
||||
AltTag.TAG_NAME,
|
||||
ClientTag.TAG_NAME,
|
||||
IMetaTag.TAG_NAME,
|
||||
-> false
|
||||
|
||||
else -> true
|
||||
|
||||
+63
-25
@@ -30,27 +30,39 @@ package com.vitorpamplona.quartz.nip50Search
|
||||
* tokens that label data rather than describe it: `music`, `eng`, `openlibrary`, `IS_A_PROPERTY_OF`,
|
||||
* `concept-header`, `OL19722168W`, `2026-06-01`.
|
||||
*
|
||||
* Allocation-free: the read path calls it once per value per event per keystroke.
|
||||
* Allocation-free, and a single scan for nearly every value: the read path calls it once per
|
||||
* value per event per keystroke.
|
||||
*/
|
||||
fun isNaturalLanguageValue(value: String): Boolean {
|
||||
var start = 0
|
||||
var end = value.length
|
||||
while (start < end && value[start].isWhitespace()) start++
|
||||
while (end > start && value[end - 1].isWhitespace()) end--
|
||||
if (start == end || isMachineValue(value, start, end)) return false
|
||||
if (start == end || isStructuredValue(value, start, end)) return false
|
||||
|
||||
var nonAscii = false
|
||||
for (i in start until end) {
|
||||
val c = value[i]
|
||||
if (c.isWhitespace() || c.code > 127) return true
|
||||
// The token shapes in isMachineToken cannot contain whitespace.
|
||||
if (c.isWhitespace()) return true
|
||||
if (c.code > 127) nonAscii = true
|
||||
}
|
||||
|
||||
// A single token. A capitalized letters-only word can only collide with a hex id, so the
|
||||
// other machine checks are needed just for the rare non-ASCII token, such as an IRI.
|
||||
return if (nonAscii) {
|
||||
!isMachineToken(value, start, end)
|
||||
} else {
|
||||
isCapitalizedWord(value, start, end) && !isHexId(value, start, end)
|
||||
}
|
||||
return isCapitalizedWord(value, start, end)
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether a tag value is an identifier or a structure rather than text: a JSON object or array,
|
||||
* a number (including a signed one and an ISBN-10 ending in `X`), a URI of any scheme
|
||||
* (`https://…`, `wss://…`, `tag:…`, `urn:…`, `isbn:…`), an event address (`kind:pubkey:d`), a
|
||||
* long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine
|
||||
* a URL (`https://…`, `wss://…`, even with unescaped spaces in its path), an event address
|
||||
* (`kind:pubkey:d`, even with spaces in `d`), or, as a single token, a number (including a
|
||||
* signed one and an ISBN-10 ending in `X`), a URI of any scheme (`tag:…`, `urn:…`, `isbn:…`),
|
||||
* a long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine
|
||||
* values too: there is nothing in them to index.
|
||||
*/
|
||||
fun isMachineValue(value: String): Boolean {
|
||||
@@ -58,10 +70,13 @@ fun isMachineValue(value: String): Boolean {
|
||||
var end = value.length
|
||||
while (start < end && value[start].isWhitespace()) start++
|
||||
while (end > start && value[end - 1].isWhitespace()) end--
|
||||
return start == end || isMachineValue(value, start, end)
|
||||
if (start == end || isStructuredValue(value, start, end)) return true
|
||||
for (i in start until end) if (value[i].isWhitespace()) return false
|
||||
return isMachineToken(value, start, end)
|
||||
}
|
||||
|
||||
private fun isMachineValue(
|
||||
/** The machine shapes recognizable from their first characters, whitespace or not. */
|
||||
private fun isStructuredValue(
|
||||
v: String,
|
||||
start: Int,
|
||||
end: Int,
|
||||
@@ -69,17 +84,20 @@ private fun isMachineValue(
|
||||
val first = v[start]
|
||||
val last = v[end - 1]
|
||||
if ((first == '{' && last == '}') || (first == '[' && last == ']')) return true
|
||||
return isUrl(v, start, end) || isAddress(v, start, end)
|
||||
}
|
||||
|
||||
// Every other shape is a single token.
|
||||
for (i in start until end) if (v[i].isWhitespace()) return false
|
||||
|
||||
return isNumber(v, start, end) ||
|
||||
/** The machine shapes that are a single token. */
|
||||
private fun isMachineToken(
|
||||
v: String,
|
||||
start: Int,
|
||||
end: Int,
|
||||
): Boolean =
|
||||
isNumber(v, start, end) ||
|
||||
isUri(v, start, end) ||
|
||||
isAddress(v, start, end) ||
|
||||
isHexId(v, start, end) ||
|
||||
isUuid(v, start, end) ||
|
||||
isBech32(v, start)
|
||||
}
|
||||
|
||||
private fun Char.isAsciiDigit() = this in '0'..'9'
|
||||
|
||||
@@ -108,21 +126,41 @@ private fun isNumber(
|
||||
return i == end
|
||||
}
|
||||
|
||||
/** An RFC 3986 scheme followed by `:` and something more, as a single token. */
|
||||
/** The index of the `:` that ends a leading RFC 3986 scheme, or -1 when there is none. */
|
||||
private fun schemeEnd(
|
||||
v: String,
|
||||
start: Int,
|
||||
end: Int,
|
||||
): Int {
|
||||
if (!v[start].isAsciiLetter()) return -1
|
||||
var i = start + 1
|
||||
while (i < end) {
|
||||
val c = v[i]
|
||||
if (c == ':') return i
|
||||
if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return -1
|
||||
i++
|
||||
}
|
||||
return -1
|
||||
}
|
||||
|
||||
/** A scheme followed by `:` and something more, as a single token. */
|
||||
private fun isUri(
|
||||
v: String,
|
||||
start: Int,
|
||||
end: Int,
|
||||
): Boolean {
|
||||
if (!v[start].isAsciiLetter()) return false
|
||||
var i = start + 1
|
||||
while (i < end) {
|
||||
val c = v[i]
|
||||
if (c == ':') return i + 1 < end
|
||||
if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return false
|
||||
i++
|
||||
}
|
||||
return false
|
||||
val colon = schemeEnd(v, start, end)
|
||||
return colon >= 0 && colon + 1 < end
|
||||
}
|
||||
|
||||
/** A scheme followed by `://`. Unlike "Song: Gold", never text. */
|
||||
private fun isUrl(
|
||||
v: String,
|
||||
start: Int,
|
||||
end: Int,
|
||||
): Boolean {
|
||||
val colon = schemeEnd(v, start, end)
|
||||
return colon >= 0 && colon + 2 < end && v[colon + 1] == '/' && v[colon + 2] == '/'
|
||||
}
|
||||
|
||||
/** `<kind>:<64-hex pubkey>` optionally followed by `:<d>`. */
|
||||
|
||||
+2
@@ -340,6 +340,8 @@ class DecentralizedListsTest {
|
||||
arrayOf("e", headerId, "wss://relay.example.com", "mention"),
|
||||
arrayOf("alt", "Book: Herfsttij der Middeleeuwen by Johan Huizinga"),
|
||||
arrayOf("client", "Brainstorm"),
|
||||
arrayOf("imeta", "url https://example.com/cover.jpg", "m image/jpeg", "alt A book cover"),
|
||||
arrayOf("artwork", "https://example.com/Cover Art.jpg"),
|
||||
),
|
||||
),
|
||||
)
|
||||
|
||||
+8
-1
@@ -76,6 +76,11 @@ class NaturalLanguageValueTest {
|
||||
"{\"word\": {\"slug\": \"ol-ol98624w\", \"name\": \"Waking with Enemies\"}}",
|
||||
"[\"title\", \"artist\"]",
|
||||
"Re:Zero",
|
||||
// URLs and addresses with unescaped spaces, as feeds publish them
|
||||
"https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg",
|
||||
"https://headstarts.uk/msp/longy/Katherines Wheel/katherines wheel.xml",
|
||||
"39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid",
|
||||
"ABCDEFABCDEFABCDEFABCDEFABCDEFAB",
|
||||
).forEach { assertFalse(isNaturalLanguageValue(it), it) }
|
||||
}
|
||||
|
||||
@@ -97,13 +102,15 @@ class NaturalLanguageValueTest {
|
||||
"c65d75f4d058f4746e12e44441cacea0",
|
||||
"8ad7c296-67e9-5f57-ba52-2ff732e62e87",
|
||||
"npub1f5pre6wl6ad87vr4hr5wppqq30sh58m4p33mthnjreh03qadcajs7gwt3z",
|
||||
"https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg",
|
||||
"39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid",
|
||||
"",
|
||||
).forEach { assertTrue(isMachineValue(it), it) }
|
||||
}
|
||||
|
||||
@Test
|
||||
fun wordsAndPhrasesAreNotMachineValues() {
|
||||
listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell").forEach {
|
||||
listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell", "{ राधेय }Rishi Verma").forEach {
|
||||
assertFalse(isMachineValue(it), it)
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user