Merge pull request #4318 from vitorpamplona/claude/pensive-pascal-wyl0to

Index natural-language tag values in Decentralized Lists
This commit is contained in:
Vitor Pamplona
2026-10-05 10:00:10 -04:00
committed by GitHub
12 changed files with 603 additions and 32 deletions
@@ -2,7 +2,7 @@
Every concrete `SearchableEvent` implementor in Quartz, with the exact `indexableContent()`
expression. **Update this file in the same PR as any change to the searchable set or to an
`indexableContent()` body** (see SKILL.md). Verified against the code 2026-09-17.
`indexableContent()` body** (see SKILL.md). Verified against the code 2026-10-04.
Counts: 139 concrete classes covering 140 kind values (`GitStatusEvent` spans 4 kinds;
kind 30063 is shared by two NIPs and kind 38000 by three classes — see the footnotes). File paths are under
@@ -65,7 +65,7 @@ Separator legend: **NL** = `joinToString("\n")`, **SP** = `joinToString(" ")`.
| 9736 | Bolt12ZapEvent | nipB1Bolt12Zaps/zap | `content` |
| 9737 | Bolt12ZapIntentEvent | nipB1Bolt12Zaps/intent | `content` |
| 9802 | HighlightEvent | nip84Highlights | `listOfNotNull(comment(), context(), content)` NL |
| 9998 | ListHeaderEvent | experimental/decentralizedLists/header | `tags.searchableListContent()` NL — `names` (singular, plural), `titles` (singular, plural), `name`, `title`, `description`, `comments`, then every `t` value; ids/pubkeys/coordinates are left to tag filters |
| 9998 | ListHeaderEvent | experimental/decentralizedLists/header | `tags.searchableListContent()` NL — `names` (singular, plural), `titles` (singular, plural), `name`, `title`, `description`, `comments`, then in tag order every `t` value that is not `isMachineValue` and every value of every other tag (except `alt`, `client`, `imeta`) that passes `isNaturalLanguageValue` (has whitespace or non-ASCII, or is one capitalized letters-only word; never JSON, numbers, URIs of any scheme, addresses, hex ids, UUIDs, bech32) |
| 9999 | ListItemEvent | experimental/decentralizedLists/item | same as 9998 |
| 10003 | BookmarkListEvent | nip51Lists/bookmarkList | `listOfNotNull(title())` NL |
| 10100 | AgentProfileEvent | buzz/agentProfiles | `profileOrNull()?.let { listOfNotNull(it.name, it.displayName).joinToString("\n") } ?: ""` |
@@ -85,7 +85,19 @@ Filter(
All four kinds are `SearchableEvent`s sharing one walk
(`forEachSearchableListField`): `names` and `titles` (singular then plural),
`name`, `title`, `description`, `comments`, then every `t` item value.
`name`, `title`, `description`, `comments`, then, in tag order, every `t`
item value and every value of every other tag that reads as natural language.
The tag set is open — deployments add `author`, `subject`, `artist`,
`relationshipType` and tags nobody has named yet — so that last step decides
by the value, not the tag (`nip50Search/NaturalLanguageValue.kt`): a value is
indexed when it has whitespace or a non-ASCII character, or is a single
capitalized word ("Fiction", "Aristotle"), and is never a machine value (JSON,
a number, a URI of any scheme, an address, a hex id, a UUID, bech32). `t`
values skip the natural-language test but not the machine one. `alt` (NIP-31
fallback text that restates other tags), `client` and `imeta` (NIP-92 `key
value` pairs) are not indexed.
`content` is not part of the spec, and ids, pubkeys and coordinates are
served by `#p`/`#e`/`#a`/`#z` filters, so none of them are indexed.
@@ -35,13 +35,20 @@ import com.vitorpamplona.quartz.experimental.decentralizedLists.tags.Description
import com.vitorpamplona.quartz.nip01Core.core.TagArray
import com.vitorpamplona.quartz.nip01Core.core.fastForEach
import com.vitorpamplona.quartz.nip01Core.tags.hashtags.HashtagTag
import com.vitorpamplona.quartz.nip31Alts.AltTag
import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor
import com.vitorpamplona.quartz.nip50Search.isMachineValue
import com.vitorpamplona.quartz.nip50Search.isNaturalLanguageValue
import com.vitorpamplona.quartz.nip89AppHandlers.clientTag.ClientTag
import com.vitorpamplona.quartz.nip92IMeta.IMetaTag
/**
* The human-authored text of any kind in the family, in a fixed order: the header's `names`
* and `titles` (singular, then plural), the item's `name` and `title`, `description`,
* `comments`, then each `t` item value. Headers and items share one walk because the spec's
* nonstandard method lets an item carry header tags; each kind simply has fewer of them set.
* The human-authored text of any kind in the family: first, in a fixed order, the header's `names`
* and `titles` (singular, then plural), the item's `name` and `title`, `description` and
* `comments`; then, in tag order, each `t` value that is not a machine value and every value of
* every other tag that reads as natural language ([forEachOtherField]). Headers and items share
* one walk because the spec's nonstandard method lets an item carry header tags; each kind simply
* has fewer of them set.
*
* `content` is not part of the spec and ids/pubkeys/coordinates are served by tag filters,
* so neither is indexed.
@@ -51,7 +58,8 @@ import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor
fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean {
// The read path runs once per event per keystroke, so this is two allocation-free passes
// instead of one full scan (and one parsed object) per field. The first pass only remembers
// the first well-formed tag of each field; the fixed visiting order is applied afterwards.
// the first well-formed tag of each named field, so their fixed order can be applied before
// the second pass visits everything else.
var names: Array<String>? = null
var titles: Array<String>? = null
var name: String? = null
@@ -83,8 +91,51 @@ fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean
title?.let { if (!visitor.visit(it)) return false }
description?.let { if (!visitor.visit(it)) return false }
comments?.let { if (!visitor.visit(it)) return false }
return forEachOtherField(visitor, withHashtags = true)
}
/**
* Every tag the first pass of [forEachSearchableListField] does not own, in tag order: each `t`
* value that is not [isMachineValue] (only [withHashtags]), and every value of every other tag
* that [isNaturalLanguageValue]. The family's tag set is open — deployments add `author`,
* `subject`, `artist`, `relationshipType` and tags nobody has named yet — so this decides by what
* a value looks like rather than by the tag it sits in.
*
* Skips the NIP-defined metadata tags that are not the event's own text: `alt` (NIP-31 fallback
* text, which here restates `title` and `artist` behind a fixed "Song: … by …" prefix), `client`
* (NIP-89 app name) and `imeta` (NIP-92 `key value` pairs, whose space would otherwise pass every
* URL and hash as text).
*
* @return false when the visitor stopped the walk.
*/
private fun TagArray.forEachOtherField(
visitor: IndexableFieldVisitor,
withHashtags: Boolean,
): Boolean {
fastForEach { tag ->
HashtagTag.parse(tag)?.let { if (!visitor.visit(it)) return false }
if (tag.size < 2) return@fastForEach
when (tag[0]) {
HashtagTag.TAG_NAME -> {
if (withHashtags && !isMachineValue(tag[1]) && !visitor.visit(tag[1])) return false
}
NamesTag.TAG_NAME,
TitlesTag.TAG_NAME,
NameTag.TAG_NAME,
TitleTag.TAG_NAME,
DescriptionTag.TAG_NAME,
CommentsTag.TAG_NAME,
AltTag.TAG_NAME,
ClientTag.TAG_NAME,
IMetaTag.TAG_NAME,
-> {}
else -> {
for (i in 1 until tag.size) {
if (isNaturalLanguageValue(tag[i]) && !visitor.visit(tag[i])) return false
}
}
}
}
return true
}
@@ -103,10 +154,19 @@ fun TagArray.searchableListTitles(): List<String?> {
/** The DESCRIPTION role of [forEachSearchableListField]: `description`, then `comments`. */
fun TagArray.searchableListDescriptions(): List<String?> = listOf(description(), comments())
/**
* The TEXT role of [forEachSearchableListField]: the natural-language values of the tags the
* fixed-order fields do not own, one per line, or null when the event has none. The `t` values
* are not here; the search funnel carries them as hashtags.
*/
fun TagArray.searchableListExtraText(): String? = joinFields { forEachOtherField(it, withHashtags = false) }.ifEmpty { null }
/** The write-path join of [forEachSearchableListField]: one field per line. */
fun TagArray.searchableListContent() =
fun TagArray.searchableListContent() = joinFields { forEachSearchableListField(it) }
private inline fun joinFields(walk: (IndexableFieldVisitor) -> Boolean) =
buildString {
forEachSearchableListField { field ->
walk { field ->
if (field != null) {
if (isNotEmpty()) append('\n')
append(field)
@@ -31,8 +31,14 @@ import com.vitorpamplona.quartz.nip01Core.core.fastAny
*
* ## What it matches
*
* Terms are ANDed and each is a case-insensitive substring — of a tag value, or of the event's
* own indexable fields. Substring rather than token because that is what Amethyst's local search
* Terms are ANDed and each is a case-insensitive substring of the event's indexable fields — the
* same text a store's full-text index holds ([SearchableEvent.forEachIndexableField] is the read
* side of [SearchableEvent.indexableContent]), so local search and a store or relay agree on what
* an event can be found by. A tag value is searched only when the kind indexes it: a hashtag, an
* id, a URL or a data label the kind leaves out of its index is left out here too. A kind that is
* not a [SearchableEvent] has no indexable fields, so it falls back to its tag values and content.
*
* Substring rather than token because that is what Amethyst's local search
* has always done (`content.contains(text, true)`), and switching to tokens would silently stop
* matching mid-word queries; AND rather than one literal phrase because that is what a relay does
* with the same string, and the point of routing local search through a `Filter` is that the two
@@ -57,8 +63,9 @@ import com.vitorpamplona.quartz.nip01Core.core.fastAny
class EventSearchMatcher(
search: String?,
/**
* Tag names whose values are not worth searching — `p`/`e`/`a` hold hex ids that only ever
* match by accident, `client` and `alt` hold text the author did not write.
* For a kind that is not a [SearchableEvent]: tag names whose values are not worth searching —
* `p`/`e`/`a` hold hex ids that only ever match by accident, `client` and `alt` hold text the
* author did not write.
*/
private val exceptTagNames: Set<String> = DEFAULT_EXCLUDED_TAGS,
) {
@@ -82,11 +89,9 @@ class EventSearchMatcher(
event: Event,
term: String,
): Boolean {
if (event.tags.fastAny { it.size > 1 && it[0] !in exceptTagNames && it[1].contains(term, true) }) {
return true
}
if (event !is SearchableEvent) return event.content.contains(term, true)
return visitor.matches(event, term)
if (event is SearchableEvent) return visitor.matches(event, term)
return event.content.contains(term, true) ||
event.tags.fastAny { it.size > 1 && it[0] !in exceptTagNames && it[1].contains(term, true) }
}
/**
@@ -0,0 +1,238 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip50Search
/**
* Whether a free-form tag value reads as natural language, for kinds whose tag set is open
* (anyone can invent a tag name) and so cannot be indexed from a list of known tag names.
*
* A value qualifies when it is not [isMachineValue] and it either contains whitespace, contains a
* non-ASCII character, or is a single capitalized ASCII word made of letters only. That keeps
* "Social history", "刘慈欣", "Fiction" and "Aristotle", and drops the single lowercase or mixed
* tokens that label data rather than describe it: `music`, `eng`, `openlibrary`, `IS_A_PROPERTY_OF`,
* `concept-header`, `OL19722168W`, `2026-06-01`.
*
* Allocation-free, and a single scan for nearly every value: the read path calls it once per
* value per event per keystroke.
*/
fun isNaturalLanguageValue(value: String): Boolean {
var start = 0
var end = value.length
while (start < end && value[start].isSpace()) start++
while (end > start && value[end - 1].isSpace()) end--
if (start == end || isStructuredValue(value, start, end)) return false
var nonAscii = false
for (i in start until end) {
val c = value[i]
// The token shapes in isMachineToken cannot contain whitespace.
if (c.isSpace()) return true
if (c.code > 127) nonAscii = true
}
// A single token. A capitalized letters-only word can only collide with a hex id, so the
// other machine checks are needed just for the rare non-ASCII token, such as an IRI.
return if (nonAscii) {
!isMachineToken(value, start, end)
} else {
isCapitalizedWord(value, start, end) && !isHexId(value, start, end)
}
}
/**
* Whether a tag value is an identifier or a structure rather than text: a JSON object or array,
* a URL (`https://…`, `wss://…`, even with unescaped spaces in its path), an event address
* (`kind:pubkey:d`, even with spaces in `d`), or, as a single token, a number (including a
* signed one and an ISBN-10 ending in `X`), a URI of any scheme (`tag:…`, `urn:…`, `isbn:…`),
* a long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine
* values too: there is nothing in them to index.
*/
fun isMachineValue(value: String): Boolean {
var start = 0
var end = value.length
while (start < end && value[start].isSpace()) start++
while (end > start && value[end - 1].isSpace()) end--
if (start == end || isStructuredValue(value, start, end)) return true
for (i in start until end) if (value[i].isSpace()) return false
return isMachineToken(value, start, end)
}
/** The machine shapes recognizable from their first characters, whitespace or not. */
private fun isStructuredValue(
v: String,
start: Int,
end: Int,
): Boolean {
val first = v[start]
val last = v[end - 1]
if ((first == '{' && last == '}') || (first == '[' && last == ']')) return true
return isUrl(v, start, end) || isAddress(v, start, end)
}
/** The machine shapes that are a single token. */
private fun isMachineToken(
v: String,
start: Int,
end: Int,
): Boolean =
isNumber(v, start, end) ||
isUri(v, start, end) ||
isHexId(v, start, end) ||
isUuid(v, start, end) ||
isBech32(v, start)
/**
* [isWhitespace] with an ASCII fast path: on the JVM, [isWhitespace] is two Unicode table lookups
* per character, and nearly every character this file scans is ASCII.
*/
private fun Char.isSpace() = if (code < 0x80) this == ' ' || this in '\t'..'\r' || this in '\u001C'..'\u001F' else isWhitespace()
private fun Char.isAsciiDigit() = this in '0'..'9'
private fun Char.isAsciiLetter() = this in 'a'..'z' || this in 'A'..'Z'
private fun Char.isHex() = this in '0'..'9' || this in 'a'..'f' || this in 'A'..'F'
/** `-1`, `473`, `0.5`, an ISBN-13, and an ISBN-10 with its `X` check digit. */
private fun isNumber(
v: String,
start: Int,
end: Int,
): Boolean {
var i = start
if (v[i] == '-' || v[i] == '+') i++
val digitsStart = i
while (i < end && v[i].isAsciiDigit()) i++
if (i == digitsStart) return false
if (i < end && v[i] == '.') {
i++
val fractionStart = i
while (i < end && v[i].isAsciiDigit()) i++
if (i == fractionStart) return false
}
if (i < end && (v[i] == 'X' || v[i] == 'x')) i++
return i == end
}
/** The index of the `:` that ends a leading RFC 3986 scheme, or -1 when there is none. */
private fun schemeEnd(
v: String,
start: Int,
end: Int,
): Int {
if (!v[start].isAsciiLetter()) return -1
var i = start + 1
while (i < end) {
val c = v[i]
if (c == ':') return i
if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return -1
i++
}
return -1
}
/** A scheme followed by `:` and something more, as a single token. */
private fun isUri(
v: String,
start: Int,
end: Int,
): Boolean {
val colon = schemeEnd(v, start, end)
return colon >= 0 && colon + 1 < end
}
/** A scheme followed by `://`. Unlike "Song: Gold", never text. */
private fun isUrl(
v: String,
start: Int,
end: Int,
): Boolean {
val colon = schemeEnd(v, start, end)
return colon >= 0 && colon + 2 < end && v[colon + 1] == '/' && v[colon + 2] == '/'
}
/** `<kind>:<64-hex pubkey>` optionally followed by `:<d>`. */
private fun isAddress(
v: String,
start: Int,
end: Int,
): Boolean {
var i = start
while (i < end && v[i].isAsciiDigit()) i++
if (i == start || i >= end || v[i] != ':') return false
i++
val pubkeyEnd = i + 64
if (pubkeyEnd > end) return false
while (i < pubkeyEnd) {
if (!v[i].isHex()) return false
i++
}
return i == end || v[i] == ':'
}
/** Event ids, pubkeys, and the shorter hashes (32+ hex characters) feeds use as ids. */
private fun isHexId(
v: String,
start: Int,
end: Int,
): Boolean {
if (end - start < 32) return false
for (i in start until end) if (!v[i].isHex()) return false
return true
}
private fun isUuid(
v: String,
start: Int,
end: Int,
): Boolean {
if (end - start != 36) return false
for (i in 0 until 36) {
val c = v[start + i]
val hyphen = i == 8 || i == 13 || i == 18 || i == 23
if (if (hyphen) c != '-' else !c.isHex()) return false
}
return true
}
private val BECH32_PREFIXES = arrayOf("npub1", "nsec1", "note1", "nevent1", "naddr1", "nprofile1", "nrelay1")
private fun isBech32(
v: String,
start: Int,
): Boolean {
for (prefix in BECH32_PREFIXES) if (v.startsWith(prefix, start)) return true
return false
}
/** `Fiction`, `Aristotle`, `RTL`, `AT&T`: an ASCII capital followed by letters only. */
private fun isCapitalizedWord(
v: String,
start: Int,
end: Int,
): Boolean {
if (v[start] !in 'A'..'Z') return false
for (i in start + 1 until end) {
val c = v[i]
if (!(c.isAsciiLetter() || c == '\'' || c == '&' || c == '.')) return false
}
return true
}
@@ -31,6 +31,7 @@ import com.vitorpamplona.quartz.experimental.birdstar.BirdDetectionEvent
import com.vitorpamplona.quartz.experimental.birdstar.BirdexEvent
import com.vitorpamplona.quartz.experimental.decentralizedLists.DecentralizedListEvent
import com.vitorpamplona.quartz.experimental.decentralizedLists.searchableListDescriptions
import com.vitorpamplona.quartz.experimental.decentralizedLists.searchableListExtraText
import com.vitorpamplona.quartz.experimental.decentralizedLists.searchableListTitles
import com.vitorpamplona.quartz.experimental.edits.TextNoteModificationEvent
import com.vitorpamplona.quartz.experimental.fitness.workout.ExerciseTemplateEvent
@@ -629,9 +630,10 @@ object SearchFieldExtractor {
// already, so -- as for 1111/1311/30382 -- they are not passed again, and
// content is not part of the spec. One branch for all four: an item may
// carry header tags (the spec's nonstandard method), and the tag helpers
// read whichever are present.
// read whichever are present. Every other natural-language tag value
// (author, subject, artist, ...) is the text tier, below both.
is DecentralizedListEvent -> {
tiers(event, event.tags.searchableListTitles(), event.tags.searchableListDescriptions(), null)
tiers(event, event.tags.searchableListTitles(), event.tags.searchableListDescriptions(), event.tags.searchableListExtraText())
}
// kind 1 LAST among the explicit branches, defensively: a future
@@ -299,7 +299,7 @@ class DecentralizedListsTest {
)
assertEquals("Fido\nVery good\nFido", item.indexableContent())
listOf<SearchableEvent>(header, item).forEach { event ->
listOf<SearchableEvent>(header, item, book).forEach { event ->
val fields = mutableListOf<String>()
event.forEachIndexableField { field ->
field?.let { fields.add(it) }
@@ -310,4 +310,56 @@ class DecentralizedListsTest {
assertEquals("", assertIs<ListItemEvent>(event(9999, emptyArray())).indexableContent())
}
// A book item as a Decentralized Lists deployment publishes it: deployment-specific tags
// around the spec's own, most of them machine data.
private val book =
assertIs<AddressableListItemEvent>(
event(
39999,
arrayOf(
arrayOf("d", "ol-ol1141875w"),
arrayOf("z", "39998:$author:books"),
arrayOf("title", "Herfsttij der Middeleeuwen"),
arrayOf("t", "history"),
arrayOf("t", "tag:soundcloud,2010:tracks/711391888"),
arrayOf("t", "https://example.com/ep/1"),
arrayOf("author", "Johan Huizinga"),
arrayOf("subject", "Middle Ages"),
arrayOf("subject", "Fiction"),
arrayOf("subject", "Civilisation médiévale"),
arrayOf("open-library-id", "OL1141875W"),
arrayOf("isbn", "9780226360935"),
arrayOf("isbn10", "057507681X"),
arrayOf("year", "1919"),
arrayOf("lang", "eng"),
arrayOf("source", "openlibrary"),
arrayOf("cover", "https://covers.openlibrary.org/b/id/7245546-L.jpg"),
arrayOf("json", "{\"word\": {\"name\": \"Herfsttij der Middeleeuwen\"}}"),
arrayOf("p", author, "wss://relay.example.com", "Johan Fan"),
arrayOf("e", headerId, "wss://relay.example.com", "mention"),
arrayOf("alt", "Book: Herfsttij der Middeleeuwen by Johan Huizinga"),
arrayOf("client", "Brainstorm"),
arrayOf("imeta", "url https://example.com/cover.jpg", "m image/jpeg", "alt A book cover"),
arrayOf("artwork", "https://example.com/Cover Art.jpg"),
),
),
)
@Test
fun searchIndexesEveryNaturalLanguageTagValue() {
assertEquals(
"Herfsttij der Middeleeuwen\nhistory\nJohan Huizinga\nMiddle Ages\nFiction\nCivilisation médiévale\nJohan Fan",
book.indexableContent(),
)
}
@Test
fun searchSkipsMachineHashtagsButKeepsSingleWordOnes() {
val item =
assertIs<ListItemEvent>(
event(9999, arrayOf(arrayOf("t", "romance"), arrayOf("t", "8ad7c296-67e9-5f57-ba52-2ff732e62e87"), arrayOf("t", "473"))),
)
assertEquals("romance", item.indexableContent())
}
}
@@ -20,6 +20,7 @@
*/
package com.vitorpamplona.quartz.nip01Core.store.sqlite
import com.vitorpamplona.quartz.experimental.decentralizedLists.item.AddressableListItemEvent
import com.vitorpamplona.quartz.experimental.trustedLists.metric
import com.vitorpamplona.quartz.experimental.trustedLists.title
import com.vitorpamplona.quartz.experimental.trustedLists.users.UserTrustedListEvent
@@ -221,6 +222,35 @@ class SearchTest : BaseDBTest() {
db.assertQuery(null, Filter(search = memberKey))
}
@Test
fun testDecentralizedListItemsAreSearchableByEveryNaturalLanguageTag() =
forEachDB { db ->
val item =
signer.sign<AddressableListItemEvent>(
TimeUtils.now(),
AddressableListItemEvent.KIND,
arrayOf(
arrayOf("d", "ol-uniqbookslug"),
arrayOf("title", "Uniqtitle Middeleeuwen"),
arrayOf("author", "Johan Uniqauthor"),
arrayOf("subject", "Uniqsubject"),
arrayOf("lang", "uniqlang"),
arrayOf("alt", "Book: Uniqalt"),
),
"",
)
db.store.insertEvent(item)
db.assertQuery(item, Filter(search = "uniqtitle"))
db.assertQuery(item, Filter(search = "uniqauthor"))
db.assertQuery(item, Filter(search = "uniqsubject"))
// the slug, a lowercase data label and the NIP-31 fallback text are not indexed
db.assertQuery(null, Filter(search = "uniqbookslug"))
db.assertQuery(null, Filter(search = "uniqlang"))
db.assertQuery(null, Filter(search = "uniqalt"))
}
@Test
fun testProfileJsonFieldsAreSearchable() =
forEachDB { db ->
@@ -66,18 +66,49 @@ class EventSearchMatcherTest {
}
@Test
fun tagValuesAreSearchedExceptTheOnesNobodyMeansToSearch() {
fun aSearchableKindMatchesWhatItIndexesAndNothingElse() {
// Kind 1 indexes its subject and content, as a store's full-text index does; a hashtag
// the kind leaves out of its index is left out here too, so local and store search agree.
assertTrue(matches("subj", note("body", arrayOf(arrayOf("subject", "a subject")))))
assertTrue(matches("bitcoin", note("body", arrayOf(arrayOf("t", "bitcoin")))))
// `p`/`e`/`a` hold hex ids, `client` and `alt` hold text the author did not write.
assertFalse(matches("bitcoin", note("body", arrayOf(arrayOf("t", "bitcoin")))))
assertFalse(matches("amethyst", note("body", arrayOf(arrayOf("client", "amethyst")))))
assertFalse(matches("deadbeef", note("body", arrayOf(arrayOf("p", "deadbeef".repeat(8))))))
}
@Test
fun aDecentralizedListItemMatchesItsNaturalLanguageTagsButNotItsData() {
val book =
note(
"",
arrayOf(
arrayOf("d", "ol-ol1141875w"),
arrayOf("title", "Herfsttij der Middeleeuwen"),
arrayOf("author", "Johan Huizinga"),
arrayOf("subject", "Middle Ages"),
arrayOf("medium", "music"),
arrayOf("cover", "https://covers.openlibrary.org/b/id/7245546-L.jpg"),
),
kind = 39999,
)
assertTrue(matches("huizinga", book))
assertTrue(matches("middle ages", book))
assertFalse(matches("music", book))
assertFalse(matches("openlibrary", book))
assertFalse(matches("ol1141875w", book))
}
@Test
fun aKindWithNoIndexableFieldsFallsBackToItsTagsAndContent() {
assertTrue(matches("bitcoin", note("+", arrayOf(arrayOf("t", "bitcoin")), kind = 7)))
assertTrue(matches("+", note("+", kind = 7)))
// `p`/`e`/`a` hold hex ids, `client` and `alt` hold text the author did not write.
assertFalse(matches("amethyst", note("+", arrayOf(arrayOf("client", "amethyst")), kind = 7)))
assertFalse(matches("deadbeef", note("+", arrayOf(arrayOf("p", "deadbeef".repeat(8))), kind = 7)))
}
@Test
fun indexableFieldsBeyondContentAreSearched() {
// A long-form title lives in a tag, but its summary reaches the matcher through the
// visitor as well — both paths must find it.
// A long-form title and summary live in tags; they reach the matcher through the visitor.
val article = note("the body", arrayOf(arrayOf("title", "Lightning"), arrayOf("summary", "a summary")), kind = 30023)
assertTrue(matches("Lightning", article))
assertTrue(matches("summary", article))
@@ -0,0 +1,117 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.quartz.nip50Search
import kotlin.test.Test
import kotlin.test.assertFalse
import kotlin.test.assertTrue
/** Values taken from the kind 9999/39999 corpus on a Decentralized Lists relay. */
class NaturalLanguageValueTest {
@Test
fun phrasesAndNonAsciiTextAreNaturalLanguage() {
listOf(
"Social history",
"Johan Huizinga",
"Guevara, ernesto, 1928-1967",
"Climbing The Mountain (1979-1990)",
"[Live] Closer To Somewhere",
"Song: Gold by Torcon 7",
"刘慈欣",
"老子",
"Moyen Âge",
" padded phrase ",
).forEach { assertTrue(isNaturalLanguageValue(it), it) }
}
@Test
fun capitalizedWordsAreNaturalLanguage() {
listOf("Fiction", "History", "Aristotle", "Plutarch", "RTL", "PromoDJ", "AT&T").forEach {
assertTrue(isNaturalLanguageValue(it), it)
}
}
@Test
fun singleLowercaseOrMixedTokensAreNot() {
listOf(
"music",
"eng",
"openlibrary",
"reference",
"string",
"forward",
"IS_THE_CONCEPT_FOR",
"concept-header",
"tagassert--ol-ol49463w--science-fiction--6aca05b8",
"OL19722168W",
"2026-06-01",
"sorita67",
"Musician:",
).forEach { assertFalse(isNaturalLanguageValue(it), it) }
}
@Test
fun machineValuesAreNotNaturalLanguageEvenWithSpaces() {
listOf(
"",
" ",
"{\"word\": {\"slug\": \"ol-ol98624w\", \"name\": \"Waking with Enemies\"}}",
"[\"title\", \"artist\"]",
"Re:Zero",
// URLs and addresses with unescaped spaces, as feeds publish them
"https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg",
"https://headstarts.uk/msp/longy/Katherines Wheel/katherines wheel.xml",
"39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid",
"ABCDEFABCDEFABCDEFABCDEFABCDEFAB",
).forEach { assertFalse(isNaturalLanguageValue(it), it) }
}
@Test
fun machineValues() {
listOf(
"473",
"-1",
"0.5",
"9781984820983",
"057507681X",
"https://covers.openlibrary.org/b/id/7245546-L.jpg",
"wss://relay.damus.io",
"tag:soundcloud,2010:tracks/711391888",
"urn:isbn:9780679428329",
"39999:6aca05b812da97601151776d13de04ae71afc9d86da1408f0e72cffef72ece4b:ol-ol2838774w",
"39998:6aca05b812da97601151776d13de04ae71afc9d86da1408f0e72cffef72ece4b",
"6aca05b812da97601151776d13de04ae71afc9d86da1408f0e72cffef72ece4b",
"c65d75f4d058f4746e12e44441cacea0",
"8ad7c296-67e9-5f57-ba52-2ff732e62e87",
"npub1f5pre6wl6ad87vr4hr5wppqq30sh58m4p33mthnjreh03qadcajs7gwt3z",
"https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg",
"39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid",
"",
).forEach { assertTrue(isMachineValue(it), it) }
}
@Test
fun wordsAndPhrasesAreNotMachineValues() {
listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell", "{ राधेय }Rishi Verma").forEach {
assertFalse(isMachineValue(it), it)
}
}
}
@@ -476,6 +476,30 @@ class SearchFieldExtractorTest {
)
}
@Test
fun decentralizedListItemExtraTagsAreTheTextTier() {
// Deployment tags rank below what the item is called and what it is about; machine
// values and the NIP-31 alt text are not indexed at all.
val tags =
arrayOf(
arrayOf("d", "ol-ol98624w"),
arrayOf("title", "Waking with Enemies"),
arrayOf("author", "Christopher Pike"),
arrayOf("subject", "Horror tales"),
arrayOf("lang", "eng"),
arrayOf("isbn", "9781534445145"),
arrayOf("alt", "Book: Waking with Enemies by Christopher Pike"),
)
val fields = SearchFieldExtractor.extract(AddressableListItemEvent("3c".repeat(32), alice, 1L, tags, "", ""))
assertEquals(
IndexableFields.Tiered(
primary = listOf("Waking with Enemies"),
text = "Christopher Pike\nHorror tales",
),
fields,
)
}
@Test
fun textTracksIndexWhatIsSaidNotTheTimings() {
val vtt = "WEBVTT\n\n1\n00:00:00.000 --> 00:00:02.000 align:start\n<v Roger>Hello nostr\n"
@@ -58,8 +58,8 @@
9736 The content body.
9737 The content body.
9802 The Comment\nThe Context\nThe content body.
9998 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2
9999 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2
9998 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2
9999 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2
10003 The Title
10100
10154 The Title\nThe Description
@@ -141,8 +141,8 @@
39092 The Title\nThe Description
39307 The content body.
39701 The Title\nThe content body.
39998 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2
39999 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2
39998 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2
39999 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2
40002 The content body.
40100 The content body.
45001 The content body.