diff --git a/.claude/skills/searchable-events/references/searchable-kinds.md b/.claude/skills/searchable-events/references/searchable-kinds.md index 6c5ef8797d..55ac82e9f6 100644 --- a/.claude/skills/searchable-events/references/searchable-kinds.md +++ b/.claude/skills/searchable-events/references/searchable-kinds.md @@ -2,7 +2,7 @@ Every concrete `SearchableEvent` implementor in Quartz, with the exact `indexableContent()` expression. **Update this file in the same PR as any change to the searchable set or to an -`indexableContent()` body** (see SKILL.md). Verified against the code 2026-09-17. +`indexableContent()` body** (see SKILL.md). Verified against the code 2026-10-04. Counts: 139 concrete classes covering 140 kind values (`GitStatusEvent` spans 4 kinds; kind 30063 is shared by two NIPs and kind 38000 by three classes — see the footnotes). File paths are under @@ -65,7 +65,7 @@ Separator legend: **NL** = `joinToString("\n")`, **SP** = `joinToString(" ")`. | 9736 | Bolt12ZapEvent | nipB1Bolt12Zaps/zap | `content` | | 9737 | Bolt12ZapIntentEvent | nipB1Bolt12Zaps/intent | `content` | | 9802 | HighlightEvent | nip84Highlights | `listOfNotNull(comment(), context(), content)` NL | -| 9998 | ListHeaderEvent | experimental/decentralizedLists/header | `tags.searchableListContent()` NL — `names` (singular, plural), `titles` (singular, plural), `name`, `title`, `description`, `comments`, then every `t` value; ids/pubkeys/coordinates are left to tag filters | +| 9998 | ListHeaderEvent | experimental/decentralizedLists/header | `tags.searchableListContent()` NL — `names` (singular, plural), `titles` (singular, plural), `name`, `title`, `description`, `comments`, then in tag order every `t` value that is not `isMachineValue` and every value of every other tag (except `alt`, `client`, `imeta`) that passes `isNaturalLanguageValue` (has whitespace or non-ASCII, or is one capitalized letters-only word; never JSON, numbers, URIs of any scheme, addresses, hex ids, UUIDs, bech32) | | 9999 | ListItemEvent | experimental/decentralizedLists/item | same as 9998 | | 10003 | BookmarkListEvent | nip51Lists/bookmarkList | `listOfNotNull(title())` NL | | 10100 | AgentProfileEvent | buzz/agentProfiles | `profileOrNull()?.let { listOfNotNull(it.name, it.displayName).joinToString("\n") } ?: ""` | diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/README.md b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/README.md index 060f22cfb4..2f7ae612f6 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/README.md +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/README.md @@ -85,7 +85,19 @@ Filter( All four kinds are `SearchableEvent`s sharing one walk (`forEachSearchableListField`): `names` and `titles` (singular then plural), -`name`, `title`, `description`, `comments`, then every `t` item value. +`name`, `title`, `description`, `comments`, then, in tag order, every `t` +item value and every value of every other tag that reads as natural language. + +The tag set is open — deployments add `author`, `subject`, `artist`, +`relationshipType` and tags nobody has named yet — so that last step decides +by the value, not the tag (`nip50Search/NaturalLanguageValue.kt`): a value is +indexed when it has whitespace or a non-ASCII character, or is a single +capitalized word ("Fiction", "Aristotle"), and is never a machine value (JSON, +a number, a URI of any scheme, an address, a hex id, a UUID, bech32). `t` +values skip the natural-language test but not the machine one. `alt` (NIP-31 +fallback text that restates other tags), `client` and `imeta` (NIP-92 `key +value` pairs) are not indexed. + `content` is not part of the spec, and ids, pubkeys and coordinates are served by `#p`/`#e`/`#a`/`#z` filters, so none of them are indexed. diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt index 27557c2bbd..51e933d96a 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/SearchExt.kt @@ -35,13 +35,20 @@ import com.vitorpamplona.quartz.experimental.decentralizedLists.tags.Description import com.vitorpamplona.quartz.nip01Core.core.TagArray import com.vitorpamplona.quartz.nip01Core.core.fastForEach import com.vitorpamplona.quartz.nip01Core.tags.hashtags.HashtagTag +import com.vitorpamplona.quartz.nip31Alts.AltTag import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor +import com.vitorpamplona.quartz.nip50Search.isMachineValue +import com.vitorpamplona.quartz.nip50Search.isNaturalLanguageValue +import com.vitorpamplona.quartz.nip89AppHandlers.clientTag.ClientTag +import com.vitorpamplona.quartz.nip92IMeta.IMetaTag /** - * The human-authored text of any kind in the family, in a fixed order: the header's `names` - * and `titles` (singular, then plural), the item's `name` and `title`, `description`, - * `comments`, then each `t` item value. Headers and items share one walk because the spec's - * nonstandard method lets an item carry header tags; each kind simply has fewer of them set. + * The human-authored text of any kind in the family: first, in a fixed order, the header's `names` + * and `titles` (singular, then plural), the item's `name` and `title`, `description` and + * `comments`; then, in tag order, each `t` value that is not a machine value and every value of + * every other tag that reads as natural language ([forEachOtherField]). Headers and items share + * one walk because the spec's nonstandard method lets an item carry header tags; each kind simply + * has fewer of them set. * * `content` is not part of the spec and ids/pubkeys/coordinates are served by tag filters, * so neither is indexed. @@ -51,7 +58,8 @@ import com.vitorpamplona.quartz.nip50Search.IndexableFieldVisitor fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean { // The read path runs once per event per keystroke, so this is two allocation-free passes // instead of one full scan (and one parsed object) per field. The first pass only remembers - // the first well-formed tag of each field; the fixed visiting order is applied afterwards. + // the first well-formed tag of each named field, so their fixed order can be applied before + // the second pass visits everything else. var names: Array? = null var titles: Array? = null var name: String? = null @@ -83,8 +91,51 @@ fun TagArray.forEachSearchableListField(visitor: IndexableFieldVisitor): Boolean title?.let { if (!visitor.visit(it)) return false } description?.let { if (!visitor.visit(it)) return false } comments?.let { if (!visitor.visit(it)) return false } + return forEachOtherField(visitor, withHashtags = true) +} + +/** + * Every tag the first pass of [forEachSearchableListField] does not own, in tag order: each `t` + * value that is not [isMachineValue] (only [withHashtags]), and every value of every other tag + * that [isNaturalLanguageValue]. The family's tag set is open — deployments add `author`, + * `subject`, `artist`, `relationshipType` and tags nobody has named yet — so this decides by what + * a value looks like rather than by the tag it sits in. + * + * Skips the NIP-defined metadata tags that are not the event's own text: `alt` (NIP-31 fallback + * text, which here restates `title` and `artist` behind a fixed "Song: … by …" prefix), `client` + * (NIP-89 app name) and `imeta` (NIP-92 `key value` pairs, whose space would otherwise pass every + * URL and hash as text). + * + * @return false when the visitor stopped the walk. + */ +private fun TagArray.forEachOtherField( + visitor: IndexableFieldVisitor, + withHashtags: Boolean, +): Boolean { fastForEach { tag -> - HashtagTag.parse(tag)?.let { if (!visitor.visit(it)) return false } + if (tag.size < 2) return@fastForEach + when (tag[0]) { + HashtagTag.TAG_NAME -> { + if (withHashtags && !isMachineValue(tag[1]) && !visitor.visit(tag[1])) return false + } + + NamesTag.TAG_NAME, + TitlesTag.TAG_NAME, + NameTag.TAG_NAME, + TitleTag.TAG_NAME, + DescriptionTag.TAG_NAME, + CommentsTag.TAG_NAME, + AltTag.TAG_NAME, + ClientTag.TAG_NAME, + IMetaTag.TAG_NAME, + -> {} + + else -> { + for (i in 1 until tag.size) { + if (isNaturalLanguageValue(tag[i]) && !visitor.visit(tag[i])) return false + } + } + } } return true } @@ -103,10 +154,19 @@ fun TagArray.searchableListTitles(): List { /** The DESCRIPTION role of [forEachSearchableListField]: `description`, then `comments`. */ fun TagArray.searchableListDescriptions(): List = listOf(description(), comments()) +/** + * The TEXT role of [forEachSearchableListField]: the natural-language values of the tags the + * fixed-order fields do not own, one per line, or null when the event has none. The `t` values + * are not here; the search funnel carries them as hashtags. + */ +fun TagArray.searchableListExtraText(): String? = joinFields { forEachOtherField(it, withHashtags = false) }.ifEmpty { null } + /** The write-path join of [forEachSearchableListField]: one field per line. */ -fun TagArray.searchableListContent() = +fun TagArray.searchableListContent() = joinFields { forEachSearchableListField(it) } + +private inline fun joinFields(walk: (IndexableFieldVisitor) -> Boolean) = buildString { - forEachSearchableListField { field -> + walk { field -> if (field != null) { if (isNotEmpty()) append('\n') append(field) diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcher.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcher.kt index 3e005d88b2..59f1a9cdd3 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcher.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcher.kt @@ -31,8 +31,14 @@ import com.vitorpamplona.quartz.nip01Core.core.fastAny * * ## What it matches * - * Terms are ANDed and each is a case-insensitive substring — of a tag value, or of the event's - * own indexable fields. Substring rather than token because that is what Amethyst's local search + * Terms are ANDed and each is a case-insensitive substring of the event's indexable fields — the + * same text a store's full-text index holds ([SearchableEvent.forEachIndexableField] is the read + * side of [SearchableEvent.indexableContent]), so local search and a store or relay agree on what + * an event can be found by. A tag value is searched only when the kind indexes it: a hashtag, an + * id, a URL or a data label the kind leaves out of its index is left out here too. A kind that is + * not a [SearchableEvent] has no indexable fields, so it falls back to its tag values and content. + * + * Substring rather than token because that is what Amethyst's local search * has always done (`content.contains(text, true)`), and switching to tokens would silently stop * matching mid-word queries; AND rather than one literal phrase because that is what a relay does * with the same string, and the point of routing local search through a `Filter` is that the two @@ -57,8 +63,9 @@ import com.vitorpamplona.quartz.nip01Core.core.fastAny class EventSearchMatcher( search: String?, /** - * Tag names whose values are not worth searching — `p`/`e`/`a` hold hex ids that only ever - * match by accident, `client` and `alt` hold text the author did not write. + * For a kind that is not a [SearchableEvent]: tag names whose values are not worth searching — + * `p`/`e`/`a` hold hex ids that only ever match by accident, `client` and `alt` hold text the + * author did not write. */ private val exceptTagNames: Set = DEFAULT_EXCLUDED_TAGS, ) { @@ -82,11 +89,9 @@ class EventSearchMatcher( event: Event, term: String, ): Boolean { - if (event.tags.fastAny { it.size > 1 && it[0] !in exceptTagNames && it[1].contains(term, true) }) { - return true - } - if (event !is SearchableEvent) return event.content.contains(term, true) - return visitor.matches(event, term) + if (event is SearchableEvent) return visitor.matches(event, term) + return event.content.contains(term, true) || + event.tags.fastAny { it.size > 1 && it[0] !in exceptTagNames && it[1].contains(term, true) } } /** diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt new file mode 100644 index 0000000000..489c5090eb --- /dev/null +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValue.kt @@ -0,0 +1,238 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip50Search + +/** + * Whether a free-form tag value reads as natural language, for kinds whose tag set is open + * (anyone can invent a tag name) and so cannot be indexed from a list of known tag names. + * + * A value qualifies when it is not [isMachineValue] and it either contains whitespace, contains a + * non-ASCII character, or is a single capitalized ASCII word made of letters only. That keeps + * "Social history", "刘慈欣", "Fiction" and "Aristotle", and drops the single lowercase or mixed + * tokens that label data rather than describe it: `music`, `eng`, `openlibrary`, `IS_A_PROPERTY_OF`, + * `concept-header`, `OL19722168W`, `2026-06-01`. + * + * Allocation-free, and a single scan for nearly every value: the read path calls it once per + * value per event per keystroke. + */ +fun isNaturalLanguageValue(value: String): Boolean { + var start = 0 + var end = value.length + while (start < end && value[start].isSpace()) start++ + while (end > start && value[end - 1].isSpace()) end-- + if (start == end || isStructuredValue(value, start, end)) return false + + var nonAscii = false + for (i in start until end) { + val c = value[i] + // The token shapes in isMachineToken cannot contain whitespace. + if (c.isSpace()) return true + if (c.code > 127) nonAscii = true + } + + // A single token. A capitalized letters-only word can only collide with a hex id, so the + // other machine checks are needed just for the rare non-ASCII token, such as an IRI. + return if (nonAscii) { + !isMachineToken(value, start, end) + } else { + isCapitalizedWord(value, start, end) && !isHexId(value, start, end) + } +} + +/** + * Whether a tag value is an identifier or a structure rather than text: a JSON object or array, + * a URL (`https://…`, `wss://…`, even with unescaped spaces in its path), an event address + * (`kind:pubkey:d`, even with spaces in `d`), or, as a single token, a number (including a + * signed one and an ISBN-10 ending in `X`), a URI of any scheme (`tag:…`, `urn:…`, `isbn:…`), + * a long hex id, a UUID, or a NIP-19 bech32 entity. Empty and blank values count as machine + * values too: there is nothing in them to index. + */ +fun isMachineValue(value: String): Boolean { + var start = 0 + var end = value.length + while (start < end && value[start].isSpace()) start++ + while (end > start && value[end - 1].isSpace()) end-- + if (start == end || isStructuredValue(value, start, end)) return true + for (i in start until end) if (value[i].isSpace()) return false + return isMachineToken(value, start, end) +} + +/** The machine shapes recognizable from their first characters, whitespace or not. */ +private fun isStructuredValue( + v: String, + start: Int, + end: Int, +): Boolean { + val first = v[start] + val last = v[end - 1] + if ((first == '{' && last == '}') || (first == '[' && last == ']')) return true + return isUrl(v, start, end) || isAddress(v, start, end) +} + +/** The machine shapes that are a single token. */ +private fun isMachineToken( + v: String, + start: Int, + end: Int, +): Boolean = + isNumber(v, start, end) || + isUri(v, start, end) || + isHexId(v, start, end) || + isUuid(v, start, end) || + isBech32(v, start) + +/** + * [isWhitespace] with an ASCII fast path: on the JVM, [isWhitespace] is two Unicode table lookups + * per character, and nearly every character this file scans is ASCII. + */ +private fun Char.isSpace() = if (code < 0x80) this == ' ' || this in '\t'..'\r' || this in '\u001C'..'\u001F' else isWhitespace() + +private fun Char.isAsciiDigit() = this in '0'..'9' + +private fun Char.isAsciiLetter() = this in 'a'..'z' || this in 'A'..'Z' + +private fun Char.isHex() = this in '0'..'9' || this in 'a'..'f' || this in 'A'..'F' + +/** `-1`, `473`, `0.5`, an ISBN-13, and an ISBN-10 with its `X` check digit. */ +private fun isNumber( + v: String, + start: Int, + end: Int, +): Boolean { + var i = start + if (v[i] == '-' || v[i] == '+') i++ + val digitsStart = i + while (i < end && v[i].isAsciiDigit()) i++ + if (i == digitsStart) return false + if (i < end && v[i] == '.') { + i++ + val fractionStart = i + while (i < end && v[i].isAsciiDigit()) i++ + if (i == fractionStart) return false + } + if (i < end && (v[i] == 'X' || v[i] == 'x')) i++ + return i == end +} + +/** The index of the `:` that ends a leading RFC 3986 scheme, or -1 when there is none. */ +private fun schemeEnd( + v: String, + start: Int, + end: Int, +): Int { + if (!v[start].isAsciiLetter()) return -1 + var i = start + 1 + while (i < end) { + val c = v[i] + if (c == ':') return i + if (!(c.isAsciiLetter() || c.isAsciiDigit() || c == '+' || c == '.' || c == '-')) return -1 + i++ + } + return -1 +} + +/** A scheme followed by `:` and something more, as a single token. */ +private fun isUri( + v: String, + start: Int, + end: Int, +): Boolean { + val colon = schemeEnd(v, start, end) + return colon >= 0 && colon + 1 < end +} + +/** A scheme followed by `://`. Unlike "Song: Gold", never text. */ +private fun isUrl( + v: String, + start: Int, + end: Int, +): Boolean { + val colon = schemeEnd(v, start, end) + return colon >= 0 && colon + 2 < end && v[colon + 1] == '/' && v[colon + 2] == '/' +} + +/** `:<64-hex pubkey>` optionally followed by `:`. */ +private fun isAddress( + v: String, + start: Int, + end: Int, +): Boolean { + var i = start + while (i < end && v[i].isAsciiDigit()) i++ + if (i == start || i >= end || v[i] != ':') return false + i++ + val pubkeyEnd = i + 64 + if (pubkeyEnd > end) return false + while (i < pubkeyEnd) { + if (!v[i].isHex()) return false + i++ + } + return i == end || v[i] == ':' +} + +/** Event ids, pubkeys, and the shorter hashes (32+ hex characters) feeds use as ids. */ +private fun isHexId( + v: String, + start: Int, + end: Int, +): Boolean { + if (end - start < 32) return false + for (i in start until end) if (!v[i].isHex()) return false + return true +} + +private fun isUuid( + v: String, + start: Int, + end: Int, +): Boolean { + if (end - start != 36) return false + for (i in 0 until 36) { + val c = v[start + i] + val hyphen = i == 8 || i == 13 || i == 18 || i == 23 + if (if (hyphen) c != '-' else !c.isHex()) return false + } + return true +} + +private val BECH32_PREFIXES = arrayOf("npub1", "nsec1", "note1", "nevent1", "naddr1", "nprofile1", "nrelay1") + +private fun isBech32( + v: String, + start: Int, +): Boolean { + for (prefix in BECH32_PREFIXES) if (v.startsWith(prefix, start)) return true + return false +} + +/** `Fiction`, `Aristotle`, `RTL`, `AT&T`: an ASCII capital followed by letters only. */ +private fun isCapitalizedWord( + v: String, + start: Int, + end: Int, +): Boolean { + if (v[start] !in 'A'..'Z') return false + for (i in start + 1 until end) { + val c = v[i] + if (!(c.isAsciiLetter() || c == '\'' || c == '&' || c == '.')) return false + } + return true +} diff --git a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractor.kt b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractor.kt index dc43fd3943..b91786e25c 100644 --- a/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractor.kt +++ b/quartz/src/commonMain/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractor.kt @@ -31,6 +31,7 @@ import com.vitorpamplona.quartz.experimental.birdstar.BirdDetectionEvent import com.vitorpamplona.quartz.experimental.birdstar.BirdexEvent import com.vitorpamplona.quartz.experimental.decentralizedLists.DecentralizedListEvent import com.vitorpamplona.quartz.experimental.decentralizedLists.searchableListDescriptions +import com.vitorpamplona.quartz.experimental.decentralizedLists.searchableListExtraText import com.vitorpamplona.quartz.experimental.decentralizedLists.searchableListTitles import com.vitorpamplona.quartz.experimental.edits.TextNoteModificationEvent import com.vitorpamplona.quartz.experimental.fitness.workout.ExerciseTemplateEvent @@ -629,9 +630,10 @@ object SearchFieldExtractor { // already, so -- as for 1111/1311/30382 -- they are not passed again, and // content is not part of the spec. One branch for all four: an item may // carry header tags (the spec's nonstandard method), and the tag helpers - // read whichever are present. + // read whichever are present. Every other natural-language tag value + // (author, subject, artist, ...) is the text tier, below both. is DecentralizedListEvent -> { - tiers(event, event.tags.searchableListTitles(), event.tags.searchableListDescriptions(), null) + tiers(event, event.tags.searchableListTitles(), event.tags.searchableListDescriptions(), event.tags.searchableListExtraText()) } // kind 1 LAST among the explicit branches, defensively: a future diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt index 4a0e5511f0..7853830fa7 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/experimental/decentralizedLists/DecentralizedListsTest.kt @@ -299,7 +299,7 @@ class DecentralizedListsTest { ) assertEquals("Fido\nVery good\nFido", item.indexableContent()) - listOf(header, item).forEach { event -> + listOf(header, item, book).forEach { event -> val fields = mutableListOf() event.forEachIndexableField { field -> field?.let { fields.add(it) } @@ -310,4 +310,56 @@ class DecentralizedListsTest { assertEquals("", assertIs(event(9999, emptyArray())).indexableContent()) } + + // A book item as a Decentralized Lists deployment publishes it: deployment-specific tags + // around the spec's own, most of them machine data. + private val book = + assertIs( + event( + 39999, + arrayOf( + arrayOf("d", "ol-ol1141875w"), + arrayOf("z", "39998:$author:books"), + arrayOf("title", "Herfsttij der Middeleeuwen"), + arrayOf("t", "history"), + arrayOf("t", "tag:soundcloud,2010:tracks/711391888"), + arrayOf("t", "https://example.com/ep/1"), + arrayOf("author", "Johan Huizinga"), + arrayOf("subject", "Middle Ages"), + arrayOf("subject", "Fiction"), + arrayOf("subject", "Civilisation médiévale"), + arrayOf("open-library-id", "OL1141875W"), + arrayOf("isbn", "9780226360935"), + arrayOf("isbn10", "057507681X"), + arrayOf("year", "1919"), + arrayOf("lang", "eng"), + arrayOf("source", "openlibrary"), + arrayOf("cover", "https://covers.openlibrary.org/b/id/7245546-L.jpg"), + arrayOf("json", "{\"word\": {\"name\": \"Herfsttij der Middeleeuwen\"}}"), + arrayOf("p", author, "wss://relay.example.com", "Johan Fan"), + arrayOf("e", headerId, "wss://relay.example.com", "mention"), + arrayOf("alt", "Book: Herfsttij der Middeleeuwen by Johan Huizinga"), + arrayOf("client", "Brainstorm"), + arrayOf("imeta", "url https://example.com/cover.jpg", "m image/jpeg", "alt A book cover"), + arrayOf("artwork", "https://example.com/Cover Art.jpg"), + ), + ), + ) + + @Test + fun searchIndexesEveryNaturalLanguageTagValue() { + assertEquals( + "Herfsttij der Middeleeuwen\nhistory\nJohan Huizinga\nMiddle Ages\nFiction\nCivilisation médiévale\nJohan Fan", + book.indexableContent(), + ) + } + + @Test + fun searchSkipsMachineHashtagsButKeepsSingleWordOnes() { + val item = + assertIs( + event(9999, arrayOf(arrayOf("t", "romance"), arrayOf("t", "8ad7c296-67e9-5f57-ba52-2ff732e62e87"), arrayOf("t", "473"))), + ) + assertEquals("romance", item.indexableContent()) + } } diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SearchTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SearchTest.kt index 23024b8be5..0d646b93fc 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SearchTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip01Core/store/sqlite/SearchTest.kt @@ -20,6 +20,7 @@ */ package com.vitorpamplona.quartz.nip01Core.store.sqlite +import com.vitorpamplona.quartz.experimental.decentralizedLists.item.AddressableListItemEvent import com.vitorpamplona.quartz.experimental.trustedLists.metric import com.vitorpamplona.quartz.experimental.trustedLists.title import com.vitorpamplona.quartz.experimental.trustedLists.users.UserTrustedListEvent @@ -221,6 +222,35 @@ class SearchTest : BaseDBTest() { db.assertQuery(null, Filter(search = memberKey)) } + @Test + fun testDecentralizedListItemsAreSearchableByEveryNaturalLanguageTag() = + forEachDB { db -> + val item = + signer.sign( + TimeUtils.now(), + AddressableListItemEvent.KIND, + arrayOf( + arrayOf("d", "ol-uniqbookslug"), + arrayOf("title", "Uniqtitle Middeleeuwen"), + arrayOf("author", "Johan Uniqauthor"), + arrayOf("subject", "Uniqsubject"), + arrayOf("lang", "uniqlang"), + arrayOf("alt", "Book: Uniqalt"), + ), + "", + ) + db.store.insertEvent(item) + + db.assertQuery(item, Filter(search = "uniqtitle")) + db.assertQuery(item, Filter(search = "uniqauthor")) + db.assertQuery(item, Filter(search = "uniqsubject")) + + // the slug, a lowercase data label and the NIP-31 fallback text are not indexed + db.assertQuery(null, Filter(search = "uniqbookslug")) + db.assertQuery(null, Filter(search = "uniqlang")) + db.assertQuery(null, Filter(search = "uniqalt")) + } + @Test fun testProfileJsonFieldsAreSearchable() = forEachDB { db -> diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcherTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcherTest.kt index cf35a7b7a7..fd6a049a3d 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcherTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/EventSearchMatcherTest.kt @@ -66,18 +66,49 @@ class EventSearchMatcherTest { } @Test - fun tagValuesAreSearchedExceptTheOnesNobodyMeansToSearch() { + fun aSearchableKindMatchesWhatItIndexesAndNothingElse() { + // Kind 1 indexes its subject and content, as a store's full-text index does; a hashtag + // the kind leaves out of its index is left out here too, so local and store search agree. assertTrue(matches("subj", note("body", arrayOf(arrayOf("subject", "a subject"))))) - assertTrue(matches("bitcoin", note("body", arrayOf(arrayOf("t", "bitcoin"))))) - // `p`/`e`/`a` hold hex ids, `client` and `alt` hold text the author did not write. + assertFalse(matches("bitcoin", note("body", arrayOf(arrayOf("t", "bitcoin"))))) assertFalse(matches("amethyst", note("body", arrayOf(arrayOf("client", "amethyst"))))) assertFalse(matches("deadbeef", note("body", arrayOf(arrayOf("p", "deadbeef".repeat(8)))))) } + @Test + fun aDecentralizedListItemMatchesItsNaturalLanguageTagsButNotItsData() { + val book = + note( + "", + arrayOf( + arrayOf("d", "ol-ol1141875w"), + arrayOf("title", "Herfsttij der Middeleeuwen"), + arrayOf("author", "Johan Huizinga"), + arrayOf("subject", "Middle Ages"), + arrayOf("medium", "music"), + arrayOf("cover", "https://covers.openlibrary.org/b/id/7245546-L.jpg"), + ), + kind = 39999, + ) + assertTrue(matches("huizinga", book)) + assertTrue(matches("middle ages", book)) + assertFalse(matches("music", book)) + assertFalse(matches("openlibrary", book)) + assertFalse(matches("ol1141875w", book)) + } + + @Test + fun aKindWithNoIndexableFieldsFallsBackToItsTagsAndContent() { + assertTrue(matches("bitcoin", note("+", arrayOf(arrayOf("t", "bitcoin")), kind = 7))) + assertTrue(matches("+", note("+", kind = 7))) + // `p`/`e`/`a` hold hex ids, `client` and `alt` hold text the author did not write. + assertFalse(matches("amethyst", note("+", arrayOf(arrayOf("client", "amethyst")), kind = 7))) + assertFalse(matches("deadbeef", note("+", arrayOf(arrayOf("p", "deadbeef".repeat(8))), kind = 7))) + } + @Test fun indexableFieldsBeyondContentAreSearched() { - // A long-form title lives in a tag, but its summary reaches the matcher through the - // visitor as well — both paths must find it. + // A long-form title and summary live in tags; they reach the matcher through the visitor. val article = note("the body", arrayOf(arrayOf("title", "Lightning"), arrayOf("summary", "a summary")), kind = 30023) assertTrue(matches("Lightning", article)) assertTrue(matches("summary", article)) diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt new file mode 100644 index 0000000000..ee79e85d3f --- /dev/null +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/NaturalLanguageValueTest.kt @@ -0,0 +1,117 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.quartz.nip50Search + +import kotlin.test.Test +import kotlin.test.assertFalse +import kotlin.test.assertTrue + +/** Values taken from the kind 9999/39999 corpus on a Decentralized Lists relay. */ +class NaturalLanguageValueTest { + @Test + fun phrasesAndNonAsciiTextAreNaturalLanguage() { + listOf( + "Social history", + "Johan Huizinga", + "Guevara, ernesto, 1928-1967", + "Climbing The Mountain (1979-1990)", + "[Live] Closer To Somewhere", + "Song: Gold by Torcon 7", + "刘慈欣", + "老子", + "Moyen Âge", + " padded phrase ", + ).forEach { assertTrue(isNaturalLanguageValue(it), it) } + } + + @Test + fun capitalizedWordsAreNaturalLanguage() { + listOf("Fiction", "History", "Aristotle", "Plutarch", "RTL", "PromoDJ", "AT&T").forEach { + assertTrue(isNaturalLanguageValue(it), it) + } + } + + @Test + fun singleLowercaseOrMixedTokensAreNot() { + listOf( + "music", + "eng", + "openlibrary", + "reference", + "string", + "forward", + "IS_THE_CONCEPT_FOR", + "concept-header", + "tagassert--ol-ol49463w--science-fiction--6aca05b8", + "OL19722168W", + "2026-06-01", + "sorita67", + "Musician:", + ).forEach { assertFalse(isNaturalLanguageValue(it), it) } + } + + @Test + fun machineValuesAreNotNaturalLanguageEvenWithSpaces() { + listOf( + "", + " ", + "{\"word\": {\"slug\": \"ol-ol98624w\", \"name\": \"Waking with Enemies\"}}", + "[\"title\", \"artist\"]", + "Re:Zero", + // URLs and addresses with unescaped spaces, as feeds publish them + "https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg", + "https://headstarts.uk/msp/longy/Katherines Wheel/katherines wheel.xml", + "39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid", + "ABCDEFABCDEFABCDEFABCDEFABCDEFAB", + ).forEach { assertFalse(isNaturalLanguageValue(it), it) } + } + + @Test + fun machineValues() { + listOf( + "473", + "-1", + "0.5", + "9781984820983", + "057507681X", + "https://covers.openlibrary.org/b/id/7245546-L.jpg", + "wss://relay.damus.io", + "tag:soundcloud,2010:tracks/711391888", + "urn:isbn:9780679428329", + "39999:6aca05b812da97601151776d13de04ae71afc9d86da1408f0e72cffef72ece4b:ol-ol2838774w", + "39998:6aca05b812da97601151776d13de04ae71afc9d86da1408f0e72cffef72ece4b", + "6aca05b812da97601151776d13de04ae71afc9d86da1408f0e72cffef72ece4b", + "c65d75f4d058f4746e12e44441cacea0", + "8ad7c296-67e9-5f57-ba52-2ff732e62e87", + "npub1f5pre6wl6ad87vr4hr5wppqq30sh58m4p33mthnjreh03qadcajs7gwt3z", + "https://wlvl.com/assets/images/podcasts/Century Podcast logo.jpg", + "39998:2efaa715bbb46dd5be6b7da8d7700266d11674b913b8178addb5c2e63d987331:first one no uuid", + "", + ).forEach { assertTrue(isMachineValue(it), it) } + } + + @Test + fun wordsAndPhrasesAreNotMachineValues() { + listOf("history", "literary-fiction", "Song: Gold", "Fiction", "Paris, 1920", "1984 Orwell", "{ राधेय }Rishi Verma").forEach { + assertFalse(isMachineValue(it), it) + } + } +} diff --git a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractorTest.kt b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractorTest.kt index 617a4fe607..81e8821edb 100644 --- a/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractorTest.kt +++ b/quartz/src/commonTest/kotlin/com/vitorpamplona/quartz/nip50Search/SearchFieldExtractorTest.kt @@ -476,6 +476,30 @@ class SearchFieldExtractorTest { ) } + @Test + fun decentralizedListItemExtraTagsAreTheTextTier() { + // Deployment tags rank below what the item is called and what it is about; machine + // values and the NIP-31 alt text are not indexed at all. + val tags = + arrayOf( + arrayOf("d", "ol-ol98624w"), + arrayOf("title", "Waking with Enemies"), + arrayOf("author", "Christopher Pike"), + arrayOf("subject", "Horror tales"), + arrayOf("lang", "eng"), + arrayOf("isbn", "9781534445145"), + arrayOf("alt", "Book: Waking with Enemies by Christopher Pike"), + ) + val fields = SearchFieldExtractor.extract(AddressableListItemEvent("3c".repeat(32), alice, 1L, tags, "", "")) + assertEquals( + IndexableFields.Tiered( + primary = listOf("Waking with Enemies"), + text = "Christopher Pike\nHorror tales", + ), + fields, + ) + } + @Test fun textTracksIndexWhatIsSaidNotTheTimings() { val vtt = "WEBVTT\n\n1\n00:00:00.000 --> 00:00:02.000 align:start\nHello nostr\n" diff --git a/quartz/src/jvmTest/resources/indexable-content.golden b/quartz/src/jvmTest/resources/indexable-content.golden index ba9a7f11a5..0c8432d694 100644 --- a/quartz/src/jvmTest/resources/indexable-content.golden +++ b/quartz/src/jvmTest/resources/indexable-content.golden @@ -58,8 +58,8 @@ 9736 The content body. 9737 The content body. 9802 The Comment\nThe Context\nThe content body. -9998 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2 -9999 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2 +9998 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2 +9999 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2 10003 The Title 10100 10154 The Title\nThe Description @@ -141,8 +141,8 @@ 39092 The Title\nThe Description 39307 The content body. 39701 The Title\nThe content body. -39998 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2 -39999 The Name\nThe Title\nThe Description\nhashtag1\nhashtag2 +39998 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2 +39999 The Name\nThe Title\nThe Description\nThe Subject\nThe Summary\nThe About\nThe Comment\nThe Context\nThe Rules\nThe Text\nThe Instructions\nhashtag1\nhashtag2 40002 The content body. 40100 The content body. 45001 The content body.