diff --git a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt
index baa7bc7568..62e46ced82 100644
--- a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt
+++ b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt
@@ -44,11 +44,18 @@ object MetaTagsParser {
private const val META = "meta"
private const val HEAD = "head"
+
+ // Elements whose content is text rather than markup: script and style hold raw text, title and
+ // textarea hold character data. A `<` inside any of them is not a tag.
private const val SCRIPT = "script"
private const val STYLE = "style"
+ private const val TITLE = "title"
+ private const val TEXTAREA = "textarea"
private const val SCRIPT_END = ""
@@ -202,15 +209,24 @@ object MetaTagsParser {
return if (nameIs(nameStart, nameEnd, HEAD)) TagKind.HEAD_END else TagKind.OTHER
}
- if (nameIs(nameStart, nameEnd, META)) return TagKind.META
+ // Text-only elements are skipped whole: `for (i = 0; i < n; i++)` in a script, a quote
+ // in a JS string, or the `<` and the apostrophe in `
5 < 6, that's math`
+ // are content, not markup, and scanning them as markup hides the tags that follow --
+ // the same way an unbalanced quote inside a comment does. Switching on the name length
+ // first keeps the common tag (a `
`, a ``) down to one comparison.
+ when (nameEnd - nameStart) {
+ META.length -> if (nameIs(nameStart, nameEnd, META)) return TagKind.META
- // Script and style bodies are raw text: a `<` in `for (i = 0; i < n; i++)` or a quote
- // in a JS string is not markup and must not be scanned as such, for the same reason
- // comments can't be.
- if (nameIs(nameStart, nameEnd, SCRIPT)) {
- skipRawText(SCRIPT_END)
- } else if (nameIs(nameStart, nameEnd, STYLE)) {
- skipRawText(STYLE_END)
+ STYLE.length ->
+ if (nameIs(nameStart, nameEnd, STYLE)) {
+ skipRawText(STYLE_END)
+ } else if (nameIs(nameStart, nameEnd, TITLE)) {
+ skipRawText(TITLE_END)
+ }
+
+ SCRIPT.length -> if (nameIs(nameStart, nameEnd, SCRIPT)) skipRawText(SCRIPT_END)
+
+ TEXTAREA.length -> if (nameIs(nameStart, nameEnd, TEXTAREA)) skipRawText(TEXTAREA_END)
}
return TagKind.OTHER
diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt
new file mode 100644
index 0000000000..a8ca36a12f
--- /dev/null
+++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt
@@ -0,0 +1,120 @@
+/*
+ * Copyright (c) 2025 Vitor Pamplona
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy of
+ * this software and associated documentation files (the "Software"), to deal in
+ * the Software without restriction, including without limitation the rights to use,
+ * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
+ * Software, and to permit persons to whom the Software is furnished to do so,
+ * subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in all
+ * copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
+ * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
+ * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
+ * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
+ * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+ */
+package com.vitorpamplona.amethyst.commons.preview
+
+import kotlinx.coroutines.test.runTest
+import kotlin.test.Test
+import kotlin.test.assertEquals
+
+/**
+ * The charset half of the meta scan: `` and `` are
+ * the tags that decide how the rest of the document -- including every og: value -- is decoded.
+ * Get this wrong and a preview renders mojibake rather than nothing, so it fails quietly.
+ */
+class HtmlParserCharsetTest {
+ private suspend fun firstContentOf(
+ bytes: ByteArray,
+ charsetName: String?,
+ ): String =
+ HtmlParser()
+ .parseHtml(bytes, charsetName)
+ .last()
+ .attr("content")
+
+ // -- HtmlCharsetParser --------------------------------------------------------------------
+
+ @Test
+ fun sniffsTheCharsetAttribute() {
+ assertEquals("iso-8859-1", HtmlCharsetParser.detectCharset("""""".encodeToByteArray()))
+ }
+
+ @Test
+ fun sniffsTheHttpEquivContentType() {
+ val html = """"""
+
+ assertEquals("shift_jis", HtmlCharsetParser.detectCharset(html.encodeToByteArray()))
+ }
+
+ @Test
+ fun defaultsToUtf8WhenNothingIsDeclared() {
+ assertEquals("UTF-8", HtmlCharsetParser.detectCharset("""x""".encodeToByteArray()))
+ }
+
+ @Test
+ fun onlySniffsTheFirstKilobyte() {
+ // The window is deliberate -- the declaration is required to be early -- but it means a
+ // charset pushed past 1 KB by a banner comment is not found, and UTF-8 is assumed.
+ val pushedOut = "" + "" + """"""
+
+ assertEquals("UTF-8", HtmlCharsetParser.detectCharset(pushedOut.encodeToByteArray()))
+ }
+
+ @Test
+ fun aCommentedOutCharsetIsNotSniffed() {
+ val html = """"""
+
+ assertEquals("utf-8", HtmlCharsetParser.detectCharset(html.encodeToByteArray()))
+ }
+
+ // -- HtmlParser: which charset wins --------------------------------------------------------
+
+ @Test
+ fun anExplicitCharsetWinsOverTheDocumentDeclaration() =
+ runTest {
+ // Content-Type said windows-1252; the document claims utf-8 and must not be believed.
+ val bytes =
+ byteArrayOf(0x3C) + // '<'
+ """head>""".encodeToByteArray()
+
+ assertEquals("café", firstContentOf(bytes, "windows-1252"))
+ }
+
+ @Test
+ fun aByteOrderMarkWinsOverTheDocumentDeclaration() =
+ runTest {
+ val utf16 = """"""
+ val bytes = byteArrayOf(0xFE.toByte(), 0xFF.toByte()) + utf16.encodeToUtf16Be()
+
+ assertEquals("café", firstContentOf(bytes, null))
+ }
+
+ @Test
+ fun fallsBackToTheDocumentDeclarationWhenTheResponseHasNoCharset() =
+ runTest {
+ val bytes =
+ """""".encodeToByteArray()
+
+ assertEquals("café", firstContentOf(bytes, null))
+ }
+
+ private fun String.encodeToUtf16Be(): ByteArray {
+ val out = ByteArray(length * 2)
+ forEachIndexed { i, c ->
+ out[i * 2] = (c.code shr 8).toByte()
+ out[i * 2 + 1] = (c.code and 0xFF).toByte()
+ }
+ return out
+ }
+}
diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt
index 34374cafd7..71f7189252 100644
--- a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt
+++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt
@@ -24,7 +24,8 @@ import kotlin.test.Test
import kotlin.test.assertEquals
/**
- * Comments, declarations and script bodies are not element markup. Scanning them for
+ * Comments, declarations and the text-only elements (script, style, title, textarea) are not
+ * element markup. Scanning them for
* attribute quotes lets an odd apostrophe -- "we don't" is enough -- leave the scanner
* inside a phantom quoted value, swallowing every tag up to the next quote character.
* That is what hid the entire og: block of https://brainstorm.world from link previews.
@@ -100,6 +101,75 @@ class MetaTagsParserCommentTest {
assertEquals("Real Title", metaTags[0].attr("content"))
}
+ @Test
+ fun titleTextWithALessThanDoesNotSwallowFollowingMetaTags() {
+ // `5 < 6, that's math` is ordinary HTML: an unescaped `<` in title text,
+ // then an apostrophe. Scanned as markup, the `<` opens a phantom tag and the apostrophe
+ // opens a phantom attribute value that runs to the end of the document.
+ val input =
+ """
+ |
+ | 5 < 6, that's math
+ |
+ |
+ """.trimMargin()
+
+ val metaTags = MetaTagsParser.parse(input).toList()
+
+ assertEquals(1, metaTags.size)
+ assertEquals("Real Title", metaTags[0].attr("content"))
+ }
+
+ @Test
+ fun metaTagsInsideTitleTextAreNotParsed() {
+ // Title content is character data, so this is a title that reads literally
+ // `a b`, not a second og:title.
+ val input =
+ """
+ |
+ | a b
+ |
+ |
+ """.trimMargin()
+
+ val metaTags = MetaTagsParser.parse(input).toList()
+
+ assertEquals(1, metaTags.size)
+ assertEquals("Real Title", metaTags[0].attr("content"))
+ }
+
+ @Test
+ fun textareaTextDoesNotSwallowFollowingMetaTags() {
+ val input =
+ """
+ |
+ |
+ |
+ |
+ """.trimMargin()
+
+ val metaTags = MetaTagsParser.parse(input).toList()
+
+ assertEquals(1, metaTags.size)
+ assertEquals("Real Title", metaTags[0].attr("content"))
+ }
+
+ @Test
+ fun aSelfClosedScriptStillOpensRawText() {
+ // Deliberate, and what a browser does: `/` on a script start tag is ignored, so everything
+ // up to `` is script data. A page written this way shows nothing after it either.
+ // Pinned so that "fixing" it never turns a JS string into an og: tag.
+ val input =
+ """
+ |
+ |
+ |
+ |
+ """.trimMargin()
+
+ assertEquals(0, MetaTagsParser.parse(input).count())
+ }
+
@Test
fun doctypeAndProcessingInstructionsAreSkipped() {
val input =
diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserEdgeCaseTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserEdgeCaseTest.kt
new file mode 100644
index 0000000000..1ece50c95e
--- /dev/null
+++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserEdgeCaseTest.kt
@@ -0,0 +1,208 @@
+/*
+ * Copyright (c) 2025 Vitor Pamplona
+ *
+ * Permission is hereby granted, free of charge, to any person obtaining a copy of
+ * this software and associated documentation files (the "Software"), to deal in
+ * the Software without restriction, including without limitation the rights to use,
+ * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
+ * Software, and to permit persons to whom the Software is furnished to do so,
+ * subject to the following conditions:
+ *
+ * The above copyright notice and this permission notice shall be included in all
+ * copies or substantial portions of the Software.
+ *
+ * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
+ * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
+ * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
+ * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
+ * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
+ * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+ */
+package com.vitorpamplona.amethyst.commons.preview
+
+import kotlin.test.Test
+import kotlin.test.assertEquals
+import kotlin.test.assertTrue
+
+/**
+ * Tag and attribute shapes a preview fetch can meet in the wild, and the ones a truncated or
+ * hostile response can produce. The server picks this input, so "it throws" and "it silently eats
+ * the rest of the head" both have to be ruled out for every shape here.
+ */
+class MetaTagsParserEdgeCaseTest {
+ private fun contents(html: String) = MetaTagsParser.parse(html).map { it.attr("content") }.toList()
+
+ // -- the end of the scan ------------------------------------------------------------------
+
+ @Test
+ fun stopsAtAnUppercaseHeadEndTag() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("1"), metas)
+ }
+
+ @Test
+ fun stopsAtAHeadEndTagWithTrailingSpace() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("1"), metas)
+ }
+
+ @Test
+ fun aHeadEndTagInsideAnAttributeValueDoesNotEndTheScan() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("", "2"), metas)
+ }
+
+ @Test
+ fun scansTheWholeDocumentWhenThereIsNoHeadEndTag() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("1", "2"), metas)
+ }
+
+ // -- truncated responses ------------------------------------------------------------------
+
+ @Test
+ fun aBodyTruncatedRightAfterASlashDoesNotThrow() {
+ // The `/` of a `/>` as the very last byte: the self-closing check must not read past it.
+ val metas = contents("""<"""))
+ }
+
+ // -- tag shapes ---------------------------------------------------------------------------
+
+ @Test
+ fun readsAnUppercaseMetaTagAndUppercaseAttributeNames() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("Real"), metas)
+ }
+
+ @Test
+ fun anEmptySelfClosedMetaIsSkippedWithoutDerailingTheScan() {
+ // `` has no separator before the `/`, so the name reads as `meta/` and the tag is
+ // dropped. It carries nothing anyway; what matters is that the next tag still parses.
+ val metas = contents("""""")
+
+ assertEquals(listOf("Real"), metas)
+ }
+
+ @Test
+ fun aSelfClosingSequenceInsideAQuotedValueDoesNotEndTheTag() {
+ val metas =
+ contents(
+ """""",
+ )
+
+ assertEquals(listOf("a /> b", "ok"), metas)
+ }
+
+ @Test
+ fun readsMetaTagsInsideNoscript() {
+ // noscript content is markup for a parser that isn't running scripts, and a redirect meta
+ // hidden in there is exactly the kind a preview wants to see.
+ val metas =
+ contents(
+ """""",
+ )
+
+ assertEquals(listOf("0", "Real"), metas)
+ }
+
+ @Test
+ fun skipsCdataSections() {
+ val metas = contents(""" y ]]>""")
+
+ assertEquals(listOf("Real"), metas)
+ }
+
+ // -- attribute shapes ---------------------------------------------------------------------
+
+ @Test
+ fun keepsAttributesWhenTheTagEndsWithAValuelessOne() {
+ val metas = MetaTagsParser.parse("""""").toList()
+
+ assertEquals(1, metas.size)
+ assertEquals("og:title", metas[0].attr("property"))
+ assertEquals("T", metas[0].attr("content"))
+ }
+
+ @Test
+ fun keepsAValueThatSpansLines() {
+ val metas = contents("")
+
+ assertEquals(listOf("line1\nline2"), metas)
+ }
+
+ @Test
+ fun anUnknownAttributeIsIgnoredNotFatal() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("Real"), metas)
+ }
+
+ // -- character references in values --------------------------------------------------------
+
+ @Test
+ fun keepsQueryStringAmpersandsIntact() {
+ // og:image URLs are full of `&`; only a real character reference may be resolved.
+ val metas =
+ contents(
+ """""",
+ )
+
+ assertEquals(listOf("https://x.com/i.png?w=1200&h=630&fit=crop&q=80"), metas)
+ }
+
+ @Test
+ fun decodesCharacterReferencesOutsideTheBasicMultilingualPlane() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("😀 😀 hi"), metas)
+ }
+
+ @Test
+ fun leavesAnUnknownCharacterReferenceAlone() {
+ val metas = contents("""""")
+
+ assertEquals(listOf("AT&T ¬areference; ZZ;"), metas)
+ }
+
+ // -- laziness -------------------------------------------------------------------------------
+
+ @Test
+ fun stopsReadingOnceTheConsumerStops() {
+ // OpenGraphParser bails as soon as it has title+description+image; the sequence must not
+ // have scanned the rest of the document by then. A tag after an unterminated comment is
+ // unreachable, so seeing the first one proves the scan was still lazy.
+ val html =
+ """
+ |
+ |
+ |