diff --git a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt index baa7bc7568..62e46ced82 100644 --- a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt +++ b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt @@ -44,11 +44,18 @@ object MetaTagsParser { private const val META = "meta" private const val HEAD = "head" + + // Elements whose content is text rather than markup: script and style hold raw text, title and + // textarea hold character data. A `<` inside any of them is not a tag. private const val SCRIPT = "script" private const val STYLE = "style" + private const val TITLE = "title" + private const val TEXTAREA = "textarea" private const val SCRIPT_END = "5 < 6, that's math` + // are content, not markup, and scanning them as markup hides the tags that follow -- + // the same way an unbalanced quote inside a comment does. Switching on the name length + // first keeps the common tag (a `
`, a ``) down to one comparison. + when (nameEnd - nameStart) { + META.length -> if (nameIs(nameStart, nameEnd, META)) return TagKind.META - // Script and style bodies are raw text: a `<` in `for (i = 0; i < n; i++)` or a quote - // in a JS string is not markup and must not be scanned as such, for the same reason - // comments can't be. - if (nameIs(nameStart, nameEnd, SCRIPT)) { - skipRawText(SCRIPT_END) - } else if (nameIs(nameStart, nameEnd, STYLE)) { - skipRawText(STYLE_END) + STYLE.length -> + if (nameIs(nameStart, nameEnd, STYLE)) { + skipRawText(STYLE_END) + } else if (nameIs(nameStart, nameEnd, TITLE)) { + skipRawText(TITLE_END) + } + + SCRIPT.length -> if (nameIs(nameStart, nameEnd, SCRIPT)) skipRawText(SCRIPT_END) + + TEXTAREA.length -> if (nameIs(nameStart, nameEnd, TEXTAREA)) skipRawText(TEXTAREA_END) } return TagKind.OTHER diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt new file mode 100644 index 0000000000..a8ca36a12f --- /dev/null +++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt @@ -0,0 +1,120 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.commons.preview + +import kotlinx.coroutines.test.runTest +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * The charset half of the meta scan: `` and `` are + * the tags that decide how the rest of the document -- including every og: value -- is decoded. + * Get this wrong and a preview renders mojibake rather than nothing, so it fails quietly. + */ +class HtmlParserCharsetTest { + private suspend fun firstContentOf( + bytes: ByteArray, + charsetName: String?, + ): String = + HtmlParser() + .parseHtml(bytes, charsetName) + .last() + .attr("content") + + // -- HtmlCharsetParser -------------------------------------------------------------------- + + @Test + fun sniffsTheCharsetAttribute() { + assertEquals("iso-8859-1", HtmlCharsetParser.detectCharset("""""".encodeToByteArray())) + } + + @Test + fun sniffsTheHttpEquivContentType() { + val html = """""" + + assertEquals("shift_jis", HtmlCharsetParser.detectCharset(html.encodeToByteArray())) + } + + @Test + fun defaultsToUtf8WhenNothingIsDeclared() { + assertEquals("UTF-8", HtmlCharsetParser.detectCharset("""x""".encodeToByteArray())) + } + + @Test + fun onlySniffsTheFirstKilobyte() { + // The window is deliberate -- the declaration is required to be early -- but it means a + // charset pushed past 1 KB by a banner comment is not found, and UTF-8 is assumed. + val pushedOut = "" + "" + """""" + + assertEquals("UTF-8", HtmlCharsetParser.detectCharset(pushedOut.encodeToByteArray())) + } + + @Test + fun aCommentedOutCharsetIsNotSniffed() { + val html = """""" + + assertEquals("utf-8", HtmlCharsetParser.detectCharset(html.encodeToByteArray())) + } + + // -- HtmlParser: which charset wins -------------------------------------------------------- + + @Test + fun anExplicitCharsetWinsOverTheDocumentDeclaration() = + runTest { + // Content-Type said windows-1252; the document claims utf-8 and must not be believed. + val bytes = + byteArrayOf(0x3C) + // '<' + """head>""".encodeToByteArray() + + assertEquals("café", firstContentOf(bytes, "windows-1252")) + } + + @Test + fun aByteOrderMarkWinsOverTheDocumentDeclaration() = + runTest { + val utf16 = """""" + val bytes = byteArrayOf(0xFE.toByte(), 0xFF.toByte()) + utf16.encodeToUtf16Be() + + assertEquals("café", firstContentOf(bytes, null)) + } + + @Test + fun fallsBackToTheDocumentDeclarationWhenTheResponseHasNoCharset() = + runTest { + val bytes = + """""".encodeToByteArray() + + assertEquals("café", firstContentOf(bytes, null)) + } + + private fun String.encodeToUtf16Be(): ByteArray { + val out = ByteArray(length * 2) + forEachIndexed { i, c -> + out[i * 2] = (c.code shr 8).toByte() + out[i * 2 + 1] = (c.code and 0xFF).toByte() + } + return out + } +} diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt index 34374cafd7..71f7189252 100644 --- a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt +++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt @@ -24,7 +24,8 @@ import kotlin.test.Test import kotlin.test.assertEquals /** - * Comments, declarations and script bodies are not element markup. Scanning them for + * Comments, declarations and the text-only elements (script, style, title, textarea) are not + * element markup. Scanning them for * attribute quotes lets an odd apostrophe -- "we don't" is enough -- leave the scanner * inside a phantom quoted value, swallowing every tag up to the next quote character. * That is what hid the entire og: block of https://brainstorm.world from link previews. @@ -100,6 +101,75 @@ class MetaTagsParserCommentTest { assertEquals("Real Title", metaTags[0].attr("content")) } + @Test + fun titleTextWithALessThanDoesNotSwallowFollowingMetaTags() { + // `5 < 6, that's math` is ordinary HTML: an unescaped `<` in title text, + // then an apostrophe. Scanned as markup, the `<` opens a phantom tag and the apostrophe + // opens a phantom attribute value that runs to the end of the document. + val input = + """ + | + | 5 < 6, that's math + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun metaTagsInsideTitleTextAreNotParsed() { + // Title content is character data, so this is a title that reads literally + // `a b`, not a second og:title. + val input = + """ + | + | a <meta property="og:title" content="Fake"> b + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun textareaTextDoesNotSwallowFollowingMetaTags() { + val input = + """ + | + | + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun aSelfClosedScriptStillOpensRawText() { + // Deliberate, and what a browser does: `/` on a script start tag is ignored, so everything + // up to `` is script data. A page written this way shows nothing after it either. + // Pinned so that "fixing" it never turns a JS string into an og: tag. + val input = + """ + | + |