diff --git a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt index 77092cadb2..62e46ced82 100644 --- a/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt +++ b/commons/src/commonMain/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParser.kt @@ -21,7 +21,6 @@ package com.vitorpamplona.amethyst.commons.preview import com.vitorpamplona.amethyst.commons.util.codePointToChars -import kotlinx.collections.immutable.toImmutableMap data class MetaTag( private val attrs: Map, @@ -32,9 +31,44 @@ data class MetaTag( fun attr(name: String): String = attrs[name.lowercase()] ?: "" } +/** + * Extracts `` tags out of a (possibly partial) HTML document. + * + * This runs on every link preview, over bytes straight off the network, so the scan touches each + * character once and allocates nothing until an actual `` shows up: `<` is found with + * [String.indexOf], tag names are compared in place against the four names that matter, and only a + * meta tag's attribute span is ever handed to [parseAttrs]. + */ object MetaTagsParser { - private val NON_ATTR_NAME_CHARS = setOf(Char(0x0), '"', '\'', '>', '/') - private val NON_UNQUOTED_ATTR_VALUE_CHARS = setOf('"', '\'', '=', '>', '<', '`') + private const val NO_QUOTE = ' ' + + private const val META = "meta" + private const val HEAD = "head" + + // Elements whose content is text rather than markup: script and style hold raw text, title and + // textarea hold character data. A `<` inside any of them is not a tag. + private const val SCRIPT = "script" + private const val STYLE = "style" + private const val TITLE = "title" + private const val TEXTAREA = "textarea" + + private const val SCRIPT_END = "` start tag: its attribute span is [TagScanner.attrsStart]..<[TagScanner.attrsEnd]. */ + META, + + /** The `` that ends the interesting part of the document. */ + HEAD_END, + + /** Anything else: other elements, comments, declarations, unparseable markup. */ + OTHER, + } /** * Lazily parse a partial HTML document and extract meta tags. @@ -43,100 +77,176 @@ object MetaTagsParser { sequence { val s = TagScanner(input) while (!s.exhausted()) { - val t = s.nextTag() ?: continue - if (t.name == "head" && t.isEnd) { - break - } - if (t.name == "meta") { - val attrs = parseAttrs(t.attrPart) ?: continue + val kind = s.nextTag() + if (kind == TagKind.HEAD_END) break + if (kind == TagKind.META) { + val attrs = parseAttrs(input, s.attrsStart, s.attrsEnd) ?: continue yield(MetaTag(attrs)) } } } - private data class RawTag( - val isEnd: Boolean, - val name: String, - val attrPart: String, - ) - private class TagScanner( private val input: String, ) { + private val length = input.length private var p = 0 - fun exhausted(): Boolean = p >= input.length + /** Attribute span of the tag [nextTag] last reported as [TagKind.META]. */ + var attrsStart = 0 + private set + var attrsEnd = 0 + private set - private fun peek(): Char = input[p] + fun exhausted(): Boolean = p >= length - private fun consume(): Char = input[p++] + /** + * True when `input[from.. Boolean) { - while (!this.exhausted() && pred(this.peek())) { - this.consume() + private fun skipComment() { + val end = input.indexOf(COMMENT_END, p) + p = if (end < 0) length else end + COMMENT_END.length + } + + private fun skipToTagEnd() { + val end = input.indexOf('>', p) + p = if (end < 0) length else end + 1 + } + + /** Leaves [p] on the `= length) return TagKind.OTHER + + // ``, `` and `` are not element markup, so the + // attribute-quote tracking below must not run over them. A comment holding an odd + // number of quote characters -- an apostrophe in "we don't", a lone `"` -- would + // otherwise leave the scanner inside a phantom quoted attribute value and make it + // swallow every tag that follows, until the next matching quote character. That is + // enough to hide a page's whole `` block from the preview. + val first = input[p] + if (first == '!' || first == '?') { + if (input.startsWith(COMMENT_START, p)) { + skipComment() + } else { + skipToTagEnd() + } + return TagKind.OTHER + } + + // read the tag name + val isEnd = first == '/' + if (isEnd) p++ val nameStart = p - skipWhile { !it.isWhitespace() && it != '>' } + while (p < length && !input[p].isWhitespace() && input[p] != '>') p++ val nameEnd = p - // seek to start of attrs part - skipSpaces() - val attrsStart = p + // seek to the start of the attrs part + while (p < length && input[p].isWhitespace()) p++ + attrsStart = p - // skip until end of tag - var quote: Char? = null - while (!exhausted()) { - val c = consume() - when { - // `/>` out of quote -> end of tag - quote == null && c == '/' && peek() == '>' -> { - consume() + // skip to the end of the tag, tracking quoted values so a `>` inside one doesn't end it + var i = p + var quote = NO_QUOTE + while (i < length) { + val c = input[i] + if (quote == NO_QUOTE) { + // `>` or `/>` out of quote -> end of tag + if (c == '>') { + i++ break } - - // `>` out of quote -> end of tag - quote == null && c == '>' -> { + if (c == '/' && i + 1 < length && input[i + 1] == '>') { + i += 2 break } - - // entering quote - quote == null && (c == '\'' || c == '"') -> { - quote = c - } - - // leaving quote - quote != null && c == quote -> { - quote = null - } + if (c == '"' || c == '\'') quote = c + } else if (c == quote) { + quote = NO_QUOTE } + i++ } - val attrsEnd = p - 1 + p = i + attrsEnd = i - 1 - val name = input.slice(nameStart..5 < 6, that's math` + // are content, not markup, and scanning them as markup hides the tags that follow -- + // the same way an unbalanced quote inside a comment does. Switching on the name length + // first keeps the common tag (a `
`, a ``) down to one comparison. + when (nameEnd - nameStart) { + META.length -> if (nameIs(nameStart, nameEnd, META)) return TagKind.META + + STYLE.length -> + if (nameIs(nameStart, nameEnd, STYLE)) { + skipRawText(STYLE_END) + } else if (nameIs(nameStart, nameEnd, TITLE)) { + skipRawText(TITLE_END) + } + + SCRIPT.length -> if (nameIs(nameStart, nameEnd, SCRIPT)) skipRawText(SCRIPT_END) + + TEXTAREA.length -> if (nameIs(nameStart, nameEnd, TEXTAREA)) skipRawText(TEXTAREA_END) + } + + return TagKind.OTHER } } + // These two are `when` branches rather than a `Set` because `Set.contains` boxes + // the char, once per attribute character of every meta tag. + private fun isNonAttrNameChar(c: Char): Boolean = + when (c) { + '\u0000', '"', '\'', '>', '/' -> true + else -> false + } + + private fun isNonUnquotedAttrValueChar(c: Char): Boolean = + when (c) { + '"', '\'', '=', '>', '<', '`' -> true + else -> false + } + // map of HTML element attribute name to its value, with additional logics: // - attribute names are matched in a case-insensitive manner // - attribute names never duplicate @@ -216,16 +326,20 @@ object MetaTagsParser { private val attrs = mutableMapOf() - fun add(attr: Pair) { - val name = attr.first.lowercase() - if (attrs.containsKey(name)) { - throw IllegalArgumentException("duplicated attribute name: $name") - } - val value = attr.second.replace(RE_CHAR_REF, Companion::replaceCharRefs) - attrs += Pair(name, value) + /** Adds an attribute, returning false if that name was already set (the first value wins). */ + fun add( + name: String, + value: String, + ): Boolean { + val key = name.lowercase() + if (attrs.containsKey(key)) return false + // Resolving character references is the expensive half of an attribute and almost no + // value has an `&` in it, so the scan for one pays for itself. + attrs[key] = if (value.indexOf('&') < 0) value else value.replace(RE_CHAR_REF, Companion::replaceCharRefs) + return true } - fun freeze(): Map = attrs.toImmutableMap() + fun freeze(): Map = attrs } private enum class State { @@ -236,16 +350,22 @@ object MetaTagsParser { SPACE, } - private fun parseAttrs(input: String): Map? { + /** Parses the attributes of a single tag, held in `input[from..? { val attrs = Attrs() var state = State.NAME - var nameBegin = 0 - var nameEnd = 0 - var valueBegin = 0 - var valueQuote: Char? = null + var nameBegin = from + var nameEnd = from + var valueBegin = from + var valueQuote = NO_QUOTE - input.forEachIndexed { i, c -> + for (i in from.. { when { @@ -259,7 +379,7 @@ object MetaTagsParser { state = State.BEFORE_EQ } - NON_ATTR_NAME_CHARS.contains(c) || c.isISOControl() || !c.isDefined() -> { + isNonAttrNameChar(c) || c.isISOControl() || !c.isDefined() -> { return null } } @@ -275,7 +395,7 @@ object MetaTagsParser { else -> { // if it is expecting = but gets another name, starts another property - runCatching { attrs.add(Pair(input.slice(nameBegin.. { valueBegin = i - valueQuote = null + valueQuote = NO_QUOTE state = State.VALUE } } } State.VALUE -> { - var attr: Pair? = null - if (valueQuote != null) { - if (c == valueQuote) { - attr = - Pair( - input.slice(nameBegin.. { - attr = - Pair( - input.slice(nameBegin.. valueEnd = i - i == input.length - 1 -> { - attr = - Pair( - input.slice(nameBegin.. valueEnd = i + 1 - NON_UNQUOTED_ATTR_VALUE_CHARS.contains(c) -> { - return null - } + isNonUnquotedAttrValueChar(c) -> return null } } - if (attr != null) { - runCatching { attrs.add(attr) }.getOrNull() ?: return null + if (valueEnd >= 0) { + val added = + attrs.add( + input.substring(nameBegin, nameEnd), + input.substring(valueBegin, valueEnd), + ) + if (!added) return null state = State.SPACE } } diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt new file mode 100644 index 0000000000..a8ca36a12f --- /dev/null +++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/HtmlParserCharsetTest.kt @@ -0,0 +1,120 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.commons.preview + +import kotlinx.coroutines.test.runTest +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * The charset half of the meta scan: `` and `` are + * the tags that decide how the rest of the document -- including every og: value -- is decoded. + * Get this wrong and a preview renders mojibake rather than nothing, so it fails quietly. + */ +class HtmlParserCharsetTest { + private suspend fun firstContentOf( + bytes: ByteArray, + charsetName: String?, + ): String = + HtmlParser() + .parseHtml(bytes, charsetName) + .last() + .attr("content") + + // -- HtmlCharsetParser -------------------------------------------------------------------- + + @Test + fun sniffsTheCharsetAttribute() { + assertEquals("iso-8859-1", HtmlCharsetParser.detectCharset("""""".encodeToByteArray())) + } + + @Test + fun sniffsTheHttpEquivContentType() { + val html = """""" + + assertEquals("shift_jis", HtmlCharsetParser.detectCharset(html.encodeToByteArray())) + } + + @Test + fun defaultsToUtf8WhenNothingIsDeclared() { + assertEquals("UTF-8", HtmlCharsetParser.detectCharset("""x""".encodeToByteArray())) + } + + @Test + fun onlySniffsTheFirstKilobyte() { + // The window is deliberate -- the declaration is required to be early -- but it means a + // charset pushed past 1 KB by a banner comment is not found, and UTF-8 is assumed. + val pushedOut = "" + "" + """""" + + assertEquals("UTF-8", HtmlCharsetParser.detectCharset(pushedOut.encodeToByteArray())) + } + + @Test + fun aCommentedOutCharsetIsNotSniffed() { + val html = """""" + + assertEquals("utf-8", HtmlCharsetParser.detectCharset(html.encodeToByteArray())) + } + + // -- HtmlParser: which charset wins -------------------------------------------------------- + + @Test + fun anExplicitCharsetWinsOverTheDocumentDeclaration() = + runTest { + // Content-Type said windows-1252; the document claims utf-8 and must not be believed. + val bytes = + byteArrayOf(0x3C) + // '<' + """head>""".encodeToByteArray() + + assertEquals("café", firstContentOf(bytes, "windows-1252")) + } + + @Test + fun aByteOrderMarkWinsOverTheDocumentDeclaration() = + runTest { + val utf16 = """""" + val bytes = byteArrayOf(0xFE.toByte(), 0xFF.toByte()) + utf16.encodeToUtf16Be() + + assertEquals("café", firstContentOf(bytes, null)) + } + + @Test + fun fallsBackToTheDocumentDeclarationWhenTheResponseHasNoCharset() = + runTest { + val bytes = + """""".encodeToByteArray() + + assertEquals("café", firstContentOf(bytes, null)) + } + + private fun String.encodeToUtf16Be(): ByteArray { + val out = ByteArray(length * 2) + forEachIndexed { i, c -> + out[i * 2] = (c.code shr 8).toByte() + out[i * 2 + 1] = (c.code and 0xFF).toByte() + } + return out + } +} diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt new file mode 100644 index 0000000000..71f7189252 --- /dev/null +++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserCommentTest.kt @@ -0,0 +1,238 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.commons.preview + +import kotlin.test.Test +import kotlin.test.assertEquals + +/** + * Comments, declarations and the text-only elements (script, style, title, textarea) are not + * element markup. Scanning them for + * attribute quotes lets an odd apostrophe -- "we don't" is enough -- leave the scanner + * inside a phantom quoted value, swallowing every tag up to the next quote character. + * That is what hid the entire og: block of https://brainstorm.world from link previews. + */ +class MetaTagsParserCommentTest { + @Test + fun commentWithUnbalancedApostropheDoesNotSwallowFollowingMetaTags() { + val input = + """ + | + | + | + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(2, metaTags.size) + assertEquals("Brainstorm", metaTags[0].attr("content")) + assertEquals("https://example.com/og-image.png", metaTags[1].attr("content")) + } + + @Test + fun metaTagsInsideCommentsAreNotParsed() { + val input = + """ + | + | + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun scriptBodyDoesNotSwallowFollowingMetaTags() { + val input = + """ + | + | + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun styleBodyDoesNotSwallowFollowingMetaTags() { + val input = + """ + | + | + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun titleTextWithALessThanDoesNotSwallowFollowingMetaTags() { + // `5 < 6, that's math` is ordinary HTML: an unescaped `<` in title text, + // then an apostrophe. Scanned as markup, the `<` opens a phantom tag and the apostrophe + // opens a phantom attribute value that runs to the end of the document. + val input = + """ + | + | 5 < 6, that's math + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun metaTagsInsideTitleTextAreNotParsed() { + // Title content is character data, so this is a title that reads literally + // `a b`, not a second og:title. + val input = + """ + | + | a <meta property="og:title" content="Fake"> b + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun textareaTextDoesNotSwallowFollowingMetaTags() { + val input = + """ + | + | + | + | + """.trimMargin() + + val metaTags = MetaTagsParser.parse(input).toList() + + assertEquals(1, metaTags.size) + assertEquals("Real Title", metaTags[0].attr("content")) + } + + @Test + fun aSelfClosedScriptStillOpensRawText() { + // Deliberate, and what a browser does: `/` on a script start tag is ignored, so everything + // up to `` is script data. A page written this way shows nothing after it either. + // Pinned so that "fixing" it never turns a JS string into an og: tag. + val input = + """ + | + | + | Brainstorm - Web of Trust for Nostr + | + | + | + | + |
+ | + """.trimMargin() + + val info = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(input)) + + assertEquals("Brainstorm - Your Network. Your Rules.", info.title) + assertEquals("The decentralized Web of Trust layer for Nostr.", info.description) + assertEquals("https://brainstorm.nosfabrica.com/og-image.png", info.image) + } +} diff --git a/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserEdgeCaseTest.kt b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserEdgeCaseTest.kt new file mode 100644 index 0000000000..1ece50c95e --- /dev/null +++ b/commons/src/commonTest/kotlin/com/vitorpamplona/amethyst/commons/preview/MetaTagsParserEdgeCaseTest.kt @@ -0,0 +1,208 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.commons.preview + +import kotlin.test.Test +import kotlin.test.assertEquals +import kotlin.test.assertTrue + +/** + * Tag and attribute shapes a preview fetch can meet in the wild, and the ones a truncated or + * hostile response can produce. The server picks this input, so "it throws" and "it silently eats + * the rest of the head" both have to be ruled out for every shape here. + */ +class MetaTagsParserEdgeCaseTest { + private fun contents(html: String) = MetaTagsParser.parse(html).map { it.attr("content") }.toList() + + // -- the end of the scan ------------------------------------------------------------------ + + @Test + fun stopsAtAnUppercaseHeadEndTag() { + val metas = contents("""""") + + assertEquals(listOf("1"), metas) + } + + @Test + fun stopsAtAHeadEndTagWithTrailingSpace() { + val metas = contents("""""") + + assertEquals(listOf("1"), metas) + } + + @Test + fun aHeadEndTagInsideAnAttributeValueDoesNotEndTheScan() { + val metas = contents("""""") + + assertEquals(listOf("", "2"), metas) + } + + @Test + fun scansTheWholeDocumentWhenThereIsNoHeadEndTag() { + val metas = contents("""""") + + assertEquals(listOf("1", "2"), metas) + } + + // -- truncated responses ------------------------------------------------------------------ + + @Test + fun aBodyTruncatedRightAfterASlashDoesNotThrow() { + // The `/` of a `/>` as the very last byte: the self-closing check must not read past it. + val metas = contents("""<""")) + } + + // -- tag shapes --------------------------------------------------------------------------- + + @Test + fun readsAnUppercaseMetaTagAndUppercaseAttributeNames() { + val metas = contents("""""") + + assertEquals(listOf("Real"), metas) + } + + @Test + fun anEmptySelfClosedMetaIsSkippedWithoutDerailingTheScan() { + // `` has no separator before the `/`, so the name reads as `meta/` and the tag is + // dropped. It carries nothing anyway; what matters is that the next tag still parses. + val metas = contents("""""") + + assertEquals(listOf("Real"), metas) + } + + @Test + fun aSelfClosingSequenceInsideAQuotedValueDoesNotEndTheTag() { + val metas = + contents( + """""", + ) + + assertEquals(listOf("a /> b", "ok"), metas) + } + + @Test + fun readsMetaTagsInsideNoscript() { + // noscript content is markup for a parser that isn't running scripts, and a redirect meta + // hidden in there is exactly the kind a preview wants to see. + val metas = + contents( + """""", + ) + + assertEquals(listOf("0", "Real"), metas) + } + + @Test + fun skipsCdataSections() { + val metas = contents(""" y ]]>""") + + assertEquals(listOf("Real"), metas) + } + + // -- attribute shapes --------------------------------------------------------------------- + + @Test + fun keepsAttributesWhenTheTagEndsWithAValuelessOne() { + val metas = MetaTagsParser.parse("""""").toList() + + assertEquals(1, metas.size) + assertEquals("og:title", metas[0].attr("property")) + assertEquals("T", metas[0].attr("content")) + } + + @Test + fun keepsAValueThatSpansLines() { + val metas = contents("") + + assertEquals(listOf("line1\nline2"), metas) + } + + @Test + fun anUnknownAttributeIsIgnoredNotFatal() { + val metas = contents("""""") + + assertEquals(listOf("Real"), metas) + } + + // -- character references in values -------------------------------------------------------- + + @Test + fun keepsQueryStringAmpersandsIntact() { + // og:image URLs are full of `&`; only a real character reference may be resolved. + val metas = + contents( + """""", + ) + + assertEquals(listOf("https://x.com/i.png?w=1200&h=630&fit=crop&q=80"), metas) + } + + @Test + fun decodesCharacterReferencesOutsideTheBasicMultilingualPlane() { + val metas = contents("""""") + + assertEquals(listOf("😀 😀 hi"), metas) + } + + @Test + fun leavesAnUnknownCharacterReferenceAlone() { + val metas = contents("""""") + + assertEquals(listOf("AT&T ¬areference; &#xZZ;"), metas) + } + + // -- laziness ------------------------------------------------------------------------------- + + @Test + fun stopsReadingOnceTheConsumerStops() { + // OpenGraphParser bails as soon as it has title+description+image; the sequence must not + // have scanned the rest of the document by then. A tag after an unterminated comment is + // unreachable, so seeing the first one proves the scan was still lazy. + val html = + """ + | + | + | + + Example - A Site + + + + + + + + + + + +
+ + """.trimIndent() + + /** + * A news/CMS head: [tags] worth of meta+link noise, analytics scripts, JSON-LD and + * boilerplate comments, with the og: block near the end -- the worst realistic ordering. + */ + fun heavyHead(tags: Int): String { + val sb = StringBuilder(64 * 1024) + sb.append("\n\n\n") + sb.append("\n") + repeat(tags) { i -> + sb.append("\n") + sb.append("\n") + sb.append("\n") + sb.append("\n") + sb.append("\n") + } + sb.append("\n") + sb.append("\n") + sb.append("\n") + sb.append("\n") + sb.append("\n\n") + // Body the parser must never reach: it stops at . + repeat(tags * 40) { i -> sb.append("

Paragraph $i with markup and \"quotes\".

\n") } + sb.append("\n\n") + return sb.toString() + } + + /** Same content, but with no `` to stop at -- the scan runs over the whole document. */ + fun unterminatedHead(tags: Int): String = heavyHead(tags).replace("", "") + + fun bench( + label: String, + input: String, + reps: Int, + op: (String) -> Int, + ) { + repeat(maxOf(reps / 4, 2)) { op(input) } // warmup + val t0 = System.nanoTime() + var sink = 0 + repeat(reps) { sink += op(input) } + val ns = (System.nanoTime() - t0) / reps + val mbps = input.length.toDouble() / ns * 1000.0 // bytes/ns -> MB/s + println( + String.format( + "%-34s %9d B %9d ns/op %8.1f MB/s (hits=%d)", + label, + input.length, + ns, + mbps, + sink / reps, + ), + ) + } + } + + @Test + fun metaScans() { + val spa = spaHead() + val heavy = heavyHead(60) + val open = unterminatedHead(60) + + // correctness first: a benchmark that finds nothing measures nothing + val spaInfo = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(spa)) + assertEquals("Example — Your Network. Your Rules.", spaInfo.title) + assertEquals("https://example.com/og-image.png", spaInfo.image) + + val heavyInfo = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(heavy)) + assertEquals("The Headline", heavyInfo.title) + assertEquals("https://example.com/lead.jpg", heavyInfo.image) + + // the body after is never scanned + assertEquals(1 + 60 + 3, MetaTagsParser.parse(heavy).count()) + + // Note: the two "heavy head" rows report MB/s over the whole document, of which only the + // ~46 KB head is actually scanned -- the scan stops at . The last row is the same + // page with no , i.e. what a hostile server can force us to read end to end. + println("MetaTagsParser") + bench("spa head, all tags", spa, 50_000) { MetaTagsParser.parse(it).count() } + bench("spa head, og: extraction", spa, 50_000) { + OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(it)).title.length + } + bench("heavy head, to ", heavy, 2_000) { MetaTagsParser.parse(it).count() } + bench("heavy head, og: extraction", heavy, 2_000) { + OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(it)).title.length + } + bench("no , whole doc", open, 2_000) { MetaTagsParser.parse(it).count() } + + assertTrue(MetaTagsParser.parse(open).count() >= 64) + } +}