mirror of
https://github.com/vitorpamplona/amethyst.git
synced 2026-10-06 03:38:23 +00:00
perf: drop the regex and the per-tag allocations from the meta scan
The tag-name check was a Regex match over a freshly cut substring, run for
every `<` in the document. Both are gone: names are compared in place against
the only four that matter (meta, head, script, style), ASCII-case-folded with
`code or 0x20`, so a non-meta tag now costs zero allocations -- no substring,
no Matcher, no RawTag. nextTag() reports a TagKind and leaves the attribute
span as two indices; only a real `<meta>` gets read, and parseAttrs() reads
that span straight out of the document instead of a copy of it.
The rest of the scan got the same treatment:
- `indexOf('<')` / `indexOf('>')` / `indexOf("-->")` instead of char-at-a-time
predicate loops -- these are intrinsified and vectorized on the JVM.
- `Set<Char>.contains` for the attribute character classes boxed a Char per
character of every meta tag; they are `when` branches now.
- one Pair, one Result and one lambda per attribute (`runCatching { add(Pair) }`)
became a boolean-returning add -- a duplicate attribute no longer throws.
- `toImmutableMap()` rebuilt a persistent map for every meta tag; the Attrs
builder is discarded at freeze(), so its own map is already private.
- the character-reference Regex only runs on values that contain an `&`.
Measured on a comment-free head, where this and the previous implementation
do identical work (same JVM, both warmed, `plainHead` corpus):
1.1 KB head, 10 metas 12.5 us -> 5.1 us ( 88 -> 215 MB/s)
28 KB head, 204 metas 267 us -> 116 us (105 -> 242 MB/s)
MetaTagsParserBenchmark joins the prodbench suite as the guard, on corpora
shaped like a Vite SPA head and a CMS head buried in analytics scripts: any
site we preview picks the input, so the scan has to stay linear in it.
Co-Authored-By: Claude <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01FumxeDJPEgPqX8mz3xkM6b
This commit is contained in:
+175
-127
@@ -21,7 +21,6 @@
|
||||
package com.vitorpamplona.amethyst.commons.preview
|
||||
|
||||
import com.vitorpamplona.amethyst.commons.util.codePointToChars
|
||||
import kotlinx.collections.immutable.toImmutableMap
|
||||
|
||||
data class MetaTag(
|
||||
private val attrs: Map<String, String>,
|
||||
@@ -32,10 +31,37 @@ data class MetaTag(
|
||||
fun attr(name: String): String = attrs[name.lowercase()] ?: ""
|
||||
}
|
||||
|
||||
/**
|
||||
* Extracts `<meta>` tags out of a (possibly partial) HTML document.
|
||||
*
|
||||
* This runs on every link preview, over bytes straight off the network, so the scan touches each
|
||||
* character once and allocates nothing until an actual `<meta>` shows up: `<` is found with
|
||||
* [String.indexOf], tag names are compared in place against the four names that matter, and only a
|
||||
* meta tag's attribute span is ever handed to [parseAttrs].
|
||||
*/
|
||||
object MetaTagsParser {
|
||||
private val TAG_NAME = Regex("""[0-9a-zA-Z]+""")
|
||||
private val NON_ATTR_NAME_CHARS = setOf(Char(0x0), '"', '\'', '>', '/')
|
||||
private val NON_UNQUOTED_ATTR_VALUE_CHARS = setOf('"', '\'', '=', '>', '<', '`')
|
||||
private const val NO_QUOTE = ' '
|
||||
|
||||
private const val META = "meta"
|
||||
private const val HEAD = "head"
|
||||
private const val SCRIPT = "script"
|
||||
private const val STYLE = "style"
|
||||
|
||||
private const val SCRIPT_END = "</script"
|
||||
private const val STYLE_END = "</style"
|
||||
private const val COMMENT_START = "!--"
|
||||
private const val COMMENT_END = "-->"
|
||||
|
||||
private enum class TagKind {
|
||||
/** A `<meta …>` start tag: its attribute span is [TagScanner.attrsStart]..<[TagScanner.attrsEnd]. */
|
||||
META,
|
||||
|
||||
/** The `</head>` that ends the interesting part of the document. */
|
||||
HEAD_END,
|
||||
|
||||
/** Anything else: other elements, comments, declarations, unparseable markup. */
|
||||
OTHER,
|
||||
}
|
||||
|
||||
/**
|
||||
* Lazily parse a partial HTML document and extract meta tags.
|
||||
@@ -44,141 +70,167 @@ object MetaTagsParser {
|
||||
sequence {
|
||||
val s = TagScanner(input)
|
||||
while (!s.exhausted()) {
|
||||
val t = s.nextTag() ?: continue
|
||||
if (t.name == "head" && t.isEnd) {
|
||||
break
|
||||
}
|
||||
if (t.name == "meta") {
|
||||
val attrs = parseAttrs(t.attrPart) ?: continue
|
||||
val kind = s.nextTag()
|
||||
if (kind == TagKind.HEAD_END) break
|
||||
if (kind == TagKind.META) {
|
||||
val attrs = parseAttrs(input, s.attrsStart, s.attrsEnd) ?: continue
|
||||
yield(MetaTag(attrs))
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private data class RawTag(
|
||||
val isEnd: Boolean,
|
||||
val name: String,
|
||||
val attrPart: String,
|
||||
)
|
||||
|
||||
private class TagScanner(
|
||||
private val input: String,
|
||||
) {
|
||||
private val length = input.length
|
||||
private var p = 0
|
||||
|
||||
fun exhausted(): Boolean = p >= input.length
|
||||
/** Attribute span of the tag [nextTag] last reported as [TagKind.META]. */
|
||||
var attrsStart = 0
|
||||
private set
|
||||
var attrsEnd = 0
|
||||
private set
|
||||
|
||||
private fun peek(): Char = input[p]
|
||||
fun exhausted(): Boolean = p >= length
|
||||
|
||||
private fun consume(): Char = input[p++]
|
||||
|
||||
private fun skipWhile(pred: (Char) -> Boolean) {
|
||||
while (!this.exhausted() && pred(this.peek())) {
|
||||
this.consume()
|
||||
/**
|
||||
* True when `input[from..<to]` equals [lower], ASCII-case-insensitively. [lower] must hold
|
||||
* only lowercase ASCII letters: `code or 0x20` lands on such a letter only when the input
|
||||
* char is that same letter in either case, so the fold cannot produce a false positive.
|
||||
*/
|
||||
private fun nameIs(
|
||||
from: Int,
|
||||
to: Int,
|
||||
lower: String,
|
||||
): Boolean {
|
||||
if (to - from != lower.length) return false
|
||||
for (i in lower.indices) {
|
||||
val c = input[from + i]
|
||||
if (c != lower[i] && (c.code or 0x20).toChar() != lower[i]) return false
|
||||
}
|
||||
}
|
||||
|
||||
private fun skipSpaces() {
|
||||
this.skipWhile { it.isWhitespace() }
|
||||
return true
|
||||
}
|
||||
|
||||
private fun skipComment() {
|
||||
val end = input.indexOf("-->", p)
|
||||
p = if (end < 0) input.length else end + 3
|
||||
val end = input.indexOf(COMMENT_END, p)
|
||||
p = if (end < 0) length else end + COMMENT_END.length
|
||||
}
|
||||
|
||||
private fun skipToTagEnd() {
|
||||
skipWhile { it != '>' }
|
||||
if (!exhausted()) consume()
|
||||
val end = input.indexOf('>', p)
|
||||
p = if (end < 0) length else end + 1
|
||||
}
|
||||
|
||||
/** Leaves [p] on the `</name` that closes a raw-text element, or at the end of the input. */
|
||||
private fun skipRawText(name: String) {
|
||||
val end = input.indexOf("</$name", p, ignoreCase = true)
|
||||
p = if (end < 0) input.length else end
|
||||
private fun skipRawText(endTag: String) {
|
||||
var i = p
|
||||
while (true) {
|
||||
i = input.indexOf('<', i)
|
||||
if (i < 0) {
|
||||
p = length
|
||||
return
|
||||
}
|
||||
if (input.regionMatches(i, endTag, 0, endTag.length, ignoreCase = true)) {
|
||||
p = i
|
||||
return
|
||||
}
|
||||
i++
|
||||
}
|
||||
}
|
||||
|
||||
fun nextTag(): RawTag? {
|
||||
skipWhile { it != '<' }
|
||||
if (this.exhausted()) return null
|
||||
consume()
|
||||
if (this.exhausted()) return null
|
||||
fun nextTag(): TagKind {
|
||||
val lt = input.indexOf('<', p)
|
||||
if (lt < 0) {
|
||||
p = length
|
||||
return TagKind.OTHER
|
||||
}
|
||||
p = lt + 1
|
||||
if (p >= length) return TagKind.OTHER
|
||||
|
||||
// `<!-- ... -->`, `<!DOCTYPE ...>` and `<?...>` are not element markup, so the
|
||||
// `<!-- ... -->`, `<!DOCTYPE ...>` and `<?...?>` are not element markup, so the
|
||||
// attribute-quote tracking below must not run over them. A comment holding an odd
|
||||
// number of quote characters -- an apostrophe in "we don't", a lone `"` -- would
|
||||
// otherwise leave the scanner inside a phantom quoted attribute value and make it
|
||||
// swallow every tag that follows, until the next matching quote character. That is
|
||||
// enough to hide a page's whole `<meta property="og:*">` block from the preview.
|
||||
if (peek() == '!' || peek() == '?') {
|
||||
if (input.startsWith("!--", p)) {
|
||||
val first = input[p]
|
||||
if (first == '!' || first == '?') {
|
||||
if (input.startsWith(COMMENT_START, p)) {
|
||||
skipComment()
|
||||
} else {
|
||||
skipToTagEnd()
|
||||
}
|
||||
return null
|
||||
return TagKind.OTHER
|
||||
}
|
||||
|
||||
// read tag name
|
||||
val isEnd = peek() == '/'
|
||||
if (isEnd) {
|
||||
consume()
|
||||
}
|
||||
// read the tag name
|
||||
val isEnd = first == '/'
|
||||
if (isEnd) p++
|
||||
val nameStart = p
|
||||
skipWhile { !it.isWhitespace() && it != '>' }
|
||||
while (p < length && !input[p].isWhitespace() && input[p] != '>') p++
|
||||
val nameEnd = p
|
||||
|
||||
// seek to start of attrs part
|
||||
skipSpaces()
|
||||
val attrsStart = p
|
||||
// seek to the start of the attrs part
|
||||
while (p < length && input[p].isWhitespace()) p++
|
||||
attrsStart = p
|
||||
|
||||
// skip until end of tag
|
||||
var quote: Char? = null
|
||||
while (!exhausted()) {
|
||||
val c = consume()
|
||||
when {
|
||||
// `/>` out of quote -> end of tag
|
||||
quote == null && c == '/' && !exhausted() && peek() == '>' -> {
|
||||
consume()
|
||||
// skip to the end of the tag, tracking quoted values so a `>` inside one doesn't end it
|
||||
var i = p
|
||||
var quote = NO_QUOTE
|
||||
while (i < length) {
|
||||
val c = input[i]
|
||||
if (quote == NO_QUOTE) {
|
||||
// `>` or `/>` out of quote -> end of tag
|
||||
if (c == '>') {
|
||||
i++
|
||||
break
|
||||
}
|
||||
|
||||
// `>` out of quote -> end of tag
|
||||
quote == null && c == '>' -> {
|
||||
if (c == '/' && i + 1 < length && input[i + 1] == '>') {
|
||||
i += 2
|
||||
break
|
||||
}
|
||||
|
||||
// entering quote
|
||||
quote == null && (c == '\'' || c == '"') -> {
|
||||
quote = c
|
||||
}
|
||||
|
||||
// leaving quote
|
||||
quote != null && c == quote -> {
|
||||
quote = null
|
||||
}
|
||||
if (c == '"' || c == '\'') quote = c
|
||||
} else if (c == quote) {
|
||||
quote = NO_QUOTE
|
||||
}
|
||||
i++
|
||||
}
|
||||
val attrsEnd = p - 1
|
||||
p = i
|
||||
attrsEnd = i - 1
|
||||
|
||||
val name = input.slice(nameStart..<nameEnd)
|
||||
if (!name.matches(TAG_NAME)) {
|
||||
return null
|
||||
if (isEnd) {
|
||||
return if (nameIs(nameStart, nameEnd, HEAD)) TagKind.HEAD_END else TagKind.OTHER
|
||||
}
|
||||
val lowercaseName = name.lowercase()
|
||||
|
||||
if (nameIs(nameStart, nameEnd, META)) return TagKind.META
|
||||
|
||||
// Script and style bodies are raw text: a `<` in `for (i = 0; i < n; i++)` or a quote
|
||||
// in a JS string is not markup and must not be scanned as such, for the same reason
|
||||
// comments can't be.
|
||||
if (!isEnd && (lowercaseName == "script" || lowercaseName == "style")) {
|
||||
skipRawText(lowercaseName)
|
||||
if (nameIs(nameStart, nameEnd, SCRIPT)) {
|
||||
skipRawText(SCRIPT_END)
|
||||
} else if (nameIs(nameStart, nameEnd, STYLE)) {
|
||||
skipRawText(STYLE_END)
|
||||
}
|
||||
|
||||
val attrsPart = input.slice(attrsStart..<attrsEnd)
|
||||
return RawTag(isEnd, lowercaseName, attrsPart)
|
||||
return TagKind.OTHER
|
||||
}
|
||||
}
|
||||
|
||||
// These two are `when` branches rather than a `Set<Char>` because `Set<Char>.contains` boxes
|
||||
// the char, once per attribute character of every meta tag.
|
||||
private fun isNonAttrNameChar(c: Char): Boolean =
|
||||
when (c) {
|
||||
'\u0000', '"', '\'', '>', '/' -> true
|
||||
else -> false
|
||||
}
|
||||
|
||||
private fun isNonUnquotedAttrValueChar(c: Char): Boolean =
|
||||
when (c) {
|
||||
'"', '\'', '=', '>', '<', '`' -> true
|
||||
else -> false
|
||||
}
|
||||
|
||||
// map of HTML element attribute name to its value, with additional logics:
|
||||
// - attribute names are matched in a case-insensitive manner
|
||||
// - attribute names never duplicate
|
||||
@@ -258,16 +310,20 @@ object MetaTagsParser {
|
||||
|
||||
private val attrs = mutableMapOf<String, String>()
|
||||
|
||||
fun add(attr: Pair<String, String>) {
|
||||
val name = attr.first.lowercase()
|
||||
if (attrs.containsKey(name)) {
|
||||
throw IllegalArgumentException("duplicated attribute name: $name")
|
||||
}
|
||||
val value = attr.second.replace(RE_CHAR_REF, Companion::replaceCharRefs)
|
||||
attrs += Pair(name, value)
|
||||
/** Adds an attribute, returning false if that name was already set (the first value wins). */
|
||||
fun add(
|
||||
name: String,
|
||||
value: String,
|
||||
): Boolean {
|
||||
val key = name.lowercase()
|
||||
if (attrs.containsKey(key)) return false
|
||||
// Resolving character references is the expensive half of an attribute and almost no
|
||||
// value has an `&` in it, so the scan for one pays for itself.
|
||||
attrs[key] = if (value.indexOf('&') < 0) value else value.replace(RE_CHAR_REF, Companion::replaceCharRefs)
|
||||
return true
|
||||
}
|
||||
|
||||
fun freeze(): Map<String, String> = attrs.toImmutableMap()
|
||||
fun freeze(): Map<String, String> = attrs
|
||||
}
|
||||
|
||||
private enum class State {
|
||||
@@ -278,16 +334,22 @@ object MetaTagsParser {
|
||||
SPACE,
|
||||
}
|
||||
|
||||
private fun parseAttrs(input: String): Map<String, String>? {
|
||||
/** Parses the attributes of a single tag, held in `input[from..<to]`. */
|
||||
private fun parseAttrs(
|
||||
input: String,
|
||||
from: Int,
|
||||
to: Int,
|
||||
): Map<String, String>? {
|
||||
val attrs = Attrs()
|
||||
|
||||
var state = State.NAME
|
||||
var nameBegin = 0
|
||||
var nameEnd = 0
|
||||
var valueBegin = 0
|
||||
var valueQuote: Char? = null
|
||||
var nameBegin = from
|
||||
var nameEnd = from
|
||||
var valueBegin = from
|
||||
var valueQuote = NO_QUOTE
|
||||
|
||||
input.forEachIndexed { i, c ->
|
||||
for (i in from..<to) {
|
||||
val c = input[i]
|
||||
when (state) {
|
||||
State.NAME -> {
|
||||
when {
|
||||
@@ -301,7 +363,7 @@ object MetaTagsParser {
|
||||
state = State.BEFORE_EQ
|
||||
}
|
||||
|
||||
NON_ATTR_NAME_CHARS.contains(c) || c.isISOControl() || !c.isDefined() -> {
|
||||
isNonAttrNameChar(c) || c.isISOControl() || !c.isDefined() -> {
|
||||
return null
|
||||
}
|
||||
}
|
||||
@@ -317,7 +379,7 @@ object MetaTagsParser {
|
||||
|
||||
else -> {
|
||||
// if it is expecting = but gets another name, starts another property
|
||||
runCatching { attrs.add(Pair(input.slice(nameBegin..<nameEnd), "")) }
|
||||
attrs.add(input.substring(nameBegin, nameEnd), "")
|
||||
|
||||
nameBegin = i
|
||||
state = State.NAME
|
||||
@@ -337,47 +399,33 @@ object MetaTagsParser {
|
||||
|
||||
else -> {
|
||||
valueBegin = i
|
||||
valueQuote = null
|
||||
valueQuote = NO_QUOTE
|
||||
state = State.VALUE
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
State.VALUE -> {
|
||||
var attr: Pair<String, String>? = null
|
||||
if (valueQuote != null) {
|
||||
if (c == valueQuote) {
|
||||
attr =
|
||||
Pair(
|
||||
input.slice(nameBegin..<nameEnd),
|
||||
input.slice(valueBegin..<i),
|
||||
)
|
||||
}
|
||||
// -1 while the value is still running; otherwise the index one past its last char
|
||||
var valueEnd = -1
|
||||
if (valueQuote != NO_QUOTE) {
|
||||
if (c == valueQuote) valueEnd = i
|
||||
} else {
|
||||
when {
|
||||
c.isWhitespace() -> {
|
||||
attr =
|
||||
Pair(
|
||||
input.slice(nameBegin..<nameEnd),
|
||||
input.slice(valueBegin..<i),
|
||||
)
|
||||
}
|
||||
c.isWhitespace() -> valueEnd = i
|
||||
|
||||
i == input.length - 1 -> {
|
||||
attr =
|
||||
Pair(
|
||||
input.slice(nameBegin..<nameEnd),
|
||||
input.slice(valueBegin..i),
|
||||
)
|
||||
}
|
||||
i == to - 1 -> valueEnd = i + 1
|
||||
|
||||
NON_UNQUOTED_ATTR_VALUE_CHARS.contains(c) -> {
|
||||
return null
|
||||
}
|
||||
isNonUnquotedAttrValueChar(c) -> return null
|
||||
}
|
||||
}
|
||||
if (attr != null) {
|
||||
runCatching { attrs.add(attr) }.getOrNull() ?: return null
|
||||
if (valueEnd >= 0) {
|
||||
val added =
|
||||
attrs.add(
|
||||
input.substring(nameBegin, nameEnd),
|
||||
input.substring(valueBegin, valueEnd),
|
||||
)
|
||||
if (!added) return null
|
||||
state = State.SPACE
|
||||
}
|
||||
}
|
||||
|
||||
+165
@@ -0,0 +1,165 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.amethyst.commons.prodbench
|
||||
|
||||
import com.vitorpamplona.amethyst.commons.preview.MetaTagsParser
|
||||
import com.vitorpamplona.amethyst.commons.preview.OpenGraphParser
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
import kotlin.test.assertTrue
|
||||
|
||||
/**
|
||||
* Measures the `<meta>` scan behind every link preview.
|
||||
*
|
||||
* Every URL in a rendered note can reach [MetaTagsParser] with a whole HTML document in hand, so
|
||||
* the scan runs on user-visible paths with attacker-shaped input (any site can serve a 1 MB head).
|
||||
* The numbers below are the guard against that: the parser must stay linear and allocation-light,
|
||||
* scanning at hundreds of MB/s rather than degrading with page size.
|
||||
*
|
||||
* Deterministic and offline. Prints ns/op and MB/s; the assertions only check that the scan still
|
||||
* finds the right tags, never wall time (CI machines vary).
|
||||
*/
|
||||
class MetaTagsParserBenchmark {
|
||||
companion object {
|
||||
/** The shape of a Vite/React SPA head: theme comment, inline script, then the og: block. */
|
||||
fun spaHead(): String =
|
||||
"""
|
||||
<!DOCTYPE html>
|
||||
<html lang="en">
|
||||
<head>
|
||||
<meta charset="UTF-8" />
|
||||
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
||||
<!-- No-flash theme: set class="dark" synchronously before first paint.
|
||||
Fresh visitors (no stored choice) stay LIGHT -- we don't auto-dark a
|
||||
dark-OS visitor until dark mode is fully reviewed. -->
|
||||
<script>
|
||||
(function () {
|
||||
var t = localStorage.getItem("app_theme");
|
||||
for (var i = 0; i < 2; i++) {
|
||||
if (t === "dark") document.documentElement.classList.add("dark");
|
||||
}
|
||||
})();
|
||||
</script>
|
||||
<title>Example - A Site</title>
|
||||
<meta name="description" content="A description that is long enough to look real." />
|
||||
<meta property="og:title" content="Example — Your Network. Your Rules." />
|
||||
<meta property="og:description" content="The decentralized layer for everything." />
|
||||
<meta property="og:image" content="https://example.com/og-image.png" />
|
||||
<meta property="og:image:width" content="1200" />
|
||||
<meta property="og:image:height" content="630" />
|
||||
<meta name="twitter:card" content="summary_large_image" />
|
||||
<link rel="icon" type="image/svg+xml" href="/favicon.svg" />
|
||||
<link rel="manifest" href="/site.webmanifest" />
|
||||
<link rel="stylesheet" crossorigin href="/assets/index-Cxc5egnN.css" />
|
||||
</head>
|
||||
<body><div id="root"></div></body>
|
||||
</html>
|
||||
""".trimIndent()
|
||||
|
||||
/**
|
||||
* A news/CMS head: [tags] worth of meta+link noise, analytics scripts, JSON-LD and
|
||||
* boilerplate comments, with the og: block near the end -- the worst realistic ordering.
|
||||
*/
|
||||
fun heavyHead(tags: Int): String {
|
||||
val sb = StringBuilder(64 * 1024)
|
||||
sb.append("<!DOCTYPE html>\n<html lang=\"en\">\n<head>\n")
|
||||
sb.append("<meta charset=\"utf-8\">\n")
|
||||
repeat(tags) { i ->
|
||||
sb.append("<!-- section $i: don't reorder, the CMS won't regenerate it -->\n")
|
||||
sb.append("<meta name=\"cms:field-$i\" content=\"value $i for a field nobody reads\">\n")
|
||||
sb.append("<link rel=\"preload\" as=\"font\" href=\"/fonts/f$i.woff2\" crossorigin>\n")
|
||||
sb.append("<script>window.__cfg$i = { id: $i, path: \"/a/b/c\", t: 0 < 1 };</script>\n")
|
||||
sb.append("<style>.cls-$i { font-family: 'Figtree', sans-serif; }</style>\n")
|
||||
}
|
||||
sb.append("<script type=\"application/ld+json\">{\"@type\":\"Article\",\"headline\":\"x < y\"}</script>\n")
|
||||
sb.append("<meta property=\"og:title\" content=\"The Headline\">\n")
|
||||
sb.append("<meta property=\"og:description\" content=\"The standfirst, in full.\">\n")
|
||||
sb.append("<meta property=\"og:image\" content=\"https://example.com/lead.jpg\">\n")
|
||||
sb.append("</head>\n<body>\n")
|
||||
// Body the parser must never reach: it stops at </head>.
|
||||
repeat(tags * 40) { i -> sb.append("<p class=\"para\">Paragraph $i with <em>markup</em> and \"quotes\".</p>\n") }
|
||||
sb.append("</body>\n</html>\n")
|
||||
return sb.toString()
|
||||
}
|
||||
|
||||
/** Same content, but with no `</head>` to stop at -- the scan runs over the whole document. */
|
||||
fun unterminatedHead(tags: Int): String = heavyHead(tags).replace("</head>", "")
|
||||
|
||||
fun bench(
|
||||
label: String,
|
||||
input: String,
|
||||
reps: Int,
|
||||
op: (String) -> Int,
|
||||
) {
|
||||
repeat(maxOf(reps / 4, 2)) { op(input) } // warmup
|
||||
val t0 = System.nanoTime()
|
||||
var sink = 0
|
||||
repeat(reps) { sink += op(input) }
|
||||
val ns = (System.nanoTime() - t0) / reps
|
||||
val mbps = input.length.toDouble() / ns * 1000.0 // bytes/ns -> MB/s
|
||||
println(
|
||||
String.format(
|
||||
"%-34s %9d B %9d ns/op %8.1f MB/s (hits=%d)",
|
||||
label,
|
||||
input.length,
|
||||
ns,
|
||||
mbps,
|
||||
sink / reps,
|
||||
),
|
||||
)
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
fun metaScans() {
|
||||
val spa = spaHead()
|
||||
val heavy = heavyHead(60)
|
||||
val open = unterminatedHead(60)
|
||||
|
||||
// correctness first: a benchmark that finds nothing measures nothing
|
||||
val spaInfo = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(spa))
|
||||
assertEquals("Example — Your Network. Your Rules.", spaInfo.title)
|
||||
assertEquals("https://example.com/og-image.png", spaInfo.image)
|
||||
|
||||
val heavyInfo = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(heavy))
|
||||
assertEquals("The Headline", heavyInfo.title)
|
||||
assertEquals("https://example.com/lead.jpg", heavyInfo.image)
|
||||
|
||||
// the body after </head> is never scanned
|
||||
assertEquals(1 + 60 + 3, MetaTagsParser.parse(heavy).count())
|
||||
|
||||
// Note: the two "heavy head" rows report MB/s over the whole document, of which only the
|
||||
// ~46 KB head is actually scanned -- the scan stops at </head>. The last row is the same
|
||||
// page with no </head>, i.e. what a hostile server can force us to read end to end.
|
||||
println("MetaTagsParser")
|
||||
bench("spa head, all tags", spa, 50_000) { MetaTagsParser.parse(it).count() }
|
||||
bench("spa head, og: extraction", spa, 50_000) {
|
||||
OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(it)).title.length
|
||||
}
|
||||
bench("heavy head, to </head>", heavy, 2_000) { MetaTagsParser.parse(it).count() }
|
||||
bench("heavy head, og: extraction", heavy, 2_000) {
|
||||
OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(it)).title.length
|
||||
}
|
||||
bench("no </head>, whole doc", open, 2_000) { MetaTagsParser.parse(it).count() }
|
||||
|
||||
assertTrue(MetaTagsParser.parse(open).count() >= 64)
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user