perf: drop the regex and the per-tag allocations from the meta scan

The tag-name check was a Regex match over a freshly cut substring, run for
every `<` in the document. Both are gone: names are compared in place against
the only four that matter (meta, head, script, style), ASCII-case-folded with
`code or 0x20`, so a non-meta tag now costs zero allocations -- no substring,
no Matcher, no RawTag. nextTag() reports a TagKind and leaves the attribute
span as two indices; only a real `<meta>` gets read, and parseAttrs() reads
that span straight out of the document instead of a copy of it.

The rest of the scan got the same treatment:

- `indexOf('<')` / `indexOf('>')` / `indexOf("-->")` instead of char-at-a-time
  predicate loops -- these are intrinsified and vectorized on the JVM.
- `Set<Char>.contains` for the attribute character classes boxed a Char per
  character of every meta tag; they are `when` branches now.
- one Pair, one Result and one lambda per attribute (`runCatching { add(Pair) }`)
  became a boolean-returning add -- a duplicate attribute no longer throws.
- `toImmutableMap()` rebuilt a persistent map for every meta tag; the Attrs
  builder is discarded at freeze(), so its own map is already private.
- the character-reference Regex only runs on values that contain an `&`.

Measured on a comment-free head, where this and the previous implementation
do identical work (same JVM, both warmed, `plainHead` corpus):

  1.1 KB head,  10 metas   12.5 us -> 5.1 us   ( 88 -> 215 MB/s)
   28 KB head, 204 metas    267 us -> 116 us   (105 -> 242 MB/s)

MetaTagsParserBenchmark joins the prodbench suite as the guard, on corpora
shaped like a Vite SPA head and a CMS head buried in analytics scripts: any
site we preview picks the input, so the scan has to stay linear in it.

Co-Authored-By: Claude <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01FumxeDJPEgPqX8mz3xkM6b
This commit is contained in:
Claude
2026-08-21 19:58:08 +00:00
parent b3d4cd924b
commit 39797b2191
2 changed files with 340 additions and 127 deletions
@@ -21,7 +21,6 @@
package com.vitorpamplona.amethyst.commons.preview
import com.vitorpamplona.amethyst.commons.util.codePointToChars
import kotlinx.collections.immutable.toImmutableMap
data class MetaTag(
private val attrs: Map<String, String>,
@@ -32,10 +31,37 @@ data class MetaTag(
fun attr(name: String): String = attrs[name.lowercase()] ?: ""
}
/**
* Extracts `<meta>` tags out of a (possibly partial) HTML document.
*
* This runs on every link preview, over bytes straight off the network, so the scan touches each
* character once and allocates nothing until an actual `<meta>` shows up: `<` is found with
* [String.indexOf], tag names are compared in place against the four names that matter, and only a
* meta tag's attribute span is ever handed to [parseAttrs].
*/
object MetaTagsParser {
private val TAG_NAME = Regex("""[0-9a-zA-Z]+""")
private val NON_ATTR_NAME_CHARS = setOf(Char(0x0), '"', '\'', '>', '/')
private val NON_UNQUOTED_ATTR_VALUE_CHARS = setOf('"', '\'', '=', '>', '<', '`')
private const val NO_QUOTE = ' '
private const val META = "meta"
private const val HEAD = "head"
private const val SCRIPT = "script"
private const val STYLE = "style"
private const val SCRIPT_END = "</script"
private const val STYLE_END = "</style"
private const val COMMENT_START = "!--"
private const val COMMENT_END = "-->"
private enum class TagKind {
/** A `<meta …>` start tag: its attribute span is [TagScanner.attrsStart]..<[TagScanner.attrsEnd]. */
META,
/** The `</head>` that ends the interesting part of the document. */
HEAD_END,
/** Anything else: other elements, comments, declarations, unparseable markup. */
OTHER,
}
/**
* Lazily parse a partial HTML document and extract meta tags.
@@ -44,141 +70,167 @@ object MetaTagsParser {
sequence {
val s = TagScanner(input)
while (!s.exhausted()) {
val t = s.nextTag() ?: continue
if (t.name == "head" && t.isEnd) {
break
}
if (t.name == "meta") {
val attrs = parseAttrs(t.attrPart) ?: continue
val kind = s.nextTag()
if (kind == TagKind.HEAD_END) break
if (kind == TagKind.META) {
val attrs = parseAttrs(input, s.attrsStart, s.attrsEnd) ?: continue
yield(MetaTag(attrs))
}
}
}
private data class RawTag(
val isEnd: Boolean,
val name: String,
val attrPart: String,
)
private class TagScanner(
private val input: String,
) {
private val length = input.length
private var p = 0
fun exhausted(): Boolean = p >= input.length
/** Attribute span of the tag [nextTag] last reported as [TagKind.META]. */
var attrsStart = 0
private set
var attrsEnd = 0
private set
private fun peek(): Char = input[p]
fun exhausted(): Boolean = p >= length
private fun consume(): Char = input[p++]
private fun skipWhile(pred: (Char) -> Boolean) {
while (!this.exhausted() && pred(this.peek())) {
this.consume()
/**
* True when `input[from..<to]` equals [lower], ASCII-case-insensitively. [lower] must hold
* only lowercase ASCII letters: `code or 0x20` lands on such a letter only when the input
* char is that same letter in either case, so the fold cannot produce a false positive.
*/
private fun nameIs(
from: Int,
to: Int,
lower: String,
): Boolean {
if (to - from != lower.length) return false
for (i in lower.indices) {
val c = input[from + i]
if (c != lower[i] && (c.code or 0x20).toChar() != lower[i]) return false
}
}
private fun skipSpaces() {
this.skipWhile { it.isWhitespace() }
return true
}
private fun skipComment() {
val end = input.indexOf("-->", p)
p = if (end < 0) input.length else end + 3
val end = input.indexOf(COMMENT_END, p)
p = if (end < 0) length else end + COMMENT_END.length
}
private fun skipToTagEnd() {
skipWhile { it != '>' }
if (!exhausted()) consume()
val end = input.indexOf('>', p)
p = if (end < 0) length else end + 1
}
/** Leaves [p] on the `</name` that closes a raw-text element, or at the end of the input. */
private fun skipRawText(name: String) {
val end = input.indexOf("</$name", p, ignoreCase = true)
p = if (end < 0) input.length else end
private fun skipRawText(endTag: String) {
var i = p
while (true) {
i = input.indexOf('<', i)
if (i < 0) {
p = length
return
}
if (input.regionMatches(i, endTag, 0, endTag.length, ignoreCase = true)) {
p = i
return
}
i++
}
}
fun nextTag(): RawTag? {
skipWhile { it != '<' }
if (this.exhausted()) return null
consume()
if (this.exhausted()) return null
fun nextTag(): TagKind {
val lt = input.indexOf('<', p)
if (lt < 0) {
p = length
return TagKind.OTHER
}
p = lt + 1
if (p >= length) return TagKind.OTHER
// `<!-- ... -->`, `<!DOCTYPE ...>` and `<?...>` are not element markup, so the
// `<!-- ... -->`, `<!DOCTYPE ...>` and `<?...?>` are not element markup, so the
// attribute-quote tracking below must not run over them. A comment holding an odd
// number of quote characters -- an apostrophe in "we don't", a lone `"` -- would
// otherwise leave the scanner inside a phantom quoted attribute value and make it
// swallow every tag that follows, until the next matching quote character. That is
// enough to hide a page's whole `<meta property="og:*">` block from the preview.
if (peek() == '!' || peek() == '?') {
if (input.startsWith("!--", p)) {
val first = input[p]
if (first == '!' || first == '?') {
if (input.startsWith(COMMENT_START, p)) {
skipComment()
} else {
skipToTagEnd()
}
return null
return TagKind.OTHER
}
// read tag name
val isEnd = peek() == '/'
if (isEnd) {
consume()
}
// read the tag name
val isEnd = first == '/'
if (isEnd) p++
val nameStart = p
skipWhile { !it.isWhitespace() && it != '>' }
while (p < length && !input[p].isWhitespace() && input[p] != '>') p++
val nameEnd = p
// seek to start of attrs part
skipSpaces()
val attrsStart = p
// seek to the start of the attrs part
while (p < length && input[p].isWhitespace()) p++
attrsStart = p
// skip until end of tag
var quote: Char? = null
while (!exhausted()) {
val c = consume()
when {
// `/>` out of quote -> end of tag
quote == null && c == '/' && !exhausted() && peek() == '>' -> {
consume()
// skip to the end of the tag, tracking quoted values so a `>` inside one doesn't end it
var i = p
var quote = NO_QUOTE
while (i < length) {
val c = input[i]
if (quote == NO_QUOTE) {
// `>` or `/>` out of quote -> end of tag
if (c == '>') {
i++
break
}
// `>` out of quote -> end of tag
quote == null && c == '>' -> {
if (c == '/' && i + 1 < length && input[i + 1] == '>') {
i += 2
break
}
// entering quote
quote == null && (c == '\'' || c == '"') -> {
quote = c
}
// leaving quote
quote != null && c == quote -> {
quote = null
}
if (c == '"' || c == '\'') quote = c
} else if (c == quote) {
quote = NO_QUOTE
}
i++
}
val attrsEnd = p - 1
p = i
attrsEnd = i - 1
val name = input.slice(nameStart..<nameEnd)
if (!name.matches(TAG_NAME)) {
return null
if (isEnd) {
return if (nameIs(nameStart, nameEnd, HEAD)) TagKind.HEAD_END else TagKind.OTHER
}
val lowercaseName = name.lowercase()
if (nameIs(nameStart, nameEnd, META)) return TagKind.META
// Script and style bodies are raw text: a `<` in `for (i = 0; i < n; i++)` or a quote
// in a JS string is not markup and must not be scanned as such, for the same reason
// comments can't be.
if (!isEnd && (lowercaseName == "script" || lowercaseName == "style")) {
skipRawText(lowercaseName)
if (nameIs(nameStart, nameEnd, SCRIPT)) {
skipRawText(SCRIPT_END)
} else if (nameIs(nameStart, nameEnd, STYLE)) {
skipRawText(STYLE_END)
}
val attrsPart = input.slice(attrsStart..<attrsEnd)
return RawTag(isEnd, lowercaseName, attrsPart)
return TagKind.OTHER
}
}
// These two are `when` branches rather than a `Set<Char>` because `Set<Char>.contains` boxes
// the char, once per attribute character of every meta tag.
private fun isNonAttrNameChar(c: Char): Boolean =
when (c) {
'\u0000', '"', '\'', '>', '/' -> true
else -> false
}
private fun isNonUnquotedAttrValueChar(c: Char): Boolean =
when (c) {
'"', '\'', '=', '>', '<', '`' -> true
else -> false
}
// map of HTML element attribute name to its value, with additional logics:
// - attribute names are matched in a case-insensitive manner
// - attribute names never duplicate
@@ -258,16 +310,20 @@ object MetaTagsParser {
private val attrs = mutableMapOf<String, String>()
fun add(attr: Pair<String, String>) {
val name = attr.first.lowercase()
if (attrs.containsKey(name)) {
throw IllegalArgumentException("duplicated attribute name: $name")
}
val value = attr.second.replace(RE_CHAR_REF, Companion::replaceCharRefs)
attrs += Pair(name, value)
/** Adds an attribute, returning false if that name was already set (the first value wins). */
fun add(
name: String,
value: String,
): Boolean {
val key = name.lowercase()
if (attrs.containsKey(key)) return false
// Resolving character references is the expensive half of an attribute and almost no
// value has an `&` in it, so the scan for one pays for itself.
attrs[key] = if (value.indexOf('&') < 0) value else value.replace(RE_CHAR_REF, Companion::replaceCharRefs)
return true
}
fun freeze(): Map<String, String> = attrs.toImmutableMap()
fun freeze(): Map<String, String> = attrs
}
private enum class State {
@@ -278,16 +334,22 @@ object MetaTagsParser {
SPACE,
}
private fun parseAttrs(input: String): Map<String, String>? {
/** Parses the attributes of a single tag, held in `input[from..<to]`. */
private fun parseAttrs(
input: String,
from: Int,
to: Int,
): Map<String, String>? {
val attrs = Attrs()
var state = State.NAME
var nameBegin = 0
var nameEnd = 0
var valueBegin = 0
var valueQuote: Char? = null
var nameBegin = from
var nameEnd = from
var valueBegin = from
var valueQuote = NO_QUOTE
input.forEachIndexed { i, c ->
for (i in from..<to) {
val c = input[i]
when (state) {
State.NAME -> {
when {
@@ -301,7 +363,7 @@ object MetaTagsParser {
state = State.BEFORE_EQ
}
NON_ATTR_NAME_CHARS.contains(c) || c.isISOControl() || !c.isDefined() -> {
isNonAttrNameChar(c) || c.isISOControl() || !c.isDefined() -> {
return null
}
}
@@ -317,7 +379,7 @@ object MetaTagsParser {
else -> {
// if it is expecting = but gets another name, starts another property
runCatching { attrs.add(Pair(input.slice(nameBegin..<nameEnd), "")) }
attrs.add(input.substring(nameBegin, nameEnd), "")
nameBegin = i
state = State.NAME
@@ -337,47 +399,33 @@ object MetaTagsParser {
else -> {
valueBegin = i
valueQuote = null
valueQuote = NO_QUOTE
state = State.VALUE
}
}
}
State.VALUE -> {
var attr: Pair<String, String>? = null
if (valueQuote != null) {
if (c == valueQuote) {
attr =
Pair(
input.slice(nameBegin..<nameEnd),
input.slice(valueBegin..<i),
)
}
// -1 while the value is still running; otherwise the index one past its last char
var valueEnd = -1
if (valueQuote != NO_QUOTE) {
if (c == valueQuote) valueEnd = i
} else {
when {
c.isWhitespace() -> {
attr =
Pair(
input.slice(nameBegin..<nameEnd),
input.slice(valueBegin..<i),
)
}
c.isWhitespace() -> valueEnd = i
i == input.length - 1 -> {
attr =
Pair(
input.slice(nameBegin..<nameEnd),
input.slice(valueBegin..i),
)
}
i == to - 1 -> valueEnd = i + 1
NON_UNQUOTED_ATTR_VALUE_CHARS.contains(c) -> {
return null
}
isNonUnquotedAttrValueChar(c) -> return null
}
}
if (attr != null) {
runCatching { attrs.add(attr) }.getOrNull() ?: return null
if (valueEnd >= 0) {
val added =
attrs.add(
input.substring(nameBegin, nameEnd),
input.substring(valueBegin, valueEnd),
)
if (!added) return null
state = State.SPACE
}
}
@@ -0,0 +1,165 @@
/*
* Copyright (c) 2025 Vitor Pamplona
*
* Permission is hereby granted, free of charge, to any person obtaining a copy of
* this software and associated documentation files (the "Software"), to deal in
* the Software without restriction, including without limitation the rights to use,
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
* Software, and to permit persons to whom the Software is furnished to do so,
* subject to the following conditions:
*
* The above copyright notice and this permission notice shall be included in all
* copies or substantial portions of the Software.
*
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
*/
package com.vitorpamplona.amethyst.commons.prodbench
import com.vitorpamplona.amethyst.commons.preview.MetaTagsParser
import com.vitorpamplona.amethyst.commons.preview.OpenGraphParser
import kotlin.test.Test
import kotlin.test.assertEquals
import kotlin.test.assertTrue
/**
* Measures the `<meta>` scan behind every link preview.
*
* Every URL in a rendered note can reach [MetaTagsParser] with a whole HTML document in hand, so
* the scan runs on user-visible paths with attacker-shaped input (any site can serve a 1 MB head).
* The numbers below are the guard against that: the parser must stay linear and allocation-light,
* scanning at hundreds of MB/s rather than degrading with page size.
*
* Deterministic and offline. Prints ns/op and MB/s; the assertions only check that the scan still
* finds the right tags, never wall time (CI machines vary).
*/
class MetaTagsParserBenchmark {
companion object {
/** The shape of a Vite/React SPA head: theme comment, inline script, then the og: block. */
fun spaHead(): String =
"""
<!DOCTYPE html>
<html lang="en">
<head>
<meta charset="UTF-8" />
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
<!-- No-flash theme: set class="dark" synchronously before first paint.
Fresh visitors (no stored choice) stay LIGHT -- we don't auto-dark a
dark-OS visitor until dark mode is fully reviewed. -->
<script>
(function () {
var t = localStorage.getItem("app_theme");
for (var i = 0; i < 2; i++) {
if (t === "dark") document.documentElement.classList.add("dark");
}
})();
</script>
<title>Example - A Site</title>
<meta name="description" content="A description that is long enough to look real." />
<meta property="og:title" content="Example &mdash; Your Network. Your Rules." />
<meta property="og:description" content="The decentralized layer for everything." />
<meta property="og:image" content="https://example.com/og-image.png" />
<meta property="og:image:width" content="1200" />
<meta property="og:image:height" content="630" />
<meta name="twitter:card" content="summary_large_image" />
<link rel="icon" type="image/svg+xml" href="/favicon.svg" />
<link rel="manifest" href="/site.webmanifest" />
<link rel="stylesheet" crossorigin href="/assets/index-Cxc5egnN.css" />
</head>
<body><div id="root"></div></body>
</html>
""".trimIndent()
/**
* A news/CMS head: [tags] worth of meta+link noise, analytics scripts, JSON-LD and
* boilerplate comments, with the og: block near the end -- the worst realistic ordering.
*/
fun heavyHead(tags: Int): String {
val sb = StringBuilder(64 * 1024)
sb.append("<!DOCTYPE html>\n<html lang=\"en\">\n<head>\n")
sb.append("<meta charset=\"utf-8\">\n")
repeat(tags) { i ->
sb.append("<!-- section $i: don't reorder, the CMS won't regenerate it -->\n")
sb.append("<meta name=\"cms:field-$i\" content=\"value $i for a field nobody reads\">\n")
sb.append("<link rel=\"preload\" as=\"font\" href=\"/fonts/f$i.woff2\" crossorigin>\n")
sb.append("<script>window.__cfg$i = { id: $i, path: \"/a/b/c\", t: 0 < 1 };</script>\n")
sb.append("<style>.cls-$i { font-family: 'Figtree', sans-serif; }</style>\n")
}
sb.append("<script type=\"application/ld+json\">{\"@type\":\"Article\",\"headline\":\"x < y\"}</script>\n")
sb.append("<meta property=\"og:title\" content=\"The Headline\">\n")
sb.append("<meta property=\"og:description\" content=\"The standfirst, in full.\">\n")
sb.append("<meta property=\"og:image\" content=\"https://example.com/lead.jpg\">\n")
sb.append("</head>\n<body>\n")
// Body the parser must never reach: it stops at </head>.
repeat(tags * 40) { i -> sb.append("<p class=\"para\">Paragraph $i with <em>markup</em> and \"quotes\".</p>\n") }
sb.append("</body>\n</html>\n")
return sb.toString()
}
/** Same content, but with no `</head>` to stop at -- the scan runs over the whole document. */
fun unterminatedHead(tags: Int): String = heavyHead(tags).replace("</head>", "")
fun bench(
label: String,
input: String,
reps: Int,
op: (String) -> Int,
) {
repeat(maxOf(reps / 4, 2)) { op(input) } // warmup
val t0 = System.nanoTime()
var sink = 0
repeat(reps) { sink += op(input) }
val ns = (System.nanoTime() - t0) / reps
val mbps = input.length.toDouble() / ns * 1000.0 // bytes/ns -> MB/s
println(
String.format(
"%-34s %9d B %9d ns/op %8.1f MB/s (hits=%d)",
label,
input.length,
ns,
mbps,
sink / reps,
),
)
}
}
@Test
fun metaScans() {
val spa = spaHead()
val heavy = heavyHead(60)
val open = unterminatedHead(60)
// correctness first: a benchmark that finds nothing measures nothing
val spaInfo = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(spa))
assertEquals("Example — Your Network. Your Rules.", spaInfo.title)
assertEquals("https://example.com/og-image.png", spaInfo.image)
val heavyInfo = OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(heavy))
assertEquals("The Headline", heavyInfo.title)
assertEquals("https://example.com/lead.jpg", heavyInfo.image)
// the body after </head> is never scanned
assertEquals(1 + 60 + 3, MetaTagsParser.parse(heavy).count())
// Note: the two "heavy head" rows report MB/s over the whole document, of which only the
// ~46 KB head is actually scanned -- the scan stops at </head>. The last row is the same
// page with no </head>, i.e. what a hostile server can force us to read end to end.
println("MetaTagsParser")
bench("spa head, all tags", spa, 50_000) { MetaTagsParser.parse(it).count() }
bench("spa head, og: extraction", spa, 50_000) {
OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(it)).title.length
}
bench("heavy head, to </head>", heavy, 2_000) { MetaTagsParser.parse(it).count() }
bench("heavy head, og: extraction", heavy, 2_000) {
OpenGraphParser().extractUrlInfo(MetaTagsParser.parse(it)).title.length
}
bench("no </head>, whole doc", open, 2_000) { MetaTagsParser.parse(it).count() }
assertTrue(MetaTagsParser.parse(open).count() >= 64)
}
}