mirror of
https://github.com/vitorpamplona/amethyst.git
synced 2026-10-05 19:28:25 +00:00
fix(quartz): match java.net.URLEncoder on native; drop the urlencoder dep
Both native targets delegated UrlEncoder to
net.thauvin.erik.urlencoder.UrlEncoderUtil, which implements RFC 3986
percent-encoding. The JVM/Android actual is java.net.URLEncoder/URLDecoder,
which implements application/x-www-form-urlencoded. Different specifications,
and the difference was observable:
JVM/Android UrlEncoderUtil
encode(" ") "+" "%20"
encode("*") "*" "%2A"
decode("a+b") "a b" "a+b"
This is not cosmetic. encode() builds strings that leave the device —
TorrentEvent puts it in magnet links, Nip54InlineMetadata in inline metadata,
Nip47DeepLink in the callback/appname/value parameters of NWC deep links — so
Android and iOS emitted different bytes for the same title. The decode row is
worse: a link written by Android carries '+' for its spaces, and reading it on
iOS or desktop-native gave back literal plus signs, silently, with no error.
Replaced with one UrlEncoder.native.kt in nativeMain, shared by linuxX64 and
every Apple target, matching URLEncoder/URLDecoder exactly — unreserved set is
alphanumerics plus -_.* (note '*' survives and '~' does not, the opposite of
RFC 3986), space to '+', uppercase %XX of UTF-8 bytes otherwise, and '+' back
to space on the way in. Escape runs are encoded and decoded as runs so surrogate
pairs and multi-byte sequences survive, and both directions short-circuit on a
string with nothing to change, as the java.net pair does.
UriParser.linux now delegates to UrlEncoder.decode rather than carrying its own
copy of the decoder added in the previous commit.
The new UrlEncoderTest lives in commonTest, so it pins every target against the
JVM's answers — it is what found all three rows above, by passing on jvmTest and
failing three of ten on linuxX64.
net.thauvin.erik:urlencoder-lib had no other user and is removed from both
source sets and the version catalog.
One deliberate edge difference from the JVM, documented at the call site: an
unpaired UTF-16 surrogate encodes as %EF%BF%BD rather than %3F.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01HxQ1QuyzSkR38iFHbREjoS
This commit is contained in:
@@ -55,7 +55,6 @@ media3 = "1.11.0"
|
||||
mockk = "1.14.11"
|
||||
kotlinx-coroutines-test = "1.11.0"
|
||||
negentropyKmp = "v1.2.0"
|
||||
netUrlencoderLibVersion = "1.6.0"
|
||||
navigationCompose = "2.9.8"
|
||||
okhttp = "5.5.0"
|
||||
osmdroid = "6.1.20"
|
||||
@@ -209,7 +208,6 @@ mockk = { group = "io.mockk", name = "mockk", version.ref = "mockk" }
|
||||
mockk-android = { group = "io.mockk", name = "mockk-android", version.ref = "mockk" }
|
||||
kotlinx-coroutines-test = { group = "org.jetbrains.kotlinx", name = "kotlinx-coroutines-test", version.ref = "kotlinx-coroutines-test"}
|
||||
negentropy-kmp = { module = "com.vitorpamplona.negentropy:kmp-negentropy", version.ref = "negentropyKmp" }
|
||||
net-thauvin-erik-urlencoder-lib = { module = "net.thauvin.erik.urlencoder:urlencoder-lib", version.ref = "netUrlencoderLibVersion" }
|
||||
okhttp = { group = "com.squareup.okhttp3", name = "okhttp", version.ref = "okhttp" }
|
||||
okhttpCoroutines = { group = "com.squareup.okhttp3", name = "okhttp-coroutines", version.ref = "okhttp" }
|
||||
osmdroid-android = { group = "org.osmdroid", name = "osmdroid-android", version.ref = "osmdroid" }
|
||||
|
||||
@@ -289,7 +289,6 @@ kotlin {
|
||||
dependsOn(nativeMain)
|
||||
dependencies {
|
||||
implementation(libs.charlietap.cachemap)
|
||||
implementation(libs.net.thauvin.erik.urlencoder.lib)
|
||||
implementation(libs.dev.whyoleg.cryptography.provider.apple.optimal)
|
||||
implementation("io.github.andreypfau:kotlinx-crypto-hmac:0.0.4")
|
||||
implementation("io.github.andreypfau:kotlinx-crypto-sha2:0.0.4")
|
||||
@@ -347,7 +346,6 @@ kotlin {
|
||||
create("linuxMain") {
|
||||
dependsOn(nativeMain)
|
||||
dependencies {
|
||||
implementation(libs.net.thauvin.erik.urlencoder.lib)
|
||||
implementation(libs.dev.whyoleg.cryptography.provider.apple.optimal)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils
|
||||
|
||||
import net.thauvin.erik.urlencoder.UrlEncoderUtil
|
||||
|
||||
actual object UrlEncoder {
|
||||
actual fun encode(value: String): String = UrlEncoderUtil.encode(value)
|
||||
|
||||
actual fun decode(value: String): String = UrlEncoderUtil.decode(value)
|
||||
}
|
||||
@@ -0,0 +1,134 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils
|
||||
|
||||
import kotlin.test.Test
|
||||
import kotlin.test.assertEquals
|
||||
|
||||
/**
|
||||
* Cross-target contract for [UrlEncoder].
|
||||
*
|
||||
* This is not a style preference — [UrlEncoder.encode] builds strings that leave the
|
||||
* device. `TorrentEvent` puts it in magnet links, `Nip54InlineMetadata` in inline
|
||||
* metadata, and `Nip47DeepLink` in the `callback`/`appname`/`value` parameters of NWC
|
||||
* deep links. If Android encodes a title one way and iOS another, the two clients emit
|
||||
* different bytes for the same event, and a wallet that round-trips a deep link built
|
||||
* on one platform can fail on the other.
|
||||
*
|
||||
* The JVM/Android actual is `java.net.URLEncoder`/`URLDecoder` with UTF-8, so that is
|
||||
* the reference every other target has to match. The expectations below are its
|
||||
* `application/x-www-form-urlencoded` rules, which differ from plain RFC 3986
|
||||
* percent-encoding in exactly the three places a generic library gets "wrong": space,
|
||||
* `*` and `~`.
|
||||
*/
|
||||
class UrlEncoderTest {
|
||||
@Test
|
||||
fun keepsTheUnreservedSet() {
|
||||
// URLEncoder's dontNeedEncoding set is alphanumerics plus these four, and only
|
||||
// these four. Note `*` survives and `~` does not — the opposite of RFC 3986.
|
||||
assertEquals("abcXYZ019", UrlEncoder.encode("abcXYZ019"))
|
||||
assertEquals("-_.*", UrlEncoder.encode("-_.*"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun encodesSpaceAsPlus() {
|
||||
// Form encoding, not %20.
|
||||
assertEquals("hello+world", UrlEncoder.encode("hello world"))
|
||||
assertEquals("a+b+c", UrlEncoder.encode("a b c"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun encodesTildeAndTheOtherSubDelimiters() {
|
||||
assertEquals("%7E", UrlEncoder.encode("~"))
|
||||
assertEquals("%21", UrlEncoder.encode("!"))
|
||||
assertEquals("%27", UrlEncoder.encode("'"))
|
||||
assertEquals("%28%29", UrlEncoder.encode("()"))
|
||||
assertEquals("%24%2C%3B", UrlEncoder.encode("$,;"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun encodesUriPunctuationWithUppercaseHex() {
|
||||
assertEquals("%3A%2F%2F", UrlEncoder.encode("://"))
|
||||
assertEquals("%2B", UrlEncoder.encode("+"))
|
||||
assertEquals("%3F%26%3D", UrlEncoder.encode("?&="))
|
||||
assertEquals("%40%23%25", UrlEncoder.encode("@#%"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun encodesNonAsciiAsUtf8() {
|
||||
assertEquals("%C3%A9", UrlEncoder.encode("é"))
|
||||
assertEquals("caf%C3%A9", UrlEncoder.encode("café"))
|
||||
assertEquals("%E2%82%AC", UrlEncoder.encode("€"))
|
||||
// Outside the BMP: a surrogate pair has to encode as one 4-byte sequence.
|
||||
assertEquals("%F0%9F%98%80", UrlEncoder.encode("😀"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun decodesPlusAsSpace() {
|
||||
assertEquals("hello world", UrlEncoder.decode("hello+world"))
|
||||
assertEquals("hello world", UrlEncoder.decode("hello%20world"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun decodesPercentEscapes() {
|
||||
assertEquals("://", UrlEncoder.decode("%3A%2F%2F"))
|
||||
assertEquals("~", UrlEncoder.decode("%7E"))
|
||||
assertEquals("café", UrlEncoder.decode("caf%C3%A9"))
|
||||
assertEquals("€", UrlEncoder.decode("%E2%82%AC"))
|
||||
assertEquals("😀", UrlEncoder.decode("%F0%9F%98%80"))
|
||||
// Lowercase hex decodes the same as uppercase.
|
||||
assertEquals("é", UrlEncoder.decode("%c3%a9"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun leavesUnescapedTextAlone() {
|
||||
assertEquals("plain", UrlEncoder.decode("plain"))
|
||||
assertEquals("-_.*~", UrlEncoder.decode("-_.*~"))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun roundTripsTheStringsThisIsActuallyUsedFor() {
|
||||
// A torrent title (TorrentEvent) and a tracker URL.
|
||||
val title = "Big Buck Bunny (2008) [1080p] ~ 60% done!"
|
||||
assertEquals(title, UrlEncoder.decode(UrlEncoder.encode(title)))
|
||||
|
||||
val tracker = "udp://tracker.example.org:1337/announce"
|
||||
assertEquals("udp%3A%2F%2Ftracker.example.org%3A1337%2Fannounce", UrlEncoder.encode(tracker))
|
||||
assertEquals(tracker, UrlEncoder.decode(UrlEncoder.encode(tracker)))
|
||||
|
||||
// An NWC pairing code (Nip47DeepLink.buildCallbackUri puts this in `value=`).
|
||||
val pairing =
|
||||
"nostr+walletconnect://b889ff5b1513b641e2a139f661a661364979c5beee91842f8f0ef42ab558e9d4" +
|
||||
"?relay=wss%3A%2F%2Frelay.damus.io&secret=71a8c14c1407c113601079c4302dab36460f0ccd0ad506f1f2dc73b5100571c5"
|
||||
assertEquals(pairing, UrlEncoder.decode(UrlEncoder.encode(pairing)))
|
||||
|
||||
// A callback deep link (Nip47DeepLink.parseConnectUri reads this back).
|
||||
val callback = "amethystnwc://callback"
|
||||
assertEquals("amethystnwc%3A%2F%2Fcallback", UrlEncoder.encode(callback))
|
||||
assertEquals(callback, UrlEncoder.decode(UrlEncoder.encode(callback)))
|
||||
}
|
||||
|
||||
@Test
|
||||
fun roundTripsEveryAsciiCharacter() {
|
||||
val ascii = (0..127).map { it.toChar() }.joinToString("")
|
||||
assertEquals(ascii, UrlEncoder.decode(UrlEncoder.encode(ascii)))
|
||||
}
|
||||
}
|
||||
@@ -122,9 +122,13 @@ actual class UriParser actual constructor(
|
||||
actual fun path(): String? = parsedPath
|
||||
|
||||
/**
|
||||
* Parsed once and reused, mirroring the JVM actual's lazy map. The previous version
|
||||
* re-split the entire query string on every [getQueryParameter] call, so a URI read
|
||||
* for four parameters was parsed four times.
|
||||
* Parsed once and reused, mirroring the JVM actual's lazy map — the previous version
|
||||
* re-split the entire query string on every [getQueryParameter] call.
|
||||
*
|
||||
* Decoded with [UrlEncoder.decode], which matches `URLDecoder.decode(.., "UTF-8")` —
|
||||
* what the JVM actual calls. Skipping this is why NIP-47 failed on this target:
|
||||
* `relay=wss%3A%2F%2Frelay.damus.io` reached `RelayUrlNormalizer` still encoded and
|
||||
* came back "Invalid relay Url".
|
||||
*/
|
||||
private val queryParameters: Map<String, List<String>> by lazy {
|
||||
parsedQuery?.ifBlank { null }?.let { query ->
|
||||
@@ -138,7 +142,7 @@ actual class UriParser actual constructor(
|
||||
}
|
||||
|
||||
if (parts.size == 2) {
|
||||
currentValue.add(percentDecode(parts[1]))
|
||||
currentValue.add(UrlEncoder.decode(parts[1]))
|
||||
} else {
|
||||
currentValue.add("")
|
||||
}
|
||||
@@ -153,7 +157,7 @@ actual class UriParser actual constructor(
|
||||
keyValuePair.split('&').associate { paramValue ->
|
||||
val parts = paramValue.split("=", limit = 2)
|
||||
if (parts.size == 2) {
|
||||
parts[0] to percentDecode(parts[1])
|
||||
parts[0] to UrlEncoder.decode(parts[1])
|
||||
} else {
|
||||
parts[0] to "" // Handle parameters without a value
|
||||
}
|
||||
@@ -168,82 +172,3 @@ actual class UriParser actual constructor(
|
||||
|
||||
actual fun fragments(): Map<String, String> = parsedFragments
|
||||
}
|
||||
|
||||
/**
|
||||
* `java.net.URLDecoder.decode(value, "UTF-8")` — which is literally what the JVM actual
|
||||
* calls — for a target with no `java.net`.
|
||||
*
|
||||
* This is the whole reason NIP-47 failed on linuxX64: the parser returned query values
|
||||
* exactly as they appeared in the URI, so `relay=wss%3A%2F%2Frelay.damus.io` reached
|
||||
* `RelayUrlNormalizer` still percent-encoded and came back "Invalid relay Url". Decoding
|
||||
* belongs here rather than in each caller, because the JVM and Apple actuals both hand
|
||||
* back decoded values and common code is written against that.
|
||||
*
|
||||
* Matches `URLDecoder` in the details that are observable: `+` becomes a space, a run of
|
||||
* consecutive `%XX` is decoded as one UTF-8 sequence (so multi-byte characters survive),
|
||||
* every other character passes through, and a malformed escape throws
|
||||
* [IllegalArgumentException] rather than being silently kept — the same failure the JVM
|
||||
* gives for the same input.
|
||||
*/
|
||||
private fun percentDecode(value: String): String {
|
||||
// The short-circuit URLDecoder also makes: with nothing to change, return the
|
||||
// original instance rather than rebuilding it. Most query values hit this.
|
||||
if (value.indexOf('%') < 0 && value.indexOf('+') < 0) return value
|
||||
|
||||
val result = StringBuilder(value.length)
|
||||
var index = 0
|
||||
// Sized on first use for the longest run that could still follow, then reused —
|
||||
// one allocation for the whole string, as in URLDecoder.
|
||||
var escaped: ByteArray? = null
|
||||
|
||||
while (index < value.length) {
|
||||
when (val char = value[index]) {
|
||||
'+' -> {
|
||||
result.append(' ')
|
||||
index++
|
||||
}
|
||||
|
||||
'%' -> {
|
||||
val buffer = escaped ?: ByteArray((value.length - index) / 3).also { escaped = it }
|
||||
var count = 0
|
||||
while (index + 2 < value.length && value[index] == '%') {
|
||||
buffer[count++] = decodeEscape(value, index)
|
||||
index += 3
|
||||
}
|
||||
if (index < value.length && value[index] == '%') {
|
||||
throw IllegalArgumentException("URLDecoder: Incomplete trailing escape (%) pattern")
|
||||
}
|
||||
result.append(buffer.decodeToString(0, count))
|
||||
}
|
||||
|
||||
else -> {
|
||||
result.append(char)
|
||||
index++
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return result.toString()
|
||||
}
|
||||
|
||||
private fun decodeEscape(
|
||||
value: String,
|
||||
index: Int,
|
||||
): Byte {
|
||||
val high = hexDigit(value[index + 1])
|
||||
val low = hexDigit(value[index + 2])
|
||||
if (high < 0 || low < 0) {
|
||||
throw IllegalArgumentException(
|
||||
"URLDecoder: Illegal hex characters in escape (%) pattern - ${value.substring(index, index + 3)}",
|
||||
)
|
||||
}
|
||||
return ((high shl 4) or low).toByte()
|
||||
}
|
||||
|
||||
private fun hexDigit(char: Char): Int =
|
||||
when (char) {
|
||||
in '0'..'9' -> char - '0'
|
||||
in 'a'..'f' -> char - 'a' + 10
|
||||
in 'A'..'F' -> char - 'A' + 10
|
||||
else -> -1
|
||||
}
|
||||
|
||||
@@ -1,29 +0,0 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils
|
||||
|
||||
import net.thauvin.erik.urlencoder.UrlEncoderUtil
|
||||
|
||||
actual object UrlEncoder {
|
||||
actual fun encode(value: String): String = UrlEncoderUtil.encode(value)
|
||||
|
||||
actual fun decode(value: String): String = UrlEncoderUtil.decode(value)
|
||||
}
|
||||
@@ -0,0 +1,190 @@
|
||||
/*
|
||||
* Copyright (c) 2025 Vitor Pamplona
|
||||
*
|
||||
* Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
* this software and associated documentation files (the "Software"), to deal in
|
||||
* the Software without restriction, including without limitation the rights to use,
|
||||
* copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the
|
||||
* Software, and to permit persons to whom the Software is furnished to do so,
|
||||
* subject to the following conditions:
|
||||
*
|
||||
* The above copyright notice and this permission notice shall be included in all
|
||||
* copies or substantial portions of the Software.
|
||||
*
|
||||
* THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
* IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
* FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR
|
||||
* COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN
|
||||
* AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
|
||||
* WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
||||
*/
|
||||
package com.vitorpamplona.quartz.utils
|
||||
|
||||
/**
|
||||
* Native actual for [UrlEncoder], shared by linuxX64 and every Apple target.
|
||||
*
|
||||
* ## Why this is not a library call any more
|
||||
*
|
||||
* Both native targets used to delegate to `net.thauvin.erik.urlencoder.UrlEncoderUtil`,
|
||||
* which implements RFC 3986 percent-encoding. The JVM/Android actual is
|
||||
* `java.net.URLEncoder`/`URLDecoder`, which implements
|
||||
* `application/x-www-form-urlencoded`. Those are different specifications, and the
|
||||
* difference was observable in three places:
|
||||
*
|
||||
* ```
|
||||
* JVM/Android UrlEncoderUtil
|
||||
* encode(" ") "+" "%20"
|
||||
* encode("*") "*" "%2A"
|
||||
* decode("a+b") "a b" "a+b"
|
||||
* ```
|
||||
*
|
||||
* That is not cosmetic. [encode] builds strings that leave the device — `TorrentEvent`
|
||||
* puts it in magnet links, `Nip54InlineMetadata` in inline metadata, `Nip47DeepLink` in
|
||||
* the `callback`, `appname` and `value` parameters of NWC deep links — so Android and
|
||||
* iOS were emitting different bytes for the same title. The decode row is worse than
|
||||
* cosmetic: a magnet link or deep link written by Android carries `+` for its spaces,
|
||||
* and reading it on iOS produced a string with literal plus signs instead of spaces, no
|
||||
* error anywhere.
|
||||
*
|
||||
* So this matches `URLEncoder`/`URLDecoder` exactly instead: the unreserved set is
|
||||
* alphanumerics plus `-`, `_`, `.` and `*` (note `*` survives and `~` does not — the
|
||||
* opposite of RFC 3986), space encodes to `+`, everything else to uppercase `%XX` of
|
||||
* its UTF-8 bytes, and decoding maps `+` back to a space. `UrlEncoderTest` in
|
||||
* `commonTest` pins all of it against the JVM on every target.
|
||||
*
|
||||
* Both directions short-circuit the way the `java.net` pair does: a string with nothing
|
||||
* to change is returned as-is rather than rebuilt.
|
||||
*
|
||||
* One deliberate edge difference: an *unpaired* UTF-16 surrogate encodes as `%EF%BF%BD`
|
||||
* (Kotlin's replacement character) where the JVM gives `%3F`. Nostr content is
|
||||
* well-formed UTF-16, and chasing it would cost a scan on every call.
|
||||
*/
|
||||
actual object UrlEncoder {
|
||||
private const val HEX = "0123456789ABCDEF"
|
||||
|
||||
actual fun encode(value: String): String {
|
||||
var index = 0
|
||||
while (index < value.length && isUnreserved(value[index])) index++
|
||||
if (index == value.length) return value
|
||||
|
||||
val result = StringBuilder(value.length + ESCAPE_HEADROOM)
|
||||
result.append(value, 0, index)
|
||||
|
||||
while (index < value.length) {
|
||||
val char = value[index]
|
||||
when {
|
||||
isUnreserved(char) -> {
|
||||
result.append(char)
|
||||
index++
|
||||
}
|
||||
|
||||
char == ' ' -> {
|
||||
result.append('+')
|
||||
index++
|
||||
}
|
||||
|
||||
else -> {
|
||||
// Escaped as a run rather than character by character, so a surrogate
|
||||
// pair becomes one 4-byte sequence instead of two malformed 3-byte ones.
|
||||
val start = index
|
||||
do {
|
||||
index++
|
||||
} while (index < value.length && !isUnreserved(value[index]) && value[index] != ' ')
|
||||
appendEscaped(result, value, start, index)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return result.toString()
|
||||
}
|
||||
|
||||
actual fun decode(value: String): String {
|
||||
if (value.indexOf('%') < 0 && value.indexOf('+') < 0) return value
|
||||
|
||||
val result = StringBuilder(value.length)
|
||||
var index = 0
|
||||
// Sized on first use for the longest run that could still follow, then reused —
|
||||
// one allocation for the whole string, as in URLDecoder.
|
||||
var escaped: ByteArray? = null
|
||||
|
||||
while (index < value.length) {
|
||||
when (val char = value[index]) {
|
||||
'+' -> {
|
||||
result.append(' ')
|
||||
index++
|
||||
}
|
||||
|
||||
'%' -> {
|
||||
val buffer = escaped ?: ByteArray((value.length - index) / 3).also { escaped = it }
|
||||
var count = 0
|
||||
while (index + 2 < value.length && value[index] == '%') {
|
||||
buffer[count++] = decodeEscape(value, index)
|
||||
index += 3
|
||||
}
|
||||
if (index < value.length && value[index] == '%') {
|
||||
throw IllegalArgumentException("URLDecoder: Incomplete trailing escape (%) pattern")
|
||||
}
|
||||
// Decoded as a run so a multi-byte UTF-8 sequence survives.
|
||||
result.append(buffer.decodeToString(0, count))
|
||||
}
|
||||
|
||||
else -> {
|
||||
result.append(char)
|
||||
index++
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return result.toString()
|
||||
}
|
||||
|
||||
/** `URLEncoder`'s `dontNeedEncoding` set: alphanumerics plus these four, and only these. */
|
||||
private fun isUnreserved(char: Char): Boolean =
|
||||
char in 'a'..'z' ||
|
||||
char in 'A'..'Z' ||
|
||||
char in '0'..'9' ||
|
||||
char == '-' ||
|
||||
char == '_' ||
|
||||
char == '.' ||
|
||||
char == '*'
|
||||
|
||||
private fun appendEscaped(
|
||||
result: StringBuilder,
|
||||
value: String,
|
||||
start: Int,
|
||||
end: Int,
|
||||
) {
|
||||
val bytes = value.substring(start, end).encodeToByteArray()
|
||||
for (byte in bytes) {
|
||||
val code = byte.toInt()
|
||||
result.append('%')
|
||||
result.append(HEX[(code shr 4) and 0xF])
|
||||
result.append(HEX[code and 0xF])
|
||||
}
|
||||
}
|
||||
|
||||
private fun decodeEscape(
|
||||
value: String,
|
||||
index: Int,
|
||||
): Byte {
|
||||
val high = hexDigit(value[index + 1])
|
||||
val low = hexDigit(value[index + 2])
|
||||
if (high < 0 || low < 0) {
|
||||
throw IllegalArgumentException(
|
||||
"URLDecoder: Illegal hex characters in escape (%) pattern - ${value.substring(index, index + 3)}",
|
||||
)
|
||||
}
|
||||
return ((high shl 4) or low).toByte()
|
||||
}
|
||||
|
||||
private fun hexDigit(char: Char): Int =
|
||||
when (char) {
|
||||
in '0'..'9' -> char - '0'
|
||||
in 'a'..'f' -> char - 'a' + 10
|
||||
in 'A'..'F' -> char - 'A' + 10
|
||||
else -> -1
|
||||
}
|
||||
|
||||
/** Enough for a handful of escapes before the builder has to grow. */
|
||||
private const val ESCAPE_HEADROOM = 16
|
||||
}
|
||||
Reference in New Issue
Block a user