Files
amethyst/tools/unicode-nfkc/generate.py
Claude 0a48ddab5d fix(quartz): real NFKC on Linux, private zaps and deriveKey over NIP-46
Linux native's UnicodeNormalizer returned its input unchanged, so NIP-49
keys encrypted under a non-ASCII password (normalized by every other
client) could not be decrypted there. Add a pure-Kotlin UAX #15 NFKC
normalizer with tables generated from the Unicode 17.0 UCD by
tools/unicode-nfkc/generate.py, and use it on Linux. It passes all 20,034
lines of Unicode's NormalizationTest.txt and matches java.text.Normalizer
for every code point the JDK defines (NfkcNormalizerJdkParityTest).

NostrSignerRemote.decryptZapEvent and deriveKey were TODO(), whose
NotImplementedError is an Error that escapes DecryptCache's handlers.
A private zap's anon payload is AES-CBC under the NIP-04 shared secret,
so the recipient now decrypts it through the bunker's nip04_decrypt.
The sender's copy and deriveKey need the raw private key, which a bunker
never exposes, so they now throw CouldNotPerformException and
UnsupportedMethodException. An end-to-end test runs the remote signer
against Quartz's own bunker over an in-process relay.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01QytMdt3MPxvmmWrAX3bJYS
2026-09-28 17:35:15 +00:00

138 lines
5.1 KiB
Python
Executable File

#!/usr/bin/env python3
"""Generate Quartz's pure-Kotlin NFKC tables from the Unicode Character Database.
Usage:
tools/unicode-nfkc/generate.py <ucd-dir> <output.kt>
<ucd-dir> must hold UnicodeData.txt and DerivedNormalizationProps.txt from
https://www.unicode.org/Public/<version>/ucd/. The version is read from the
DerivedNormalizationProps.txt header and stamped into the output.
Three tables are emitted, all as hex strings the Kotlin side parses once:
CCC - "start-end:ccc" ranges of non-zero canonical combining classes
DECOMP - "cp:x y z" full compatibility (NFKD) decompositions, already
expanded recursively. Hangul syllables are algorithmic and omitted.
COMPOSE - "first second composite" primary composites, i.e. two-code-point
canonical decompositions minus Full_Composition_Exclusion.
"""
import os
import re
import sys
CHUNK = 16000 # keeps each string literal far below the JVM's 64 KB constant limit
def read_lines(path):
with open(path, encoding="utf-8") as f:
for line in f:
line = line.split("#", 1)[0].strip()
if line:
yield line
def parse_range(text):
if ".." in text:
a, b = text.split("..")
return range(int(a, 16), int(b, 16) + 1)
return range(int(text, 16), int(text, 16) + 1)
def main(ucd_dir, out_path):
props_path = os.path.join(ucd_dir, "DerivedNormalizationProps.txt")
with open(props_path, encoding="utf-8") as f:
version = re.search(r"DerivedNormalizationProps-([\d.]+)\.txt", f.readline()).group(1)
ccc = {}
decomp = {} # cp -> (is_compat, [cps])
for line in read_lines(os.path.join(ucd_dir, "UnicodeData.txt")):
fields = line.split(";")
cp = int(fields[0], 16)
if int(fields[3]):
ccc[cp] = int(fields[3])
if fields[5]:
parts = fields[5].split()
compat = parts[0].startswith("<")
if compat:
parts = parts[1:]
decomp[cp] = (compat, [int(p, 16) for p in parts])
exclusions = set()
for line in read_lines(props_path):
fields = [x.strip() for x in line.split(";")]
if fields[1] == "Full_Composition_Exclusion":
exclusions.update(parse_range(fields[0]))
def expand(cp):
if cp not in decomp:
return [cp]
out = []
for c in decomp[cp][1]:
out.extend(expand(c))
return out
ccc_ranges = []
for cp in sorted(ccc):
last = ccc_ranges[-1] if ccc_ranges else None
if last and last[1] == cp - 1 and last[2] == ccc[cp]:
last[1] = cp
else:
ccc_ranges.append([cp, cp, ccc[cp]])
ccc_entries = ["%X-%X:%X" % (a, b, c) for a, b, c in ccc_ranges]
decomp_entries = [
"%X:%s" % (cp, " ".join("%X" % c for c in expand(cp))) for cp in sorted(decomp)
]
compose_entries = [
"%X %X %X" % (parts[0], parts[1], cp)
for cp, (compat, parts) in sorted(decomp.items())
if not compat and len(parts) == 2 and cp not in exclusions
]
def chunks(entries):
out, cur = [], ""
for e in entries:
if cur and len(cur) + len(e) + 1 > CHUNK:
out.append(cur)
cur = ""
cur = e if not cur else cur + ";" + e
if cur:
out.append(cur)
return out
def kotlin_array(name, entries):
body = "".join(' "%s",\n' % c for c in chunks(entries))
return " val %s =\n arrayOf(\n%s )\n" % (name, body)
header_path = os.path.join(os.path.dirname(os.path.abspath(__file__)), "..", "..", "quartz", "src",
"commonMain", "kotlin", "com", "vitorpamplona", "quartz", "utils",
"UnicodeNormalizer.kt")
with open(header_path, encoding="utf-8") as f:
license_header = f.read().split("package ", 1)[0]
with open(out_path, "w", encoding="utf-8") as out:
out.write(license_header)
out.write("package com.vitorpamplona.quartz.utils.unicode\n\n")
out.write("// GENERATED by tools/unicode-nfkc/generate.py from the Unicode %s Character Database.\n" % version)
out.write("// Do not edit by hand; re-run the script to update.\n")
out.write("//\n")
out.write("// Derived from Unicode data files, Copyright (c) 1991-2025 Unicode, Inc.,\n")
out.write("// distributed under the Unicode License v3: https://www.unicode.org/license.txt\n")
out.write("internal object NfkcData {\n")
out.write(' const val UNICODE_VERSION = "%s"\n\n' % version)
out.write(kotlin_array("CCC", ccc_entries))
out.write("\n")
out.write(kotlin_array("DECOMP", decomp_entries))
out.write("\n")
out.write(kotlin_array("COMPOSE", compose_entries))
out.write("}\n")
print("Unicode %s: %d ccc ranges, %d decompositions, %d compositions"
% (version, len(ccc_ranges), len(decomp_entries), len(compose_entries)))
if __name__ == "__main__":
if len(sys.argv) != 3:
sys.exit(__doc__)
main(sys.argv[1], sys.argv[2])