diff --git a/www/js/stego.mjs b/www/js/stego.mjs new file mode 100644 index 0000000..b18274b --- /dev/null +++ b/www/js/stego.mjs @@ -0,0 +1,345 @@ +/** + * LLM-based steganography — JavaScript port of stego.py. + * + * Half-splitting entropy coding on top of GPT-2's next-token distribution. + * Both encoder and decoder run the same JS code with the same PRNG seed, + * so they stay synchronized regardless of cross-language PRNG differences. + */ + +import { Tensor } from "https://cdn.jsdelivr.net/npm/@huggingface/transformers@3.0.0"; + +// --------------------------------------------------------------------------- +// PRNG — mulberry32 (deterministic given seed) +// --------------------------------------------------------------------------- +export function mulberry32(seed) { + let a = seed >>> 0; + return function random() { + a = (a + 0x6D2B79F5) >>> 0; + let t = Math.imul(a ^ (a >>> 15), 1 | a); + t = (t + Math.imul(t ^ (t >>> 7), 61 | t)) ^ t; + return ((t ^ (t >>> 14)) >>> 0) / 4294967296; + }; +} + +// --------------------------------------------------------------------------- +// Bit <-> string conversion (UTF-8) +// --------------------------------------------------------------------------- +export function strToBits(s) { + const data = new TextEncoder().encode(s); + const bits = []; + for (const byte of data) { + for (let i = 7; i >= 0; i--) { + bits.push((byte >> i) & 1); + } + } + return bits; +} + +export function bitsToStr(bits) { + // Pad to a multiple of 8 (should already be). + const padded = bits.slice(); + while (padded.length % 8 !== 0) padded.push(0); + const bytes = new Uint8Array(Math.floor(padded.length / 8)); + for (let i = 0; i < padded.length; i += 8) { + let byte = 0; + for (let j = 0; j < 8; j++) { + byte = (byte << 1) | padded[i + j]; + } + bytes[i / 8] = byte; + } + return new TextDecoder("utf-8", { fatal: false }).decode(bytes); +} + +// --------------------------------------------------------------------------- +// Softmax over a TypedArray / Array of logits (numerically stable) +// --------------------------------------------------------------------------- +function softmax(logits) { + let max = -Infinity; + for (let i = 0; i < logits.length; i++) { + if (logits[i] > max) max = logits[i]; + } + const probs = new Float32Array(logits.length); + let sum = 0; + for (let i = 0; i < logits.length; i++) { + const e = Math.exp(logits[i] - max); + probs[i] = e; + sum += e; + } + if (sum <= 0) sum = 1; + for (let i = 0; i < logits.length; i++) { + probs[i] /= sum; + } + return probs; +} + +// --------------------------------------------------------------------------- +// Next-token probability distribution from the model. +// Returns a Float32Array of length vocab_size. +// --------------------------------------------------------------------------- +async function nextTokenProbs(model, tokenizer, tokenIds) { + // Build the model inputs directly from the token-id array. We must NOT + // pass the id array through tokenizer(): in Transformers.js v3 the + // tokenizer call expects a string and will fail with + // "e.split is not a function" when given numbers. Instead we construct + // int64 tensors ourselves and feed them straight to the model. + const seqLength = tokenIds.length; + const idData = BigInt64Array.from(tokenIds, (x) => BigInt(x)); + const maskData = BigInt64Array.from({ length: seqLength }, () => 1n); + const inputs = { + input_ids: new Tensor("int64", idData, [1, seqLength]), + attention_mask: new Tensor("int64", maskData, [1, seqLength]), + }; + const output = await model(inputs); + const logits = output.logits; + + // logits.dims = [batch, seq_len, vocab_size] + const dims = logits.dims; + const seqLen = dims[1]; + const vocabSize = dims[2]; + const data = logits.data; + + // Extract the last token's logits: row = (batch=0, last token, :) + const offset = (seqLen - 1) * vocabSize; + const lastLogits = new Float32Array(vocabSize); + for (let i = 0; i < vocabSize; i++) { + lastLogits[i] = data[offset + i]; + } + return softmax(lastLogits); +} + +// --------------------------------------------------------------------------- +// Split the sorted vocabulary into two halves at cumulative prob 0.50. +// Returns { firstIds, firstProbs, secondIds, secondProbs }. +// --------------------------------------------------------------------------- +export function splitHalves(probs) { + const vocabSize = probs.length; + + // Build an index array covering EVERY token id [0 .. vocabSize-1] and + // sort it by probability descending. Sorting indices (rather than + // dropping any) guarantees that all token ids are retained and that the + // two slices below form a complete, disjoint partition of the vocabulary. + // Ties are broken by ascending token id so the ordering is fully + // deterministic — important under int8 quantization where many logits can + // be exactly equal (an unstable sort could otherwise diverge between the + // encoder and decoder passes). + const sortedIds = new Array(vocabSize); + for (let i = 0; i < vocabSize; i++) sortedIds[i] = i; + sortedIds.sort((a, b) => { + const diff = probs[b] - probs[a]; + if (diff !== 0) return diff; + return a - b; // stable tie-break by token id + }); + + // Find the split point: the first index at which the cumulative + // probability mass reaches/exceeds 0.50. Everything up to and including + // that token forms the first half; the remainder forms the second half. + let cum = 0.0; + let splitIdx = vocabSize; // default: all mass in first half (edge case) + for (let i = 0; i < vocabSize; i++) { + cum += probs[sortedIds[i]]; + if (cum >= 0.50) { + splitIdx = i + 1; // include this token in the first half + break; + } + } + + // Clamp so BOTH halves are guaranteed non-empty. The split index must be + // at least 1 (first half non-empty) and at most vocabSize-1 (second half + // non-empty). This handles degenerate distributions where a single token + // already holds >= 0.50 of the mass, or where the mass never reaches 0.50. + if (splitIdx < 1) splitIdx = 1; + if (splitIdx > vocabSize - 1) splitIdx = vocabSize - 1; + + // slice(0, splitIdx) + slice(splitIdx) ALWAYS covers the entire sorted + // array, so every token id appears in exactly one half — none missing, + // none duplicated. Ids are plain Numbers (0..vocabSize-1). + const firstIds = sortedIds.slice(0, splitIdx); + const secondIds = sortedIds.slice(splitIdx); + const firstProbs = firstIds.map(id => probs[id]); + const secondProbs = secondIds.map(id => probs[id]); + + return { firstIds, firstProbs, secondIds, secondProbs }; +} + +// --------------------------------------------------------------------------- +// Sample a token id from the given half using renormalized probabilities. +// Advances the PRNG exactly once (one call to rng()). +// --------------------------------------------------------------------------- +export function sampleWithinHalf(ids, probs, rng) { + let total = 0; + for (let i = 0; i < probs.length; i++) total += probs[i]; + if (total <= 0) total = 1; + const r = rng(); + let cum = 0.0; + let chosen = ids[ids.length - 1]; + for (let i = 0; i < ids.length; i++) { + cum += probs[i] / total; + if (r < cum) { + chosen = ids[i]; + break; + } + } + return chosen; +} + +// --------------------------------------------------------------------------- +// Encoder +// --------------------------------------------------------------------------- +export async function encode(model, tokenizer, secretMessage, context, key, padTokens = 3, onToken = null) { + const rng = mulberry32(key >>> 0); + const bits = strToBits(secretMessage); + const numBits = bits.length; + const totalBits = numBits; + const totalChars = secretMessage.length; + + // Tokenize the context. Transformers.js tokenizer() returns an object + // with input_ids; we extract the raw id array. + const tokenIds = tokenizeToIds(tokenizer, context); + + for (let bitIndex = 0; bitIndex < bits.length; bitIndex++) { + const bit = bits[bitIndex]; + const probs = await nextTokenProbs(model, tokenizer, tokenIds); + const { firstIds, firstProbs, secondIds, secondProbs } = splitHalves(probs); + let chosen; + if (bit === 0) { + chosen = sampleWithinHalf(firstIds, firstProbs, rng); + } else { + chosen = sampleWithinHalf(secondIds, secondProbs, rng); + } + tokenIds.push(chosen); + + // Stream the partial cover text back to the caller and yield to the + // browser so the UI can repaint between (slow) forward passes. + if (onToken) { + const textSoFar = tokenizer.decode(tokenIds, { skip_special_tokens: true }); + const charIndex = Math.floor(bitIndex / 8) + 1; + const currentChar = secretMessage[charIndex - 1] ?? ""; + onToken(textSoFar, bitIndex + 1, totalBits, currentChar, charIndex, totalChars); + } + await new Promise(resolve => setTimeout(resolve, 0)); + } + + // Padding tokens: sample from the full distribution using the same PRNG. + for (let p = 0; p < padTokens; p++) { + const probs = await nextTokenProbs(model, tokenizer, tokenIds); + // Sort descending. + const pairs = []; + for (let i = 0; i < probs.length; i++) pairs.push([probs[i], i]); + pairs.sort((a, b) => b[0] - a[0]); + const sp = pairs.map(x => x[0]); + const si = pairs.map(x => x[1]); + let total = 0; + for (let i = 0; i < sp.length; i++) total += sp[i]; + if (total <= 0) total = 1; + const r = rng(); + let cum = 0.0; + let chosen = si[0]; + for (let i = 0; i < si.length; i++) { + cum += sp[i] / total; + if (r < cum) { + chosen = si[i]; + break; + } + } + tokenIds.push(chosen); + } + + const coverText = tokenizer.decode(tokenIds, { skip_special_tokens: true }); + return { coverText, numBits, tokenCount: tokenIds.length }; +} + +// --------------------------------------------------------------------------- +// Decoder +// --------------------------------------------------------------------------- +export async function decode(model, tokenizer, coverText, context, key, numBits, padTokens = 3, onProgress = null) { + const rng = mulberry32(key >>> 0); + const totalBits = numBits; + + const contextIds = tokenizeToIds(tokenizer, context); + const coverIds = tokenizeToIds(tokenizer, coverText); + + // Generated tokens are everything after the context prefix. + const genIds = coverIds.slice(contextIds.length); + + const tokenIds = contextIds.slice(); + const bits = []; + + for (const genToken of genIds) { + if (bits.length >= numBits) break; + const probs = await nextTokenProbs(model, tokenizer, tokenIds); + const { firstIds, firstProbs, secondIds, secondProbs } = splitHalves(probs); + + // splitHalves yields plain Number ids and partitions the entire + // vocabulary, so a normalized (Number) genToken is guaranteed to be in + // exactly one half. tokenizeToIds normalizes ids to Number, but coerce + // defensively here too in case a BigInt slips through. + const token = typeof genToken === "bigint" ? Number(genToken) : genToken; + const firstSet = new Set(firstIds); + const secondSet = new Set(secondIds); + + let recoveredBit; + if (firstSet.has(token)) { + recoveredBit = 0; + bits.push(0); + // Advance PRNG identically to the encoder. + sampleWithinHalf(firstIds, firstProbs, rng); + } else if (secondSet.has(token)) { + recoveredBit = 1; + bits.push(1); + sampleWithinHalf(secondIds, secondProbs, rng); + } else { + throw new Error(`Generated token ${token} not found in either half.`); + } + tokenIds.push(token); + + // Report progress and yield so the UI can repaint between passes. + if (onProgress) { + const currentToken = tokenizer.decode([token], { skip_special_tokens: true }); + const completedBytes = Math.floor(bits.length / 8); + const recoveredChars = bitsToStr(bits.slice(0, completedBytes * 8)); + onProgress( + bits.length, + totalBits, + recoveredChars, + currentToken, + recoveredBit, + ); + } + await new Promise(resolve => setTimeout(resolve, 0)); + } + + return bitsToStr(bits.slice(0, numBits)); +} + +// --------------------------------------------------------------------------- +// Helper: tokenize a string to a plain array of token ids. +// Handles the various shapes Transformers.js may return. +// --------------------------------------------------------------------------- +function tokenizeToIds(tokenizer, text) { + const inputs = tokenizer(text, { add_special_tokens: false }); + let ids; + if (inputs && inputs.input_ids) { + ids = inputs.input_ids; + } else if (Array.isArray(inputs)) { + ids = inputs; + } else { + ids = inputs; + } + // ids may be a Tensor or a nested array like [[1,2,3]]. + let arr; + if (ids && typeof ids.data !== "undefined" && ids.dims) { + arr = Array.from(ids.data); + } else if (Array.isArray(ids)) { + // Flatten one level if nested: [[...]] -> [...] + arr = (ids.length > 0 && Array.isArray(ids[0])) ? ids[0].slice() : ids.slice(); + } else { + // Fallback: try to iterate. + arr = Array.from(ids); + } + // CRITICAL: token tensors are int64, so ids.data is a BigInt64Array and + // Array.from() yields BigInts (e.g. 351n). splitHalves produces plain + // Number ids, and `new Set([...Numbers]).has(351n)` is false — which is + // exactly what caused "Generated token 351 not found in either half". + // Normalize every id to a plain Number so membership checks line up. + return arr.map(v => (typeof v === "bigint" ? Number(v) : Number(v))); +} diff --git a/www/js/version.json b/www/js/version.json index 943a4e9..df14d11 100644 --- a/www/js/version.json +++ b/www/js/version.json @@ -1,5 +1,5 @@ { - "VERSION": "v0.7.56", - "VERSION_NUMBER": "0.7.56", - "BUILD_DATE": "2026-06-27T12:17:42.524Z" + "VERSION": "v0.7.57", + "VERSION_NUMBER": "0.7.57", + "BUILD_DATE": "2026-06-29T00:17:04.254Z" } diff --git a/www/llm-steganography.html b/www/llm-steganography.html new file mode 100644 index 0000000..b0d56dc --- /dev/null +++ b/www/llm-steganography.html @@ -0,0 +1,1518 @@ + + + + + + + LLM STEGANOGRAPHY + + + + + + + + + + + + + +
+ +
+ + +
+
+ +
+ +
+
+
+ +
+ +
+
+ + +
+ + + + +
+
+

LLM Steganography Demo

+

Hide secret messages in innocent-looking AI-generated text

+
+ + +
+

Model Loading

+

Initializing...

+
+
+
+
+
+ + +
+ + +
+

Shared Parameters

+

Both encoder and decoder must use the same context and key.

+
+ + +
+
+ + +
+
+ + +
+

Encode

+
+ + +
+ +

+
+
+ Encoding character (0/0) +
+
+ bit 0 of 0 +
+
+
+
+
+
+ + +
+

+
+ + +
+

Decode

+ +

+
+
+ Processing token: → bit - +
+
+ bit 0 of 0 +
+
+
+
+
+ Recovered so far: +
+
+
+ + +
+

+
+ + +
+

+ +

+
+

+ This demo uses a half-splitting entropy coding scheme + built on top of GPT-2's next-token probability distribution. +

+
    +
  1. The secret message is converted to a bit string (UTF-8 → bits).
  2. +
  3. For each secret bit, GPT-2 produces a probability distribution over + the entire vocabulary for the next token.
  4. +
  5. Tokens are sorted by probability (descending) and split into two + halves at the 50% cumulative probability mark.
  6. +
  7. Bit 0 → the next token is sampled from the + first (higher-probability) half; bit 1 → + from the second half.
  8. +
  9. A shared-key PRNG (mulberry32) selects the exact token within the + chosen half, so the decoder can reproduce the same random draws.
  10. +
  11. The decoder re-runs GPT-2 on the same context, observes which half + each cover token fell into, and recovers the bits → original message.
  12. +
+

+ Because both sides share the same model, context, and PRNG seed, the + decoder can perfectly reconstruct the hidden bits. The resulting cover + text reads like normal GPT-2 output, hiding the secret in plain sight. +

+
+
+ +
+ + +
+
+ + +
+
+
+
+
0 sats
+
+ + +
+
+ +
+ +
+
+
+ +
+
AI
+
+
No saved providers yet.
+
+
+ + +
+
+ リレー +
+
+ Loading relays... +
+
+ +
+ +
ブロッサム
+ +
Loading blossom servers...
+ +
+ + +
+ v0.0.1 +
+ + +
+
+
+ + + + + + + + + + + +