From 190316eb2849f0d1e80b6b464b417d573717532c Mon Sep 17 00:00:00 2001 From: davotoula Date: Tue, 9 Jun 2026 21:02:28 +0200 Subject: [PATCH 1/3] Replace GPLv3 TarsosDSP with in-house pitch shifter --- amethyst/build.gradle.kts | 3 - .../ui/actions/uploads/PitchShifter.kt | 185 ++++++++++++++++++ .../ui/actions/uploads/VoiceAnonymizer.kt | 135 ++----------- .../ui/actions/uploads/PitchShifterTest.kt | 119 +++++++++++ gradle/libs.versions.toml | 2 - settings.gradle.kts | 1 - 6 files changed, 317 insertions(+), 128 deletions(-) create mode 100644 amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt create mode 100644 amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt diff --git a/amethyst/build.gradle.kts b/amethyst/build.gradle.kts index 22cd9f66ac..47db3d09d8 100644 --- a/amethyst/build.gradle.kts +++ b/amethyst/build.gradle.kts @@ -474,9 +474,6 @@ dependencies { // EXIF metadata stripping implementation(libs.androidx.exifinterface) - // Voice anonymization DSP - implementation(libs.tarsosdsp) - // WebRTC for voice/video calls implementation(libs.stream.webrtc.android) diff --git a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt new file mode 100644 index 0000000000..0295de852b --- /dev/null +++ b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt @@ -0,0 +1,185 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.ui.actions.uploads + +import kotlin.math.PI +import kotlin.math.cos +import kotlin.math.roundToInt + +/** + * Pure-Kotlin pitch shifter for mono float PCM. + * + * Shifts pitch by a frequency [ratio][shift] while preserving the original + * duration, by combining two textbook DSP steps: + * + * 1. **WSOLA time-stretch** (Waveform Similarity Overlap-Add) stretches the + * signal in time by the pitch ratio without changing its pitch. Unlike plain + * overlap-add, WSOLA searches a small window around each analysis frame for + * the offset that best correlates with the natural continuation of the + * previous frame, which avoids the phase-discontinuity "warble" of OLA. + * 2. **Linear resampling** back to the original length raises (or lowers) the + * pitch by the same ratio and restores the duration. + * + * This is a clean-room implementation of the standard WSOLA algorithm; it + * carries no third-party code and keeps the app free of GPL-licensed DSP + * dependencies. Quality is more than sufficient for voice anonymization, whose + * pitch factors are modest (~0.7x–1.5x). + */ +class PitchShifter( + private val sampleRate: Int, + /** Analysis/synthesis window length in samples. ~46 ms at 44.1 kHz. */ + private val frameSize: Int = 2048, + /** Half-radius (in samples) of the WSOLA cross-correlation search. */ + private val seekWindow: Int = 256, +) { + private val synthesisHop = frameSize / 2 + private val overlap = frameSize - synthesisHop + private val window = hannWindow(frameSize) + + /** + * Returns [input] pitch-shifted by [frequencyRatio] with the same length. + * + * @param frequencyRatio output-to-input frequency multiplier. `> 1` raises + * the pitch, `< 1` lowers it, `1.0` is a no-op copy. + */ + fun shift( + input: FloatArray, + frequencyRatio: Double, + ): FloatArray { + if (input.isEmpty()) return FloatArray(0) + if (frequencyRatio == 1.0) return input.copyOf() + + // Stretch in time by the ratio (pitch unchanged, length scaled), then + // resample back to the original length (pitch scaled by the ratio). + val stretched = timeStretch(input, frequencyRatio) + return resampleLinear(stretched, input.size) + } + + /** + * WSOLA time-stretch. Output length is approximately `input.size * stretch`. + */ + private fun timeStretch( + input: FloatArray, + stretch: Double, + ): FloatArray { + if (input.size < frameSize + seekWindow) { + // Too short for the overlap machinery; resampling alone still yields + // the requested length change without crashing on tiny clips. + return resampleLinear(input, (input.size * stretch).roundToInt().coerceAtLeast(1)) + } + + val analysisHop = synthesisHop / stretch + val outLength = (input.size * stretch).roundToInt() + frameSize + val out = FloatArray(outLength) + val norm = FloatArray(outLength) + + // Integer start of the current analysis frame inside the input. + var analysisStart = 0 + // Fractional nominal position; the next frame is sought around it. + var nominal = 0.0 + var synthesisPos = 0 + + while (analysisStart + frameSize < input.size && synthesisPos + frameSize < outLength) { + // Overlap-add the windowed analysis frame at the synthesis position. + for (i in 0 until frameSize) { + val w = window[i] + out[synthesisPos + i] += input[analysisStart + i] * w + norm[synthesisPos + i] += w + } + + synthesisPos += synthesisHop + nominal += analysisHop + + // The samples that should naturally follow what we just wrote: the + // tail of the current frame advanced by one synthesis hop. + val naturalStart = analysisStart + synthesisHop + if (naturalStart + overlap >= input.size) break + + // Search around the nominal next analysis position for the offset + // whose head best correlates with that natural continuation. + val center = nominal.roundToInt() + analysisStart = bestMatchOffset(input, naturalStart, center) + } + + // Normalize where windows overlapped to keep unity gain at the edges. + for (i in out.indices) { + if (norm[i] > 1e-6f) out[i] /= norm[i] + } + // Trim the padding tail to the expected stretched length. + val expected = (input.size * stretch).roundToInt() + return if (expected < out.size) out.copyOf(expected) else out + } + + /** + * Finds, within ±[seekWindow] of [center], the input offset whose + * `overlap`-length head best cross-correlates with the `overlap`-length + * segment starting at [naturalStart]. Returns a bounds-safe offset. + */ + private fun bestMatchOffset( + input: FloatArray, + naturalStart: Int, + center: Int, + ): Int { + val low = (center - seekWindow).coerceAtLeast(0) + val high = (center + seekWindow).coerceAtMost(input.size - frameSize - 1) + if (high <= low) return low.coerceIn(0, input.size - frameSize - 1) + + var bestOffset = low + var bestCorr = Double.NEGATIVE_INFINITY + for (offset in low..high) { + var corr = 0.0 + for (i in 0 until overlap) { + corr += input[naturalStart + i].toDouble() * input[offset + i] + } + if (corr > bestCorr) { + bestCorr = corr + bestOffset = offset + } + } + return bestOffset + } + + /** Linear-interpolation resample of [input] to exactly [targetLength] samples. */ + private fun resampleLinear( + input: FloatArray, + targetLength: Int, + ): FloatArray { + if (targetLength <= 0 || input.isEmpty()) return FloatArray(0) + if (input.size == 1) return FloatArray(targetLength) { input[0] } + + val out = FloatArray(targetLength) + val step = (input.size - 1).toDouble() / (targetLength - 1).coerceAtLeast(1) + for (i in 0 until targetLength) { + val pos = i * step + val idx = pos.toInt() + val frac = pos - idx + val a = input[idx] + val b = if (idx + 1 < input.size) input[idx + 1] else a + out[i] = (a + (b - a) * frac).toFloat() + } + return out + } + + private fun hannWindow(size: Int): FloatArray = + FloatArray(size) { i -> + (0.5 * (1.0 - cos(2.0 * PI * i / (size - 1)))).toFloat() + } +} diff --git a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt index f9fd2fc65a..ea064ce41b 100644 --- a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt +++ b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt @@ -25,13 +25,6 @@ import android.media.MediaCodecInfo import android.media.MediaExtractor import android.media.MediaFormat import android.media.MediaMuxer -import be.tarsos.dsp.AudioDispatcher -import be.tarsos.dsp.AudioEvent -import be.tarsos.dsp.AudioProcessor -import be.tarsos.dsp.WaveformSimilarityBasedOverlapAdd -import be.tarsos.dsp.io.TarsosDSPAudioFloatConverter -import be.tarsos.dsp.io.TarsosDSPAudioFormat -import be.tarsos.dsp.resample.RateTransposer import com.vitorpamplona.quartz.utils.Log import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.currentCoroutineContext @@ -57,9 +50,10 @@ data class AnonymizedResult( /** * Processes audio files to alter voice characteristics for privacy. * - * Uses TarsosDSP's WSOLA (Waveform Similarity Overlap-Add) algorithm combined with - * rate transposition to shift pitch while preserving duration. Note that in TarsosDSP, - * pitch factors work inversely: factor < 1 raises pitch, factor > 1 lowers pitch. + * Uses the in-house [PitchShifter] (WSOLA time-stretch + resampling) to shift pitch + * while preserving duration. Preset pitch factors work inversely: factor < 1 raises + * pitch, factor > 1 lowers pitch — so the shifter receives the reciprocal as its + * output-to-input frequency ratio. */ class VoiceAnonymizer { companion object { @@ -73,7 +67,7 @@ class VoiceAnonymizer { * * The process involves three stages: * 1. Decode input audio to PCM (0-30% progress) - * 2. Apply pitch shifting with TarsosDSP (30-70% progress) + * 2. Apply pitch shifting with [PitchShifter] (30-70% progress) * 3. Encode processed audio to AAC (70-100% progress) * * @param inputFile Source audio file (supports formats decodable by MediaCodec) @@ -102,7 +96,7 @@ class VoiceAnonymizer { } val processedPcm = - processPcmWithTarsos(pcmData, preset, sampleRate) { progress -> + applyPitchShift(pcmData, preset, sampleRate) { progress -> onProgress(0.3f + progress * 0.4f) } @@ -292,7 +286,7 @@ class VoiceAnonymizer { release() } - private fun processPcmWithTarsos( + private fun applyPitchShift( pcmData: FloatArray, preset: VoicePreset, sampleRate: Int, @@ -311,75 +305,13 @@ class VoiceAnonymizer { baseFactor } } - val totalSamples = pcmData.size - val processedSamples = ArrayList(totalSamples) - val wsola = - WaveformSimilarityBasedOverlapAdd( - WaveformSimilarityBasedOverlapAdd.Parameters.musicDefaults( - factor, - sampleRate.toDouble(), - ), - ) - val rateTransposer = RateTransposer(factor) - - val bufferSize = wsola.inputBufferSize - val overlap = wsola.overlap - - val tarsosDspFormat = - TarsosDSPAudioFormat( - sampleRate.toFloat(), - 16, - 1, - true, - false, - ) - - val collector = - object : AudioProcessor { - override fun process(audioEvent: AudioEvent): Boolean { - val buffer = audioEvent.floatBuffer - for (i in 0 until audioEvent.bufferSize) { - processedSamples.add(buffer[i]) - } - return true - } - - override fun processingFinished() { - // No-op: no cleanup needed - } - } - - val dispatcher = - AudioDispatcher( - FloatArrayAudioInputStream(pcmData, tarsosDspFormat, pcmData.size.toLong()), - bufferSize, - overlap, - ) - - wsola.setDispatcher(dispatcher) - dispatcher.addAudioProcessor(wsola) - dispatcher.addAudioProcessor(rateTransposer) - dispatcher.addAudioProcessor(collector) - - var samplesProcessed = 0 - val progressProcessor = - object : AudioProcessor { - override fun process(audioEvent: AudioEvent): Boolean { - samplesProcessed += audioEvent.bufferSize - onProgress((samplesProcessed.toFloat() / totalSamples).coerceIn(0f, 1f)) - return true - } - - override fun processingFinished() { - // No-op: no cleanup needed - } - } - dispatcher.addAudioProcessor(progressProcessor) - - dispatcher.run() - - return processedSamples.toFloatArray() + // Presets express factor inversely (factor > 1 lowers pitch), so the + // output-to-input frequency ratio is the reciprocal. + val frequencyRatio = 1.0 / factor + val shifted = PitchShifter(sampleRate).shift(pcmData, frequencyRatio) + onProgress(1f) + return shifted } private fun extractWaveform( @@ -559,44 +491,3 @@ class VoiceAnonymizer { release() } } - -private class FloatArrayAudioInputStream( - private val floatArray: FloatArray, - private val format: TarsosDSPAudioFormat, - private val frameLength: Long, -) : be.tarsos.dsp.io.TarsosDSPAudioInputStream { - private var position = 0 - - override fun getFormat(): TarsosDSPAudioFormat = format - - override fun getFrameLength(): Long = frameLength - - override fun read( - buffer: ByteArray, - offset: Int, - length: Int, - ): Int { - val converter = TarsosDSPAudioFloatConverter.getConverter(format) - val floatBuffer = FloatArray(length / 2) - val samplesToRead = minOf(floatBuffer.size, floatArray.size - position) - - if (samplesToRead <= 0) return -1 - - System.arraycopy(floatArray, position, floatBuffer, 0, samplesToRead) - position += samplesToRead - - converter.toByteArray(floatBuffer, samplesToRead, buffer, offset) - return samplesToRead * 2 - } - - override fun skip(bytesToSkip: Long): Long { - val samplesToSkip = (bytesToSkip / 2).toInt() - val actualSkip = minOf(samplesToSkip, floatArray.size - position) - position += actualSkip - return actualSkip.toLong() * 2 - } - - override fun close() { - // No-op: no cleanup needed - } -} diff --git a/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt b/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt new file mode 100644 index 0000000000..2b5239e4d9 --- /dev/null +++ b/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt @@ -0,0 +1,119 @@ +/* + * Copyright (c) 2025 Vitor Pamplona + * + * Permission is hereby granted, free of charge, to any person obtaining a copy of + * this software and associated documentation files (the "Software"), to deal in + * the Software without restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell copies of the + * Software, and to permit persons to whom the Software is furnished to do so, + * subject to the following conditions: + * + * The above copyright notice and this permission notice shall be included in all + * copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR + * IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS + * FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR + * COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN + * AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION + * WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. + */ +package com.vitorpamplona.amethyst.ui.actions.uploads + +import org.junit.Assert.assertEquals +import org.junit.Assert.assertTrue +import org.junit.Test +import kotlin.math.PI +import kotlin.math.sin + +class PitchShifterTest { + private val sampleRate = 44100 + + /** One second of a clean mono sine at [freq] Hz, amplitude 0.8. */ + private fun sine( + freq: Double, + seconds: Double = 1.0, + ): FloatArray { + val n = (sampleRate * seconds).toInt() + return FloatArray(n) { i -> (0.8 * sin(2.0 * PI * freq * i / sampleRate)).toFloat() } + } + + /** + * Estimates the fundamental frequency of a (near-)periodic signal via + * autocorrelation over a stable middle window. Dependency-free and robust + * for sine inputs. + */ + private fun estimateFrequency(signal: FloatArray): Double { + // Use the central 60% to avoid edge ramp-up/ramp-down artifacts. + val start = (signal.size * 0.2).toInt() + val end = (signal.size * 0.8).toInt() + val chunk = signal.copyOfRange(start, end) + + val minFreq = 50.0 + val maxFreq = 2000.0 + val minLag = (sampleRate / maxFreq).toInt() + val maxLag = (sampleRate / minFreq).toInt().coerceAtMost(chunk.size - 1) + + var bestLag = minLag + var bestCorr = Double.NEGATIVE_INFINITY + for (lag in minLag..maxLag) { + var corr = 0.0 + for (i in 0 until chunk.size - lag) { + corr += chunk[i] * chunk[i + lag] + } + if (corr > bestCorr) { + bestCorr = corr + bestLag = lag + } + } + return sampleRate.toDouble() / bestLag + } + + @Test + fun `ratio of one preserves length`() { + val input = sine(440.0) + val output = PitchShifter(sampleRate).shift(input, 1.0) + // Duration must be preserved (within a frame of tolerance). + assertEquals(input.size.toDouble(), output.size.toDouble(), 4096.0) + } + + @Test + fun `shift up one octave roughly doubles the fundamental`() { + val output = PitchShifter(sampleRate).shift(sine(440.0), 2.0) + val measured = estimateFrequency(output) + assertEquals(880.0, measured, 880.0 * 0.06) + } + + @Test + fun `shift down one octave roughly halves the fundamental`() { + val output = PitchShifter(sampleRate).shift(sine(440.0), 0.5) + val measured = estimateFrequency(output) + assertEquals(220.0, measured, 220.0 * 0.06) + } + + @Test + fun `preserves duration for the deep preset ratio`() { + // DEEP preset lowers pitch: frequencyRatio = 1 / 1.4. + val input = sine(300.0) + val output = PitchShifter(sampleRate).shift(input, 1.0 / 1.4) + assertEquals(input.size.toDouble(), output.size.toDouble(), 4096.0) + val measured = estimateFrequency(output) + assertEquals(300.0 / 1.4, measured, (300.0 / 1.4) * 0.06) + } + + @Test + fun `preserves duration for the high preset ratio`() { + // HIGH preset raises pitch: frequencyRatio = 1 / 0.75. + val input = sine(300.0) + val output = PitchShifter(sampleRate).shift(input, 1.0 / 0.75) + assertEquals(input.size.toDouble(), output.size.toDouble(), 4096.0) + val measured = estimateFrequency(output) + assertEquals(300.0 / 0.75, measured, (300.0 / 0.75) * 0.06) + } + + @Test + fun `empty input returns empty output`() { + val output = PitchShifter(sampleRate).shift(FloatArray(0), 1.5) + assertTrue(output.isEmpty()) + } +} diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index bc655df85e..d1ad815bc7 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -60,7 +60,6 @@ securityCryptoKtx = "1.1.0" slf4j = "2.0.18" spotless = "8.6.0" streamWebrtcAndroid = "1.3.10" -tarsosdsp = "2.5" translate = "17.0.3" jetbrainsCompose = "1.11.0" unifiedpush = "3.0.10" @@ -198,7 +197,6 @@ secp256k1-kmp-jni-android = { group = "fr.acinq.secp256k1", name = "secp256k1-km secp256k1-kmp-jni-jvm = { group = "fr.acinq.secp256k1", name = "secp256k1-kmp-jni-jvm", version.ref = "secp256k1KmpJniAndroid" } schnorr256k1-kmp = { group = "com.vitorpamplona.schnorr256k1", name = "schnorr256k1-kmp", version.ref = "schnorr256k1Kmp" } stream-webrtc-android = { group = "io.getstream", name = "stream-webrtc-android", version.ref = "streamWebrtcAndroid" } -tarsosdsp = { group = "be.tarsos.dsp", name = "core", version.ref = "tarsosdsp" } unifiedpush = { group = "com.github.UnifiedPush", name = "android-connector", version.ref = "unifiedpush" } play-services-cast-framework = { group = "com.google.android.gms", name = "play-services-cast-framework", version.ref = "playServicesCast" } vico-charts-compose = { group = "com.patrykandpatrick.vico", name = "compose", version.ref = "vico-charts-compose" } diff --git a/settings.gradle.kts b/settings.gradle.kts index eb3c2a3768..f30ae000ba 100644 --- a/settings.gradle.kts +++ b/settings.gradle.kts @@ -26,7 +26,6 @@ dependencyResolutionManagement { mavenCentral() maven { url = uri("https://jitpack.io") } maven { url = uri("https://raw.githubusercontent.com/guardianproject/gpmaven/master") } - maven { url = uri("https://mvn.0110.be/releases") } } } From 66c345717c5bfc56f74f3726da11750798cdbae8 Mon Sep 17 00:00:00 2001 From: davotoula Date: Tue, 9 Jun 2026 21:28:20 +0200 Subject: [PATCH 2/3] Code review fixes: - honor cancellation in voice anonymizer - reuse shared Hann window and drop vestigial progress callback --- .../ui/actions/uploads/PitchShifter.kt | 11 +---- .../ui/actions/uploads/VoiceAnonymizer.kt | 48 ++++++++++++------- .../ui/actions/uploads/PitchShifterTest.kt | 12 ++--- 3 files changed, 38 insertions(+), 33 deletions(-) diff --git a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt index 0295de852b..cb620f8294 100644 --- a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt +++ b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifter.kt @@ -20,8 +20,7 @@ */ package com.vitorpamplona.amethyst.ui.actions.uploads -import kotlin.math.PI -import kotlin.math.cos +import com.vitorpamplona.amethyst.commons.audio.AudioWindow import kotlin.math.roundToInt /** @@ -44,7 +43,6 @@ import kotlin.math.roundToInt * pitch factors are modest (~0.7x–1.5x). */ class PitchShifter( - private val sampleRate: Int, /** Analysis/synthesis window length in samples. ~46 ms at 44.1 kHz. */ private val frameSize: Int = 2048, /** Half-radius (in samples) of the WSOLA cross-correlation search. */ @@ -52,7 +50,7 @@ class PitchShifter( ) { private val synthesisHop = frameSize / 2 private val overlap = frameSize - synthesisHop - private val window = hannWindow(frameSize) + private val window = AudioWindow.hann(frameSize) /** * Returns [input] pitch-shifted by [frequencyRatio] with the same length. @@ -177,9 +175,4 @@ class PitchShifter( } return out } - - private fun hannWindow(size: Int): FloatArray = - FloatArray(size) { i -> - (0.5 * (1.0 - cos(2.0 * PI * i / (size - 1)))).toFloat() - } } diff --git a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt index ea064ce41b..0a53b6412e 100644 --- a/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt +++ b/amethyst/src/main/java/com/vitorpamplona/amethyst/ui/actions/uploads/VoiceAnonymizer.kt @@ -26,11 +26,13 @@ import android.media.MediaExtractor import android.media.MediaFormat import android.media.MediaMuxer import com.vitorpamplona.quartz.utils.Log +import kotlinx.coroutines.CancellationException import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.currentCoroutineContext import kotlinx.coroutines.isActive import kotlinx.coroutines.withContext import java.io.File +import java.nio.ByteBuffer import java.nio.ByteOrder import kotlin.math.abs @@ -95,10 +97,8 @@ class VoiceAnonymizer { onProgress(progress * 0.3f) } - val processedPcm = - applyPitchShift(pcmData, preset, sampleRate) { progress -> - onProgress(0.3f + progress * 0.4f) - } + val processedPcm = applyPitchShift(pcmData, preset) + onProgress(0.7f) val waveform = extractWaveform(processedPcm, sampleRate) @@ -108,6 +108,10 @@ class VoiceAnonymizer { onProgress(1f) Result.success(AnonymizedResult(outputFile, waveform, duration)) + } catch (e: CancellationException) { + // Let cancellation propagate so structured concurrency works (e.g. the + // upload screen being dismissed cancels this job). + throw e } catch (e: Exception) { Log.e(TAG, "Failed to anonymize audio", e) Result.failure(e) @@ -170,7 +174,10 @@ class VoiceAnonymizer { val inputBufferIndex = decoder.dequeueInputBuffer(10000) if (inputBufferIndex < 0) return - val inputBuffer = decoder.getInputBuffer(inputBufferIndex)!! + val inputBuffer = + requireNotNull(decoder.getInputBuffer(inputBufferIndex)) { + "Decoder input buffer $inputBufferIndex was null (codec in async mode?)" + } val sampleSize = extractor.readSampleData(inputBuffer, 0) if (sampleSize < 0) { @@ -226,7 +233,10 @@ class VoiceAnonymizer { } private fun extractPcmSamples(outputBufferIndex: Int) { - val outputBuffer = decoder.getOutputBuffer(outputBufferIndex)!! + val outputBuffer = + requireNotNull(decoder.getOutputBuffer(outputBufferIndex)) { + "Decoder output buffer $outputBufferIndex was null (codec in async mode?)" + } val shortBuffer = outputBuffer.order(ByteOrder.nativeOrder()).asShortBuffer() while (shortBuffer.hasRemaining()) { pcmSamples.add(shortBuffer.get() / 32768f) @@ -289,8 +299,6 @@ class VoiceAnonymizer { private fun applyPitchShift( pcmData: FloatArray, preset: VoicePreset, - sampleRate: Int, - onProgress: (Float) -> Unit, ): FloatArray { val baseFactor = preset.pitchFactor val factor = @@ -309,9 +317,7 @@ class VoiceAnonymizer { // Presets express factor inversely (factor > 1 lowers pitch), so the // output-to-input frequency ratio is the reciprocal. val frequencyRatio = 1.0 / factor - val shifted = PitchShifter(sampleRate).shift(pcmData, frequencyRatio) - onProgress(1f) - return shifted + return PitchShifter().shift(pcmData, frequencyRatio) } private fun extractWaveform( @@ -353,7 +359,10 @@ class VoiceAnonymizer { val inputBufferIndex = encoder.dequeueInputBuffer(10000) if (inputBufferIndex < 0) return - val inputBuffer = encoder.getInputBuffer(inputBufferIndex)!! + val inputBuffer = + requireNotNull(encoder.getInputBuffer(inputBufferIndex)) { + "Encoder input buffer $inputBufferIndex was null (codec in async mode?)" + } inputBuffer.clear() val samplesToWrite = minOf((inputBuffer.capacity() / 2), pcmData.size - inputOffset) @@ -377,7 +386,7 @@ class VoiceAnonymizer { private fun queueSampleData( inputBufferIndex: Int, - inputBuffer: java.nio.ByteBuffer, + inputBuffer: ByteBuffer, samplesToWrite: Int, ) { writePcmSamplesToBuffer(inputBuffer, samplesToWrite) @@ -388,7 +397,7 @@ class VoiceAnonymizer { } private fun writePcmSamplesToBuffer( - inputBuffer: java.nio.ByteBuffer, + inputBuffer: ByteBuffer, samplesToWrite: Int, ) { for (i in 0 until samplesToWrite) { @@ -429,13 +438,16 @@ class VoiceAnonymizer { } private fun processOutputBuffer(outputBufferIndex: Int) { - val outputBuffer = encoder.getOutputBuffer(outputBufferIndex)!! + val outputBuffer = + requireNotNull(encoder.getOutputBuffer(outputBufferIndex)) { + "Encoder output buffer $outputBufferIndex was null (codec in async mode?)" + } writeToMuxerIfReady(outputBuffer) encoder.releaseOutputBuffer(outputBufferIndex, false) checkForEndOfStream() } - private fun writeToMuxerIfReady(outputBuffer: java.nio.ByteBuffer) { + private fun writeToMuxerIfReady(outputBuffer: ByteBuffer) { if (isMuxerStarted && bufferInfo.size > 0) { outputBuffer.position(bufferInfo.offset) outputBuffer.limit(bufferInfo.offset + bufferInfo.size) @@ -456,7 +468,7 @@ class VoiceAnonymizer { setInteger(MediaFormat.KEY_BIT_RATE, BIT_RATE) } - private fun encodePcmToAac( + private suspend fun encodePcmToAac( pcmData: FloatArray, sampleRate: Int, outputFile: File, @@ -473,7 +485,7 @@ class VoiceAnonymizer { val inputFeeder = EncoderInputFeeder(encoder, pcmData, sampleRate, onProgress) val outputDrainer = EncoderOutputDrainer(encoder, muxer) - while (!outputDrainer.isDone) { + while (!outputDrainer.isDone && currentCoroutineContext().isActive) { inputFeeder.feedInput() outputDrainer.drainOutput() } diff --git a/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt b/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt index 2b5239e4d9..318f7690b5 100644 --- a/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt +++ b/amethyst/src/test/java/com/vitorpamplona/amethyst/ui/actions/uploads/PitchShifterTest.kt @@ -72,21 +72,21 @@ class PitchShifterTest { @Test fun `ratio of one preserves length`() { val input = sine(440.0) - val output = PitchShifter(sampleRate).shift(input, 1.0) + val output = PitchShifter().shift(input, 1.0) // Duration must be preserved (within a frame of tolerance). assertEquals(input.size.toDouble(), output.size.toDouble(), 4096.0) } @Test fun `shift up one octave roughly doubles the fundamental`() { - val output = PitchShifter(sampleRate).shift(sine(440.0), 2.0) + val output = PitchShifter().shift(sine(440.0), 2.0) val measured = estimateFrequency(output) assertEquals(880.0, measured, 880.0 * 0.06) } @Test fun `shift down one octave roughly halves the fundamental`() { - val output = PitchShifter(sampleRate).shift(sine(440.0), 0.5) + val output = PitchShifter().shift(sine(440.0), 0.5) val measured = estimateFrequency(output) assertEquals(220.0, measured, 220.0 * 0.06) } @@ -95,7 +95,7 @@ class PitchShifterTest { fun `preserves duration for the deep preset ratio`() { // DEEP preset lowers pitch: frequencyRatio = 1 / 1.4. val input = sine(300.0) - val output = PitchShifter(sampleRate).shift(input, 1.0 / 1.4) + val output = PitchShifter().shift(input, 1.0 / 1.4) assertEquals(input.size.toDouble(), output.size.toDouble(), 4096.0) val measured = estimateFrequency(output) assertEquals(300.0 / 1.4, measured, (300.0 / 1.4) * 0.06) @@ -105,7 +105,7 @@ class PitchShifterTest { fun `preserves duration for the high preset ratio`() { // HIGH preset raises pitch: frequencyRatio = 1 / 0.75. val input = sine(300.0) - val output = PitchShifter(sampleRate).shift(input, 1.0 / 0.75) + val output = PitchShifter().shift(input, 1.0 / 0.75) assertEquals(input.size.toDouble(), output.size.toDouble(), 4096.0) val measured = estimateFrequency(output) assertEquals(300.0 / 0.75, measured, (300.0 / 0.75) * 0.06) @@ -113,7 +113,7 @@ class PitchShifterTest { @Test fun `empty input returns empty output`() { - val output = PitchShifter(sampleRate).shift(FloatArray(0), 1.5) + val output = PitchShifter().shift(FloatArray(0), 1.5) assertTrue(output.isEmpty()) } } From 26fc0ea7bfd2f500f37b1a402fdb46501187387e Mon Sep 17 00:00:00 2001 From: davotoula Date: Tue, 9 Jun 2026 22:10:19 +0200 Subject: [PATCH 3/3] docs(claude): require a license check when adding any dependency --- .claude/CLAUDE.md | 31 +++++++++++++++++++++++++++++++ 1 file changed, 31 insertions(+) diff --git a/.claude/CLAUDE.md b/.claude/CLAUDE.md index c080d01670..0a55a49f52 100644 --- a/.claude/CLAUDE.md +++ b/.claude/CLAUDE.md @@ -189,6 +189,37 @@ version. `quartz/` is protocol-only — no composables. ./gradlew spotlessApply ``` +## Dependency Licensing + +**MANDATORY whenever you introduce a new third-party dependency** — in *any* +module (`quartz`, `commons`, `amethyst`, `desktopApp`, `cli`, `quic`, +`nestsClient`, …), whether you add it to `gradle/libs.versions.toml` or to a +module's `build.gradle.kts`: determine its license **before** wiring it in. +Amethyst ships under the **MIT** license, so a copyleft dependency linked into a +distributed artifact (APK, desktop binary) can force that artifact's +combined-work terms onto the whole project. + +Verify against the dependency's actual `LICENSE`/`COPYING` file or its published +POM — **not from memory**. Then classify and act: + +- **Permissive** (MIT, Apache-2.0, BSD, ISC, MPL-2.0, zlib, …) → **OK**, + proceed. +- **LGPL, or GPL/EPL with a linking / Classpath exception** → **WARN.** + Acceptable to link (the exception keeps our own code MIT), but call it out in + your summary so the human knows. Confirm the exception actually exists in the + LICENSE text — don't assume it does. +- **Stricter than LGPL** — GPL/AGPL **without** a linking exception, SSPL, + proprietary/commercial-only, or anything where the linking-exception check is + "no" → **STRONGLY WARN and STOP.** Do not add it silently. Surface it + prominently and **require an explicit call-out in the PR description** so a + maintainer makes the decision. Prefer a permissive alternative, a clean-room + implementation, or dropping the feature. + +For any GPL-family hit the decisive question is always **"is there a linking +(LGPL/Classpath) exception?"** — that is what separates a WARN from a STOP. +(Example: TarsosDSP, GPLv3 with no exception, was removed from `amethyst` and +replaced with an in-house pitch shifter for exactly this reason.) + ## Quartz KMP Structure Quartz uses expect/actual for platform-specific implementations (e.g. crypto