From 828fd0fb6f5b89954538491862320b131ee79a00 Mon Sep 17 00:00:00 2001 From: QIU Date: Sun, 23 Aug 2026 11:40:22 +0000 Subject: [PATCH] fix(cw): stop the transcript stalling while audio waits to be archived The history box appeared to delete text. Audio leaving the 20 s live window is decoded into the archive only once a full 15 s batch has accumulated, so until then its characters were in neither place: not in the live decode, which had scrolled past them, and not in the history, which had not seen them yet. Measured on a 20 WPM timeline, the concatenated transcript held at 40 characters from t=24 s to t=34.5 s - eleven seconds of no growth - then jumped to 70 when the batch flushed. Up to 30 characters sat in that gap. Reading it as deletion is reasonable; the text really was missing from the box. The pending batch is now decoded too, on the same 1.5 s cycle as the live window, and shown as a provisional tail after the committed text. The final archive decode replaces it, having the whole batch for context. The transcript is monotonic afterwards: +3 characters every cycle with no stalls. Decoding each 100 ms capture chunk instead would have removed the gap entirely but measured 14x the inference load - over 250% of one core across ten minutes - and a chunk that short carries under two dot-lengths of context, so the decode would be poor as well as expensive. One extra inference per redecode cycle costs 24.7% against 18.3%. Discarding buffered audio drops the provisional text with it, since that text describes audio that no longer exists. Committed text stays: it was correct for audio that really was archived. --- .../look4sat/core/data/cw/CwDeepDecoder.kt | 45 +++++++++++++++++-- 1 file changed, 42 insertions(+), 3 deletions(-) diff --git a/core/data/src/main/java/com/rtbishop/look4sat/core/data/cw/CwDeepDecoder.kt b/core/data/src/main/java/com/rtbishop/look4sat/core/data/cw/CwDeepDecoder.kt index 2a0fd270..e8ea53bb 100644 --- a/core/data/src/main/java/com/rtbishop/look4sat/core/data/cw/CwDeepDecoder.kt +++ b/core/data/src/main/java/com/rtbishop/look4sat/core/data/cw/CwDeepDecoder.kt @@ -103,9 +103,24 @@ class CwDeepDecoder( private val _decodedText = MutableStateFlow("") override val decodedText: StateFlow = _decodedText.asStateFlow() + /** + * Archived text, plus a provisional decode of audio not yet archived. + * + * Kept as one flow of the two parts concatenated. Without the provisional part the + * transcript visibly shrank: audio leaving the 20 s window waits for a full + * [ARCHIVE_SECONDS] batch before it is decoded into the archive, so for up to 15 s its + * characters were in neither place - measured, up to 30 characters at 20 WPM would + * vanish and reappear later, which reads as the box deleting text. + */ private val _historyText = MutableStateFlow("") override val historyText: StateFlow = _historyText.asStateFlow() + /** Permanently archived text; the provisional tail is appended to this for display. */ + private var committedText = "" + + /** Provisional decode of the pending archive batch, replaced when it is archived. */ + private var pendingText = "" + private val _estimatedPitch = MutableStateFlow(null) override val estimatedPitch: StateFlow = _estimatedPitch.asStateFlow() @@ -373,6 +388,10 @@ class CwDeepDecoder( private fun dropBufferedAudio() { buffer.reset() archiveSize = 0 + // The provisional text describes audio being discarded, so it goes with it. + // Committed text stays: it was correct for audio that really was archived. + pendingText = "" + _historyText.value = committedText } /** @@ -467,9 +486,27 @@ class CwDeepDecoder( if (audio.size < CwDeepSpectrogram.FFT_LENGTH) return@withContext val spectrogram = CwDeepSpectrogram.compute(audio) val text = runInference(activeSession, activeEnvironment, spectrogram) - if (text.isNotEmpty()) { - _historyText.value += text - } + // This batch is final, so its provisional decode is superseded rather than kept. + committedText += text + pendingText = "" + _historyText.value = committedText + } + + /** + * Decode the audio waiting to be archived, so it stays on screen until it is. + * + * Provisional: the batch is still growing, and the final decode sees all of it at once + * with more context. Runs on the redecode cycle rather than per capture chunk - + * decoding every 100 ms chunk separately measured 14x the inference load, over 250% of + * one core, and a chunk that short carries under two dot-lengths of context anyway. + */ + private suspend fun decodePending(audio: FloatArray) = withContext(Dispatchers.Default) { + val activeSession = session ?: return@withContext + val activeEnvironment = environment ?: return@withContext + if (audio.size < CwDeepSpectrogram.FFT_LENGTH) return@withContext + val spectrogram = CwDeepSpectrogram.compute(audio) + pendingText = runInference(activeSession, activeEnvironment, spectrogram) + _historyText.value = committedText + pendingText } /** Run the ONNX model over a pre-computed spectrogram and return the decoded text. */ @@ -550,6 +587,8 @@ class CwDeepDecoder( buffer.reset() _decodedText.value = "" _historyText.value = "" + committedText = "" + pendingText = "" archiveSize = 0 _estimatedPitch.value = null _detectedToneHz.value = null