fix(cw): stop the transcript stalling while audio waits to be archived
The history box appeared to delete text. Audio leaving the 20 s live window is decoded into the archive only once a full 15 s batch has accumulated, so until then its characters were in neither place: not in the live decode, which had scrolled past them, and not in the history, which had not seen them yet. Measured on a 20 WPM timeline, the concatenated transcript held at 40 characters from t=24 s to t=34.5 s - eleven seconds of no growth - then jumped to 70 when the batch flushed. Up to 30 characters sat in that gap. Reading it as deletion is reasonable; the text really was missing from the box. The pending batch is now decoded too, on the same 1.5 s cycle as the live window, and shown as a provisional tail after the committed text. The final archive decode replaces it, having the whole batch for context. The transcript is monotonic afterwards: +3 characters every cycle with no stalls. Decoding each 100 ms capture chunk instead would have removed the gap entirely but measured 14x the inference load - over 250% of one core across ten minutes - and a chunk that short carries under two dot-lengths of context, so the decode would be poor as well as expensive. One extra inference per redecode cycle costs 24.7% against 18.3%. Discarding buffered audio drops the provisional text with it, since that text describes audio that no longer exists. Committed text stays: it was correct for audio that really was archived.
This commit is contained in:
1 parent
07df22d287
commit
828fd0fb6f
1 file changed
+42
-3
@@ -103,9 +103,24 @@ class CwDeepDecoder(
|
|||||||
private val _decodedText = MutableStateFlow("")
|
private val _decodedText = MutableStateFlow("")
|
||||||
override val decodedText: StateFlow<String> = _decodedText.asStateFlow()
|
override val decodedText: StateFlow<String> = _decodedText.asStateFlow()
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Archived text, plus a provisional decode of audio not yet archived.
|
||||||
|
*
|
||||||
|
* Kept as one flow of the two parts concatenated. Without the provisional part the
|
||||||
|
* transcript visibly shrank: audio leaving the 20 s window waits for a full
|
||||||
|
* [ARCHIVE_SECONDS] batch before it is decoded into the archive, so for up to 15 s its
|
||||||
|
* characters were in neither place - measured, up to 30 characters at 20 WPM would
|
||||||
|
* vanish and reappear later, which reads as the box deleting text.
|
||||||
|
*/
|
||||||
private val _historyText = MutableStateFlow("")
|
private val _historyText = MutableStateFlow("")
|
||||||
override val historyText: StateFlow<String> = _historyText.asStateFlow()
|
override val historyText: StateFlow<String> = _historyText.asStateFlow()
|
||||||
|
|
||||||
|
/** Permanently archived text; the provisional tail is appended to this for display. */
|
||||||
|
private var committedText = ""
|
||||||
|
|
||||||
|
/** Provisional decode of the pending archive batch, replaced when it is archived. */
|
||||||
|
private var pendingText = ""
|
||||||
|
|
||||||
private val _estimatedPitch = MutableStateFlow<Float?>(null)
|
private val _estimatedPitch = MutableStateFlow<Float?>(null)
|
||||||
override val estimatedPitch: StateFlow<Float?> = _estimatedPitch.asStateFlow()
|
override val estimatedPitch: StateFlow<Float?> = _estimatedPitch.asStateFlow()
|
||||||
|
|
||||||
@@ -373,6 +388,10 @@ class CwDeepDecoder(
|
|||||||
private fun dropBufferedAudio() {
|
private fun dropBufferedAudio() {
|
||||||
buffer.reset()
|
buffer.reset()
|
||||||
archiveSize = 0
|
archiveSize = 0
|
||||||
|
// The provisional text describes audio being discarded, so it goes with it.
|
||||||
|
// Committed text stays: it was correct for audio that really was archived.
|
||||||
|
pendingText = ""
|
||||||
|
_historyText.value = committedText
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -467,9 +486,27 @@ class CwDeepDecoder(
|
|||||||
if (audio.size < CwDeepSpectrogram.FFT_LENGTH) return@withContext
|
if (audio.size < CwDeepSpectrogram.FFT_LENGTH) return@withContext
|
||||||
val spectrogram = CwDeepSpectrogram.compute(audio)
|
val spectrogram = CwDeepSpectrogram.compute(audio)
|
||||||
val text = runInference(activeSession, activeEnvironment, spectrogram)
|
val text = runInference(activeSession, activeEnvironment, spectrogram)
|
||||||
if (text.isNotEmpty()) {
|
// This batch is final, so its provisional decode is superseded rather than kept.
|
||||||
_historyText.value += text
|
committedText += text
|
||||||
}
|
pendingText = ""
|
||||||
|
_historyText.value = committedText
|
||||||
|
}
|
||||||
|
|
||||||
|
/**
|
||||||
|
* Decode the audio waiting to be archived, so it stays on screen until it is.
|
||||||
|
*
|
||||||
|
* Provisional: the batch is still growing, and the final decode sees all of it at once
|
||||||
|
* with more context. Runs on the redecode cycle rather than per capture chunk -
|
||||||
|
* decoding every 100 ms chunk separately measured 14x the inference load, over 250% of
|
||||||
|
* one core, and a chunk that short carries under two dot-lengths of context anyway.
|
||||||
|
*/
|
||||||
|
private suspend fun decodePending(audio: FloatArray) = withContext(Dispatchers.Default) {
|
||||||
|
val activeSession = session ?: return@withContext
|
||||||
|
val activeEnvironment = environment ?: return@withContext
|
||||||
|
if (audio.size < CwDeepSpectrogram.FFT_LENGTH) return@withContext
|
||||||
|
val spectrogram = CwDeepSpectrogram.compute(audio)
|
||||||
|
pendingText = runInference(activeSession, activeEnvironment, spectrogram)
|
||||||
|
_historyText.value = committedText + pendingText
|
||||||
}
|
}
|
||||||
|
|
||||||
/** Run the ONNX model over a pre-computed spectrogram and return the decoded text. */
|
/** Run the ONNX model over a pre-computed spectrogram and return the decoded text. */
|
||||||
@@ -550,6 +587,8 @@ class CwDeepDecoder(
|
|||||||
buffer.reset()
|
buffer.reset()
|
||||||
_decodedText.value = ""
|
_decodedText.value = ""
|
||||||
_historyText.value = ""
|
_historyText.value = ""
|
||||||
|
committedText = ""
|
||||||
|
pendingText = ""
|
||||||
archiveSize = 0
|
archiveSize = 0
|
||||||
_estimatedPitch.value = null
|
_estimatedPitch.value = null
|
||||||
_detectedToneHz.value = null
|
_detectedToneHz.value = null
|
||||||
|
|||||||
Reference in new issue
Block a user