- C1: stop feeding the unconsumed barge-in TTSQueue in the reply path (deadlocked after 64 chunks on a real long reply) - C2: encode reply Opus at the TTS/codec sample rate (24 kHz), not hardcoded 16 kHz; encoder now rate-generic - M3: bound the mic PCM buffer to the 60 s max utterance - M4: tts_sentence_start per speakable sentence, not per LLM token - Minor: send tts.stop on a failed turn; constant-time token compare - Sec: untrack + gitignore config.yaml; add config.example.yaml with placeholder secrets - Tests: 23 pass (2 new regressions: long-reply deadlock, rate-generic encoder)
45 lines
1.4 KiB
Python
45 lines
1.4 KiB
Python
"""In-memory PCM ring buffer for a single utterance.
|
|
|
|
Kept deliberately dependency-free (REQ-002). The buffer accumulates the
|
|
current utterance's PCM and is flushed at end-of-speech, so STT always
|
|
transcribes one utterance, not a rolling window.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
|
|
class PCMBuffer:
|
|
"""In-memory PCM ring for the current utterance (bounded).
|
|
|
|
``max_bytes`` caps the buffer: when it would grow past the cap, the
|
|
oldest audio is dropped so a quiet room that never latches speech cannot
|
|
accumulate unbounded (review Major 3). A real utterance flushes at
|
|
end-of-speech, long before the cap.
|
|
"""
|
|
|
|
def __init__(self, max_bytes: int | None = None) -> None:
|
|
self._chunks: list[bytes] = []
|
|
self._bytes = 0
|
|
self.max_bytes = max_bytes
|
|
|
|
def feed(self, pcm: bytes) -> None:
|
|
self._chunks.append(bytes(pcm))
|
|
self._bytes += len(pcm)
|
|
if self.max_bytes is not None and self._bytes > self.max_bytes:
|
|
overflow = self._bytes - self.max_bytes
|
|
while self._chunks and overflow > 0:
|
|
head = self._chunks.pop(0)
|
|
overflow -= len(head)
|
|
self._bytes -= len(head)
|
|
|
|
def total_bytes(self) -> int:
|
|
return self._bytes
|
|
|
|
def get(self) -> bytes:
|
|
data = b"".join(self._chunks)
|
|
self.clear()
|
|
return data
|
|
|
|
def clear(self) -> None:
|
|
self._chunks = []
|
|
self._bytes = 0
|