Files
esp32-server/app/audio/buffer.py
kunthawat e96498b67e fix: resolve fresh-eyes review criticals + majors on the audio path
- C1: stop feeding the unconsumed barge-in TTSQueue in the reply path (deadlocked after 64 chunks on a real long reply)
- C2: encode reply Opus at the TTS/codec sample rate (24 kHz), not hardcoded 16 kHz; encoder now rate-generic
- M3: bound the mic PCM buffer to the 60 s max utterance
- M4: tts_sentence_start per speakable sentence, not per LLM token
- Minor: send tts.stop on a failed turn; constant-time token compare
- Sec: untrack + gitignore config.yaml; add config.example.yaml with placeholder secrets
- Tests: 23 pass (2 new regressions: long-reply deadlock, rate-generic encoder)
2026-10-03 21:55:32 +07:00

45 lines
1.4 KiB
Python

"""In-memory PCM ring buffer for a single utterance.
Kept deliberately dependency-free (REQ-002). The buffer accumulates the
current utterance's PCM and is flushed at end-of-speech, so STT always
transcribes one utterance, not a rolling window.
"""
from __future__ import annotations
class PCMBuffer:
"""In-memory PCM ring for the current utterance (bounded).
``max_bytes`` caps the buffer: when it would grow past the cap, the
oldest audio is dropped so a quiet room that never latches speech cannot
accumulate unbounded (review Major 3). A real utterance flushes at
end-of-speech, long before the cap.
"""
def __init__(self, max_bytes: int | None = None) -> None:
self._chunks: list[bytes] = []
self._bytes = 0
self.max_bytes = max_bytes
def feed(self, pcm: bytes) -> None:
self._chunks.append(bytes(pcm))
self._bytes += len(pcm)
if self.max_bytes is not None and self._bytes > self.max_bytes:
overflow = self._bytes - self.max_bytes
while self._chunks and overflow > 0:
head = self._chunks.pop(0)
overflow -= len(head)
self._bytes -= len(head)
def total_bytes(self) -> int:
return self._bytes
def get(self) -> bytes:
data = b"".join(self._chunks)
self.clear()
return data
def clear(self) -> None:
self._chunks = []
self._bytes = 0