- C1: stop feeding the unconsumed barge-in TTSQueue in the reply path (deadlocked after 64 chunks on a real long reply) - C2: encode reply Opus at the TTS/codec sample rate (24 kHz), not hardcoded 16 kHz; encoder now rate-generic - M3: bound the mic PCM buffer to the 60 s max utterance - M4: tts_sentence_start per speakable sentence, not per LLM token - Minor: send tts.stop on a failed turn; constant-time token compare - Sec: untrack + gitignore config.yaml; add config.example.yaml with placeholder secrets - Tests: 23 pass (2 new regressions: long-reply deadlock, rate-generic encoder)
68 lines
2.2 KiB
YAML
68 lines
2.2 KiB
YAML
# Hermes-Xiaozhi Bridge configuration.
|
|
# All backends default to "mock" so the bridge runs/tests on any machine
|
|
# (no GPU, no device, no native libs). Point at real endpoints on the
|
|
# voice-server by changing engine + *_endpoint values.
|
|
#
|
|
# Protocol values marked FIRMWARE are PoC assumptions — verify against the
|
|
# real Xiaozhi firmware before locking (CON-004).
|
|
|
|
host: 0.0.0.0 # LAN reachable (device on 192.168.1.x)
|
|
port: 8766
|
|
ws_path: /ws/xiaozhi
|
|
|
|
protocol:
|
|
hello_version: 1 # FIRMWARE
|
|
opus_sample_rate: 16000 # FIRMWARE
|
|
opus_channels: 1 # FIRMWARE
|
|
frame_ms: 20 # FIRMWARE
|
|
|
|
audio:
|
|
opus: real # real = decode device Opus | passthrough = raw PCM (tests)
|
|
vad_min_speech_frames: 5 # 100 ms of speech before an utterance counts
|
|
vad_min_silence_frames: 8 # 160 ms of silence ends the utterance
|
|
vad_max_utterance_frames: 3000 # 60 s safety flush
|
|
|
|
stt:
|
|
engine: mock # mock | typhoon | faster-whisper
|
|
language: th
|
|
sample_rate: 16000
|
|
# typhoon_endpoint: http://127.0.0.1:8888
|
|
# whisper_model: large-v3
|
|
|
|
tts:
|
|
engine: mock # mock | jaitts
|
|
voice: default
|
|
device: cuda:0
|
|
sample_rate: 24000
|
|
# jaitts_endpoint: http://127.0.0.1:8889
|
|
|
|
hermes:
|
|
transport: mock # mock | openai_http
|
|
reasoning: none # REQ-004 — voice profile is reasoning-none
|
|
tools: [session_search] # REQ-004/REQ-008 — read/recall only
|
|
# base_url: http://127.0.0.1:11434/v1
|
|
max_context_tokens: 16000
|
|
max_history_turns: 12
|
|
|
|
security:
|
|
enabled: true
|
|
devices:
|
|
# FIRMWARE: Device-Id header = MAC address (xiaozhi v2.5.0), not "xiaozhi-main"
|
|
- device_id: "aa:bb:cc:dd:ee:ff"
|
|
token_hash: "REPLACE_WITH_sha256_hex_of_your_device_token"
|
|
|
|
# OTA onboarding: the device POSTs to CONFIG_OTA_URL (baked into firmware at
|
|
# build time) and reads websocket.url/token here. MUST match the firmware
|
|
# build's OTA_URL = https://zhi.moreminimore.com/ota
|
|
ota:
|
|
server_ws_url: "wss://zhi.moreminimore.com/ws/xiaozhi"
|
|
server_token: "REPLACE_WITH_YOUR_DEVICE_TOKEN"
|
|
protocol_version: 1
|
|
timezone_minutes: 420 # UTC+07:00
|
|
firmware_version: "2.5.0-custom"
|
|
|
|
gpu:
|
|
monitor_enabled: true
|
|
stt_vram_budget_gb: 2.0
|
|
tts_vram_budget_gb: 4.0
|