Xiaozhi WebSocket endpoint with handshake/auth, mock STT/TTS/Hermes backends, Thai chunker, barge-in queue, latency logger, GPU planner, voice-profile guard (reasoning=none, session_search only). CON-002: no GPU/audio libs loaded at import. MUST-NOT-001: all model names/tokens from config.
98 lines
3.2 KiB
Python
98 lines
3.2 KiB
Python
"""Configuration for the Hermes-Xiaozhi bridge.
|
|
|
|
Everything is data. The bridge code never hard-codes a model name, a device
|
|
token, or a firmware constant (MUST-NOT-001, CON-002, CON-004).
|
|
|
|
``Config.from_file`` / ``Config.from_dict`` produce a pydantic model so bad
|
|
config fails loudly at startup, not mid-conversation.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from typing import Literal, Optional
|
|
|
|
from pydantic import BaseModel, Field
|
|
|
|
|
|
class STTConfig(BaseModel):
|
|
engine: Literal["mock", "typhoon", "faster-whisper"] = "mock"
|
|
language: str = "th"
|
|
sample_rate: int = 16000
|
|
# Optional per-engine knobs (endpoint, model path) — kept as free-form so a
|
|
# backend can be swapped without a schema change (MUST-NOT-001).
|
|
typhoon_endpoint: Optional[str] = None
|
|
whisper_model: str = "large-v3"
|
|
|
|
|
|
class TTSConfig(BaseModel):
|
|
engine: Literal["mock", "jaitts"] = "mock"
|
|
voice: str = "default"
|
|
device: str = "cuda:0"
|
|
sample_rate: int = 24000
|
|
jaitts_endpoint: Optional[str] = None
|
|
|
|
|
|
class HermesConfig(BaseModel):
|
|
# Transport for the local Qwen/Hermes voice profile. "mock" for tests.
|
|
transport: Literal["mock", "openai_http"] = "mock"
|
|
base_url: Optional[str] = None
|
|
api_key: Optional[str] = None
|
|
# REQ-004 — voice profile is reasoning-none, read/recall tools only.
|
|
reasoning: str = "none"
|
|
tools: list[str] = Field(default_factory=lambda: ["session_search"])
|
|
max_context_tokens: int = 16000 # REQ-005 / Phase 17 working context
|
|
max_history_turns: int = 12
|
|
|
|
|
|
class DeviceConfig(BaseModel):
|
|
device_id: str
|
|
token_hash: str # SHA-256 hex of the shared token (REQ-009)
|
|
|
|
|
|
class ProtocolConfig(BaseModel):
|
|
# FIRMWARE: verify against the real Xiaozhi firmware before locking (CON-004).
|
|
hello_version: int = 1
|
|
opus_sample_rate: int = 16000
|
|
opus_channels: int = 1
|
|
frame_ms: int = 20 # typical Xiaozhi Opus frame size
|
|
|
|
|
|
class SecurityConfig(BaseModel):
|
|
enabled: bool = True
|
|
devices: list[DeviceConfig] = Field(default_factory=list)
|
|
|
|
|
|
class GpuConfig(BaseModel):
|
|
monitor_enabled: bool = True
|
|
stt_vram_budget_gb: float = 2.0
|
|
tts_vram_budget_gb: float = 4.0
|
|
|
|
|
|
class Config(BaseModel):
|
|
host: str = "127.0.0.1" # REQ-012 — LAN/localhost only; tunnel is external
|
|
port: int = 8765
|
|
ws_path: str = "/ws/xiaozhi"
|
|
stt: STTConfig = Field(default_factory=STTConfig)
|
|
tts: TTSConfig = Field(default_factory=TTSConfig)
|
|
hermes: HermesConfig = Field(default_factory=HermesConfig)
|
|
protocol: ProtocolConfig = Field(default_factory=ProtocolConfig)
|
|
security: SecurityConfig = Field(default_factory=SecurityConfig)
|
|
gpu: GpuConfig = Field(default_factory=GpuConfig)
|
|
|
|
@classmethod
|
|
def from_dict(cls, data: dict) -> "Config":
|
|
return cls.model_validate(data)
|
|
|
|
@classmethod
|
|
def from_file(cls, path: str) -> "Config":
|
|
import yaml
|
|
|
|
with open(path, "r", encoding="utf-8") as fh:
|
|
data = yaml.safe_load(fh) or {}
|
|
return cls.from_dict(data)
|
|
|
|
|
|
# A tiny, importable default used by tests and as a fallback when no file is
|
|
# present. It is intentionally all-mock so the bridge is usable anywhere
|
|
# (CON-002) without a GPU, device, or native libs.
|
|
DEFAULT = Config()
|