Xiaozhi WebSocket endpoint with handshake/auth, mock STT/TTS/Hermes backends, Thai chunker, barge-in queue, latency logger, GPU planner, voice-profile guard (reasoning=none, session_search only). CON-002: no GPU/audio libs loaded at import. MUST-NOT-001: all model names/tokens from config.
70 lines
2.1 KiB
Python
70 lines
2.1 KiB
Python
"""GPU resource management (REQ-010).
|
|
|
|
Manual, not automatic (Phase 13: "don't automate too much early on"). Three
|
|
named modes; a :class:`GpuMonitor` reports VRAM/util so a human (or a later
|
|
auto policy) can decide. The mode switch is a pure decision — no process
|
|
spawning here, so it's testable on a machine with no GPU.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
from dataclasses import dataclass
|
|
|
|
|
|
@dataclass
|
|
class GpuState:
|
|
vram_total_gb: float
|
|
vram_free_gb: float
|
|
util_percent: int = 0
|
|
stt_resident: bool = False
|
|
tts_resident: bool = False
|
|
comfyui_active: bool = False
|
|
|
|
|
|
class GpuMonitor:
|
|
"""Reads GPU state. ``mock`` for tests; a real one would shell to nvidia-smi."""
|
|
|
|
def __init__(self, state: GpuState | None = None) -> None:
|
|
self._state = state or GpuState(vram_total_gb=16.0, vram_free_gb=16.0)
|
|
|
|
@property
|
|
def state(self) -> GpuState:
|
|
return self._state
|
|
|
|
|
|
class ResourceMode:
|
|
FULL = "voice-full"
|
|
LITE = "voice-lite"
|
|
OFF = "voice-off"
|
|
|
|
|
|
class GpuResourceManager:
|
|
"""Decide which voice models should be resident for a given mode.
|
|
|
|
* FULL — STT + TTS resident (normal voice use).
|
|
* LITE — STT resident, TTS loaded on demand (for gaming; frees VRAM).
|
|
* OFF — nothing resident (ComfyUI / other GPU work).
|
|
"""
|
|
|
|
def __init__(self, monitor: GpuMonitor) -> None:
|
|
self.monitor = monitor
|
|
|
|
def plan(self, mode: str) -> dict:
|
|
if mode == ResourceMode.FULL:
|
|
return {"stt": True, "tts": True, "unload_tts": False}
|
|
if mode == ResourceMode.LITE:
|
|
return {"stt": True, "tts": False, "unload_tts": True}
|
|
if mode == ResourceMode.OFF:
|
|
return {"stt": False, "tts": False, "unload_tts": True}
|
|
raise ValueError(f"unknown gpu mode: {mode}")
|
|
|
|
def can_fit(self, mode: str) -> bool:
|
|
"""Heuristic: does the current free VRAM fit the plan's budgets?"""
|
|
s = self.monitor.state
|
|
plan = self.plan(mode)
|
|
needed = 0.0
|
|
if plan["stt"]:
|
|
needed += 2.0
|
|
if plan["tts"]:
|
|
needed += 4.0
|
|
return s.vram_free_gb >= needed
|