diff --git a/README.md b/README.md index 9f729fe..dbffe1b 100644 --- a/README.md +++ b/README.md @@ -29,7 +29,8 @@ pass on any machine — no GPU, no device, no native libs. ```bash cd hermes-xiaozhi-bridge python3 -m venv .venv && .venv/bin/pip install -r requirements.txt -.venv/bin/python -m uvicorn app.main:app --host 127.0.0.1 --port 8765 +.venv/bin/python -m app.main +# (loads config.yaml from the repo root; override with BRIDGE_CONFIG=/path) .venv/bin/python -m pytest -q ``` diff --git a/app/main.py b/app/main.py index 4c734fe..030455b 100644 --- a/app/main.py +++ b/app/main.py @@ -41,9 +41,22 @@ def make_app(cfg: Config | None = None) -> FastAPI: def main() -> None: import uvicorn - cfg = Config() + cfg = _load_config() uvicorn.run(make_app(cfg), host=cfg.host, port=cfg.port) +def _load_config() -> Config: + """Load config.yaml from the repo root; fall back to all-mock defaults.""" + import os + + path = os.environ.get( + "BRIDGE_CONFIG", + os.path.join(os.path.dirname(os.path.dirname(__file__)), "config.yaml"), + ) + if os.path.exists(path): + return Config.from_file(path) + return Config() + + if __name__ == "__main__": main() diff --git a/config.yaml b/config.yaml index d03ef71..a2abc8e 100644 --- a/config.yaml +++ b/config.yaml @@ -7,7 +7,7 @@ # real Xiaozhi firmware before locking (CON-004). host: 127.0.0.1 -port: 8765 +port: 8766 ws_path: /ws/xiaozhi protocol: @@ -42,7 +42,7 @@ security: enabled: true devices: - device_id: xiaozhi-main - token_hash: "0000000000000000000000000000000000000000000000000000000000000000" + token_hash: "b324d2dc2d1242b4df53aca6d1691b062e1d4c085ff90ac45c03e2cec9f18942" gpu: monitor_enabled: true diff --git a/docs/HANDOFF.md b/docs/HANDOFF.md index d10f82b..4151089 100644 --- a/docs/HANDOFF.md +++ b/docs/HANDOFF.md @@ -23,8 +23,14 @@ ```bash .venv/bin/python -m pytest -q # → 18 passed .venv/bin/python -c "import app.main" # → OK +.venv/bin/python -m app.main # → serves /health + /ws/xiaozhi on config.yaml host:port +.venv/bin/python smoke_live.py # → live WS smoke: health, bad-token/bad-device rejection, + # handshake, spoken turn (stt→state→text→audio) — PASS ``` CON-002 checked: no torch/typhoon/whisper/jait/opus/numpy in `sys.modules` after import. +Bug found+fixed on this host: `main()` ignored `config.yaml` (used `Config()` defaults — +wrong port + zero-hashes token); now loads config.yaml from the repo root +(override with `BRIDGE_CONFIG`). Real device token issued for `xiaozhi-main`. ## Do this first on the voice-server (ordered) @@ -36,7 +42,13 @@ CON-002 checked: no torch/typhoon/whisper/jait/opus/numpy in `sys.modules` after 4. **TTS**: `tts.engine: jaitts` — point `jaitts_endpoint` at JaiTTS on the 5060 Ti. First-audio latency is the headline metric. 5. **Hermes transport**: `hermes.transport: openai_http` + `base_url`/`api_key` for the Qwen 3.8 vLLM endpoint (V100). Streaming SSE path already implemented in `app/hermes/__init__.py`. 6. **Device auth**: generate a real token per device, `token_hash: sha256(token)` hex in `config.yaml`. -7. **Bind**: keep `host: 127.0.0.1` — the Cloudflare tunnel is the only external surface (REQ-012). Add the tunnel hostname → `127.0.0.1:8765` on the existing tunnel. + - Done on this host (2026-10-03): `xiaozhi-main` token = `8a07643d106f9282a8f2eb1161da4f73657ce99b6602b98df89d011b8caf48a2` + (store in the ESP32 firmware config; `token_hash` in config.yaml is its SHA-256). + - ⚠️ config.yaml is committed to the repo — this token is public. Rotate it (new token, + new hash) before the device leaves the lab, or move config to a gitignored overlay. +7. **Bind**: keep `host: 127.0.0.1` — the Cloudflare tunnel is the only external surface (REQ-012). + Point the tunnel hostname at `127.0.0.1:` where `` is the value in `config.yaml` + (this host uses 8766 — 8765 is taken by another local service). ## Known sharp edges diff --git a/plan.md b/plan.md index 7d163bb..8878a82 100644 --- a/plan.md +++ b/plan.md @@ -1,6 +1,6 @@ # Plan — hermes-xiaozhi-bridge -Status: **Phase 1 complete — testable skeleton green (18 tests)** +Status: **Phase 1.5 — live-verified (CPU host): 18 tests + live WS smoke PASS** Last update: 2026-10-03 Next action: voice-server — wire real STT/TTS/Opus backends, verify protocol constants against firmware (CON-004). @@ -49,6 +49,10 @@ imports that activate only on the voice-server. ## Evidence - 2026-10-03 `pytest -q` → **18 passed** (protocol, config, auth, chunker, barge-in queue, latency, voice-profile guard, GPU planner, VAD, e2e mock turn, app import + live WS handshake+turn). - 2026-10-03 `python -c "import app.main"` → OK; `sys.modules` check → no torch/typhoon/whisper/jait/opus/numpy loaded (CON-002). +- 2026-10-03 Live boot on Windows CPU host: fixed `main()` ignoring config.yaml (was `Config()` + defaults); port 8765 taken by another local service → 8766. Issued real token for `xiaozhi-main` + (hash in config.yaml). `smoke_live.py` against running server: health ok, bad-token rejected, + bad-device rejected, handshake → spoken turn (stt → state → text → audio) → **PASS**. - Chunker verified by direct call: `chunk_text("สวัสดี. แล้วไง? ครับ")` → `["สวัสดี.", "แล้วไง?", "ครับ"]`; 200-char fragment → 4×60-char hard cap. - Known bug found+fixed in review: `_SENT_END` originally split on Thai vowel marks (U+0E40/41/48) — would break every word; now ASCII punctuation + ฯ (U+0E3F) only. diff --git a/smoke_live.py b/smoke_live.py new file mode 100644 index 0000000..910a790 --- /dev/null +++ b/smoke_live.py @@ -0,0 +1,98 @@ +"""Live smoke test against the running bridge (real WS, real auth, mock backends).""" +import asyncio +import json +import sys + +import httpx +import websockets + +BASE = "http://127.0.0.1:8766" +WS = "ws://127.0.0.1:8766/ws/xiaozhi" +TOKEN = "8a07643d106f9282a8f2eb1161da4f73657ce99b6602b98df89d011b8caf48a2" + +results = {} + +# 1. health +results["health"] = httpx.get(f"{BASE}/health", timeout=5).json() + +# 2. bad token must be rejected +async def bad_token(): + async with websockets.connect(WS) as ws: + await ws.send(json.dumps({ + "type": "hello", + "device_id": "xiaozhi-main", + "authorization": "wrong-token", + "audio": {"format": "opus", "sample_rate": 16000, "channels": 1, "frame_duration": 20}, + })) + return await ws.recv() + +# 3. unknown device must be rejected +async def bad_device(): + async with websockets.connect(WS) as ws: + await ws.send(json.dumps({ + "type": "hello", + "device_id": "intruder", + "authorization": TOKEN, + "audio": {"format": "opus", "sample_rate": 16000, "channels": 1, "frame_duration": 20}, + })) + return await ws.recv() + +# 4. real handshake + a spoken turn +async def real_turn(): + frames = [] + audio_bytes = 0 + async with websockets.connect(WS) as ws: + await ws.send(json.dumps({ + "type": "hello", + "device_id": "xiaozhi-main", + "client_id": "esp32-01", + "authorization": TOKEN, + "protocol_version": 1, + "audio": {"format": "opus", "sample_rate": 16000, "channels": 1, "frame_duration": 20}, + })) + for _ in range(20): + data = await ws.recv() + if isinstance(data, bytes): + audio_bytes += len(data) + frames.append("audio") + else: + frames.append(json.loads(data).get("type")) + if frames[-1] == "hello": # hello reply → session live + break + # one spoken turn: 4800 bytes of PCM (0.15s @ 16k mono 16-bit) + await ws.send(bytes(range(256)) * 19) + got_text = False + got_audio = False + for _ in range(50): + data = await ws.recv() + if isinstance(data, bytes): + got_audio = True + audio_bytes += len(data) + else: + frames.append(json.loads(data).get("type")) + if json.loads(data).get("type") == "text": + got_text = True + break + # keep reading TTS audio chunks that follow the text + while not got_audio: + data = await ws.recv() + if isinstance(data, bytes): + got_audio = True + audio_bytes += len(data) + else: + frames.append(json.loads(data).get("type")) + return {"frames": frames, "audio_bytes": audio_bytes, "got_text": got_text, "got_audio_out": got_audio} + +results["bad_token"] = asyncio.run(bad_token()) +results["bad_device"] = asyncio.run(bad_device()) +results["real_turn"] = asyncio.run(real_turn()) + +ok = ( + results["health"].get("status") == "ok" + and json.loads(results["bad_token"]).get("type") == "error" + and json.loads(results["bad_device"]).get("type") == "error" + and results["real_turn"]["got_audio_out"] +) +print(json.dumps(results, ensure_ascii=False, indent=2)) +print("LIVE SMOKE:", "PASS" if ok else "FAIL") +sys.exit(0 if ok else 1)