The llama.cpp endpoint writes the model's hidden reasoning INLINE into content unless enable_thinking=true is sent (which relocates it to the separate reasoning_content field). The app sent the field absent, so: - complete_json got corrupted JSON -> live 500 'Unterminated string char 143' - persona JSON generation failed (same root cause, earlier incident) - persona replies were bloated with the reasoning chain Send the flag on all three call paths (complete/complete_json/ complete_conversation). thinking param retained for API compat, now a no-op on this provider (always separated — the only correct mode here). - tests/test_llm_enable_thinking.py pins the flag on all three paths - full suite: 543 passed; independent reviewer PASS (22 focused passed, single OpenAI client in the app, both fixed sites confirmed, no stale tests) - cleanup stale comments that encoded the backwards understanding (endpoint 'ignores enable_thinking' — it honors it; that misconception is what caused this incident)
67 lines
2.1 KiB
Python
67 lines
2.1 KiB
Python
"""The llama.cpp endpoint writes the model's hidden reasoning INLINE into
|
|
`content` UNLESS `enable_thinking` is sent as `true` — which relocates it to the
|
|
separate `reasoning_content` field and keeps `content` clean.
|
|
|
|
Omitting the field (or sending false) corrupts every JSON response with inline
|
|
thinking: the live 2026-10-02 500 was `LLM returned invalid JSON:
|
|
Unterminated string` and persona generation was failing for the same reason.
|
|
|
|
These tests pin the flag so a future edit that drops it fails loudly.
|
|
"""
|
|
from __future__ import annotations
|
|
|
|
import app.llm as llm_mod
|
|
from app.llm import LLMClient
|
|
|
|
|
|
def _make_client() -> LLMClient:
|
|
c = LLMClient(base_url="http://localhost:9/v1", api_key="test-key", model="m")
|
|
c.max_attempts = 1
|
|
c.retry_delay = 0.0
|
|
return c
|
|
|
|
|
|
def _resp(content: str = '{"ok": 1}'):
|
|
message = type("M", (), {"content": content})()
|
|
choice = type("C", (), {"message": message})()
|
|
return type("R", (), {"choices": [choice]})()
|
|
|
|
|
|
def test_complete_always_sends_enable_thinking_true(monkeypatch):
|
|
c = _make_client()
|
|
seen: dict = {}
|
|
|
|
def spy(**kwargs):
|
|
seen["extra_body"] = kwargs.get("extra_body")
|
|
return _resp()
|
|
|
|
c.client.chat.completions.create = spy
|
|
c.complete("sys", "user")
|
|
assert seen["extra_body"] == {"enable_thinking": True}
|
|
|
|
|
|
def test_complete_json_always_sends_enable_thinking_true(monkeypatch):
|
|
c = _make_client()
|
|
seen: dict = {}
|
|
|
|
def spy(**kwargs):
|
|
seen["extra_body"] = kwargs.get("extra_body")
|
|
return _resp('{"mood": 1}')
|
|
|
|
c.client.chat.completions.create = spy
|
|
c.complete_json("sys", "user")
|
|
assert seen["extra_body"] == {"enable_thinking": True}
|
|
|
|
|
|
def test_complete_conversation_always_sends_enable_thinking_true(monkeypatch):
|
|
c = _make_client()
|
|
seen: dict = {}
|
|
|
|
def spy(**kwargs):
|
|
seen["extra_body"] = kwargs.get("extra_body")
|
|
return _resp("สวัสดีค่ะ")
|
|
|
|
c.client.chat.completions.create = spy
|
|
c.complete_conversation([{"role": "seller", "text": "hi"}])
|
|
assert seen["extra_body"] == {"enable_thinking": True}
|