diff --git a/backend/app/services/simulator.py b/backend/app/services/simulator.py index dc284a4..205f44f 100644 --- a/backend/app/services/simulator.py +++ b/backend/app/services/simulator.py @@ -412,7 +412,12 @@ class Simulator: system, user_prompt, temperature=0.2, - max_tokens=1200, + # 6000 (was 1200): same hidden-reasoning truncation class as the final + # judge (500 #4). At 1200 the Qwen GGUF's internal chain consumed the + # shared cap on a full seller transcript, truncated the JSON, and the + # coaching silently failed closed to {} — the loss debrief then showed + # only the outcome banner + persona, with no "why" and no coaching. + max_tokens=6000, ) return result if isinstance(result, dict) else {} diff --git a/backend/tests/test_llm_budgets.py b/backend/tests/test_llm_budgets.py index 5d848f4..8f49fa4 100644 --- a/backend/tests/test_llm_budgets.py +++ b/backend/tests/test_llm_budgets.py @@ -59,6 +59,22 @@ def test_judge_budget_covers_hidden_reasoning(): assert stub.json_kwargs[0]["max_tokens"] >= 1600 +def test_public_debrief_budget_covers_hidden_reasoning(): + # The public coaching pass (why / failurePoints / coaching) analyzes the full + # seller transcript. Same hidden-reasoning truncation class as the final judge + # (500 #4): at 1200 the internal chain ate the budget, the JSON got truncated, + # complete_json threw LLMError, and _public_debrief failed closed to {} — so the + # loss debrief showed only the outcome banner + persona, no "why" and no coaching. + # 6000 matches the final judge. + stub = _CapturingLLM("") + Simulator(stub).public_debrief( + messages=[{"role": "seller", "text": "สวัสดี"}], + outcome="lost", + score=30, + ) + assert stub.json_kwargs[0]["max_tokens"] >= 6000 + + def test_final_judge_budget_covers_hidden_reasoning(): # The FINAL judge (end-of-session verdict) is the fatal path — a truncation # there 500s the whole chat send (the 2026-10-02 live incident: JSON cut off