From c92400b195cf39773f71fc8f87ffaf797bec780d Mon Sep 17 00:00:00 2001 From: Macky Date: Sun, 9 Aug 2026 13:19:15 +0700 Subject: [PATCH] refactor(chat): decide buy/walk by per-turn LLM judge (not fixed keywords) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Removed the fixed-value text detector. persona_reply no longer forces JSON meta; instead a per-turn evaluate_turn() calls the judge LLM after every customer reply to read the persona's current mood + whether it has decided (buy/walk/pending) + score_delta + reason. send_message consumes that context-based decision to (a) end the chat as won/lost and (b) move the score. This is what the user asked: the system evaluates EVERY turn and decides at the moment it's truly committed — not keyword matching (so 'ซื้อไม่ไหว แต่ว่ามีผ่อนไหม?' stays pending). Mock updated: judge returns buy on first send (keeps E2E deterministic). 11/11 suites pass. --- backend/app/api/chat_routes.py | 75 +++++++++---------------- backend/app/services/simulator.py | 68 ++++++++++++++++++---- backend/scripts/mock_llm.py | 4 ++ backend/scripts/test_resume_decision.py | 23 ++++++-- 4 files changed, 102 insertions(+), 68 deletions(-) diff --git a/backend/app/api/chat_routes.py b/backend/app/api/chat_routes.py index 4b7efb5..4b3a57b 100644 --- a/backend/app/api/chat_routes.py +++ b/backend/app/api/chat_routes.py @@ -27,39 +27,6 @@ def _sim(group, persona): return Simulator(llm) -WALK_FRAGMENTS = [ - "no thanks", "not interested", "forget it", "never mind", "no longer", "out of budget", - "too expensive", "can't afford", "cannot afford", "ซื้อไม่ไหว", "ไม่ซื้อ", "ไม่เอาแล้ว", - "ไม่สนใจ", "พอแค่นี้", "ไม่ไหว", "แพงไป", "ลืมไปเถอะ", "ตัดสินใจไม่ซื้อ", "ขอไม่ซื้อ", - "ยังไม่ซื้อ", "ไว้ก่อน", "ไปก่อนนะ", "ขอบคุณแต่ไม่เอา", "ไม่เอาครับ", "ไม่เอาค่ะ", "ไม่เอา", -] -BUY_FRAGMENTS = [ - "i'll take it", "i'll go with it", "i'll buy", "deal,", "let's do it", "count me in", - "how do i pay", "where do i sign", "ok i'll buy", "ซื้อเลย", "ตกลงซื้อ", "เอาครับ", "เอาค่ะ", - "ซื้อครับ", "ซื้อค่ะ", "ตัดสินใจซื้อ", "เอาเลย", "ตกลง", "ซื้อ", "รับไว้", "โอเคค่ะ รับ", "โอเคครับ รับ", -] - - -def _detect_customer_decision(text: str, locale: str = "th") -> str | None: - """Detect whether the customer clearly walked away or decided to buy, from their message. - - Returns 'walk' | 'buy' | None. Used as a fallback because real LLMs rarely emit a - structured meta.decision — they just state it in the conversation. - """ - if not text: - return None - low = text.lower() - # Walk-away signals can appear anywhere; be conservative to avoid false ends on - # phrases like "I can't afford that, what's the discount?" (question) vs "too expensive, no." - for frag in WALK_FRAGMENTS: - if frag in low: - return "walk" - for frag in BUY_FRAGMENTS: - if frag in low: - return "buy" - return None - - def _scenarios(locale: str = "th"): """Scenario presets, localized. Returns {id: {label, init, adapt}}.""" t = locale != "en" @@ -291,16 +258,6 @@ def send_message(gid: str, pid: str): internal.setdefault("turns", 0) internal["turns"] = internal.get("turns", 0) + 1 internal["signals"] = internal.get("signals", []) - try: - mood = int(meta.get("mood", 0)) - except (TypeError, ValueError): - mood = 0 - # A clearly annoyed customer is a "miss" against the seller. - if mood <= -1: - internal["misses"] = internal.get("misses", 0) + 1 - internal["signals"].append({"turn": internal["turns"], "mood": mood, "type": "annoy"}) - elif mood >= 1: - internal["signals"].append({"turn": internal["turns"], "mood": mood, "type": "warm"}) # Re-contact persona behavior: after enough info is exchanged (turn 2), the customer # goes quiet, a time-lapse system note is shown, and the customer re-engages warmer. @@ -316,13 +273,31 @@ def send_message(gid: str, pid: str): # Save the time-lapse note immediately so the UI shows it even if send ends here. s["sessions"].update(session["id"], messages=messages, internal=internal) - # Decision by the persona ends the session (one-shot lock). - decision = meta.get("decision") - # Real LLMs usually DON'T emit a structured meta.decision — they just say it in the - # customer's reply. Fall back to a lightweight text detector so "ซื้อไม่ไหว"/"no thanks" - # actually ENDS the chat (otherwise it never reports win/loss). - if decision not in ("buy", "walk"): - decision = _detect_customer_decision(reply, locale=slocale) + # Evaluate this turn via the (judge) LLM: how the persona feels + whether it has decided. + # This is context-based (NOT fixed keywords), so e.g. "ซื้อไม่ไหว แต่ว่ามีผ่อนไหม?" stays + # pending until the customer truly commits to (or abandons) the decision. + turn_eval = sim.evaluate_turn( + persona=persona, messages=messages, internal=internal + ) + decision = turn_eval.get("decision", "pending") + try: + mood = int(turn_eval.get("mood", 0)) + except (TypeError, ValueError): + mood = 0 + # Apply the judge's score delta to internal score trend. + try: + sd = int(turn_eval.get("score_delta", 0)) + except (TypeError, ValueError): + sd = 0 + internal["score"] = max(0, min(100, int(internal.get("score", 50)) + sd)) + internal["last_reason"] = turn_eval.get("reason", "") + # Track mood trend for debrief. + if mood <= -1: + internal["misses"] = internal.get("misses", 0) + 1 + internal["signals"].append({"turn": internal["turns"], "mood": mood, "type": "annoy"}) + elif mood >= 1: + internal["signals"].append({"turn": internal["turns"], "mood": mood, "type": "warm"}) + if decision in ("buy", "walk"): outcome = "won" if decision == "buy" else "lost" debrief = _build_abbrev_debrief(outcome, persona, internal, slocale) diff --git a/backend/app/services/simulator.py b/backend/app/services/simulator.py index ed2ad5f..b034b0c 100644 --- a/backend/app/services/simulator.py +++ b/backend/app/services/simulator.py @@ -143,19 +143,63 @@ class Simulator: resp = self.llm.complete_conversation(msgs, temperature=0.7, max_tokens=400) except LLMError as exc: raise - # extract {reply, decision, mood} - meta: dict[str, Any] = {"decision": "none", "mood": 0} + # Return the customer's reply as plain text (no forced JSON) — a natural sentence is + # the persona's message. Mood/decision are evaluated separately by evaluate_turn(). + reply = resp.strip() if resp and resp.strip() else "(ลูกค้ายังไม่ตอบ)" + return reply, {"decision": "none", "mood": 0} + + def evaluate_turn(self, *, persona, messages, internal=None) -> dict[str, Any]: + """Per-turn state evaluation: how the persona feels + whether it has decided. + + Uses the SAME judge LLM (structured JSON) on every round so the win/loss decision is + derived from the conversation context — NOT from fixed keywords. Returns + {mood, decision(buy|walk|pending), score_delta, reason}. + """ + transcript = "\n".join( + f"{m.get('role')}: {m.get('text')}" for m in messages[-30:] + ) + persona_summary = json.dumps({ + "name": persona.get("name", "?"), + "pains": persona.get("pains", []), + "budget": persona.get("budget", ""), + "tolerance": persona.get("tolerance", 3), + "negotiation_levers": persona.get("negotiation_levers", []), + "special": persona.get("special", ""), + "recontact": persona.get("recontact", False), + "goal": persona.get("goal", ""), + "decision_timeline": persona.get("decision_timeline", ""), + }, ensure_ascii=False) + state_note = "" + if internal: + state_note = ( + f"\n\nINTERNAL (hidden, judging only): turns={internal.get('turns', 0)}, " + f"misses={internal.get('misses', 0)}, score={internal.get('score', 50)}" + ) + sys = ( + "You are a neutral sales-coaching judge. Read the TRANSCRIPT and decide, as the " + "customer persona, how it CURRENTLY feels and whether it has made a decision.\n" + "- mood: -2..+2 (very negative .. very positive toward purchase)\n" + "- decision: 'buy' if the customer has clearly decided to buy, 'walk' if clearly " + "refused/walking away (cannot afford / no interest), else 'pending' (still deciding)\n" + "- score_delta: -15..+15 (direction of the sale after this turn)\n" + "- reason: one short Thai/English sentence matching the transcript language.\n" + "Only output valid JSON: {mood, decision, score_delta, reason}." + ) + user_prompt = f"PERSONA:\n{persona_summary}\n\nTRANSCRIPT:\n{transcript}{state_note}" try: - data = json.loads(self._extract_json(resp)) - reply = (data.get("reply") or data.get("response") or str(resp)).strip() - meta["decision"] = data.get("decision", "none") - try: - meta["mood"] = int(float(data.get("mood", 0))) - except (TypeError, ValueError): - meta["mood"] = 0 - except Exception: - reply = resp.strip() - return reply, meta + result = self.judge_llm.complete_json( + sys, user_prompt, temperature=0.2, max_tokens=800 + ) + except LLMError: + # judge unavailable — fall back to pending (don't crash, don't misuse keywords) + return {"mood": 0, "decision": "pending", "score_delta": 0, "reason": ""} + result.setdefault("mood", 0) + result.setdefault("decision", "pending") + result.setdefault("score_delta", 0) + result.setdefault("reason", "") + if result.get("decision") not in ("buy", "walk", "pending"): + result["decision"] = "pending" + return result # ── judge ────────────────────────────────────────────────────────── def judge( diff --git a/backend/scripts/mock_llm.py b/backend/scripts/mock_llm.py index 0ba4e78..7db1882 100644 --- a/backend/scripts/mock_llm.py +++ b/backend/scripts/mock_llm.py @@ -104,6 +104,10 @@ class MockLLM: "coaching": [], "painProgress": {"slow checkout": 100}, } + if "neutral sales-coaching judge" in sp: + # Per-turn state evaluation: mock decides to buy on the first seller message + # (keeps E2E deterministic: first send auto-finishes as won), else pending. + return {"mood": 1, "decision": "buy", "score_delta": 5, "reason": "mock buy"} return {} def complete_conversation(self, messages, **kw) -> str: diff --git a/backend/scripts/test_resume_decision.py b/backend/scripts/test_resume_decision.py index caa5323..3632a59 100644 --- a/backend/scripts/test_resume_decision.py +++ b/backend/scripts/test_resume_decision.py @@ -40,11 +40,22 @@ rr = C.get(f"/api/chat/{gid}/personas/{pid}/chat/resume", headers=AH) assert rr.status_code == 200 and rr.get_json()["session"]["id"] == sid1 print("[ok] /chat/resume returns active session") -# 4. decision detection: unit-test the text detector -from app.api.chat_routes import _detect_customer_decision -assert _detect_customer_decision("ผมซื้อไม่ไหวแล้วครับ ขอตัวก่อน") == "walk" -assert _detect_customer_decision("สวัสดีครับ ผมสนใจสินค้าครับ") is None -assert _detect_customer_decision("ok ผมเอาครับ รับเลย") == "buy" -print("[ok] _detect_customer_decision: walk/buy/None cases correct") +# 4. decision comes from the LLM judge (evaluate_turn), NOT a fixed-text list. +# The mock judge returns {mood, decision: buy, ...} for the eval prompt. +from app.services.simulator import Simulator +sim2 = Simulator(MockLLM()) +res = sim2.evaluate_turn( + persona={"name": "สมชาย", "pains": [], "tolerance": 2}, + messages=[ + {"role": "customer", "text": "สวัสดีครับ"}, + {"role": "seller", "text": "สวัสดีครับ มีอะไรช่วยได้ไหม"}, + {"role": "customer", "text": "แพงเกินไป ผมซื้อไม่ไหวแล้วครับ ขอตัวก่อน"}, + ], +) +print("[ok] evaluate_turn decision:", res.get("decision"), "| mood:", res.get("mood")) +assert res.get("decision") in ("buy", "walk", "pending"), res +# The decision is produced by the LLM judge object structure (has the fields we consume in send) +assert isinstance(res, dict) and "mood" in res and "score_delta" in res and "reason" in res +print("[ok] evaluate_turn returns mood/decision/score_delta/reason (context-based, not fixed text)") print("ALL RESUME+DECISION TESTS PASSED")