"""Persona generator: builds 15 personas (5 per tier) from a Sales Kit + scenario.""" from __future__ import annotations import json from typing import Any from ..llm import LLMClient from .persona_prompts import PERSONA_SYSTEM TIERS = ["A", "B", "C"] PER_TIER = 5 # 5 per tier = 15 total TARGET = 15 # total personas the system generates (no "add more" button needed) class PersonaGenerator: def __init__(self, llm: LLMClient) -> None: self.llm = llm def generate( self, *, sales_kit: dict[str, Any], language: str = "en", channel: str = "social", ) -> list[dict[str, Any]]: kit_json = json.dumps(sales_kit, ensure_ascii=False)[:12000] lang_name = "Thai" if language == "th" else "English" scenario = (sales_kit.get("scenarioFrame") or "").strip() or "a general product sale" user_prompt = ( f"Platform/channel preference: {channel}\n" f"Language: {lang_name} (all persona text in {lang_name})\n" f"Sales Kit:\n{kit_json}\n\n" f"Generate exactly 15 personas (5 per tier A/B/C) as JSON." ) # Real LLMs sometimes return fewer than 15 (truncation / merge). Retry up to 2 extra # times with a nudge; the body below already ACCEPTS short results (>= 8) instead of # hard-failing, so these retries are just best-effort to reach a fuller set. attempt = 0 while True: attempt += 1 prompt = user_prompt + ( "" if attempt == 1 else "\n\n(Note: you left some personas out — please output all 15, one JSON object per persona, no extra prose.)" ) result = self.llm.complete_json( PERSONA_SYSTEM, prompt, temperature=0.8, max_tokens=14000 ) personas = result.get("personas") or [] if (isinstance(personas, list) and len(personas) >= TARGET) or attempt >= 3: break if not isinstance(personas, list) or not personas: raise ValueError("persona generator returned no personas") normalized, counts = [], {"A": 0, "B": 0, "C": 0} for idx, p in enumerate(personas, start=1): if len(normalized) >= TARGET: break # already reached 20 total if not isinstance(p, dict): continue tier = p.get("tier", p.get("intent_tier")) if tier not in TIERS: tier = "B" if counts[tier] >= PER_TIER: continue # skip overflow per tier counts[tier] += 1 p["id"] = f"persona-{idx:02d}" p["tier"] = tier p["channel"] = p.get("channel", channel) p.setdefault("initiation_mode", "customer") p.setdefault("special", "") p.setdefault("difficulty", 1) p.setdefault("pains", []) p.setdefault("negotiation_levers", []) p.setdefault("objections", []) p.setdefault("tolerance", 3) p.setdefault("recontact", False) normalized.append(p) # Wrap tier-C: ensure at least one wrong_text persona if "C" in counts and not any( p.get("special") == "wrong_text" for p in normalized ): # find first tier-C and mark it for p in normalized: if p["tier"] == "C": p["special"] = "wrong_text" break return normalized def generate_variant( self, source: dict[str, Any], sales_kit: dict[str, Any] | None = None, language: str = "en", ) -> dict[str, Any]: """Create ONE new persona that is a fresh incarnation of a source persona. The variant LOCKS the source's core traits — pain points, objections, negotiation levers, tolerance (temper), and any special/recontact behavior — so it practices the SAME selling challenge, but gets a NEW identity (name, profession, age, location, background, personality, income, opener) so it isn't an identical copy. Because the seller already knows how this customer 'plays', we vary the new identity so the trainee still has to re-read and re-adjust rather than memorizing exact answers. """ kit_note = ( f"Sales Kit\\n{json.dumps(sales_kit, ensure_ascii=False)[:6000]}" if sales_kit else "" ) src = json.dumps( { "pains": source.get("pains", []), "objections": source.get("objections", []), "negotiation_levers": source.get("negotiation_levers", []), "tolerance": source.get("tolerance", 3), "special": source.get("special", ""), "recontact": source.get("recontact", False), "goal": source.get("goal", ""), "decision_timeline": source.get("decision_timeline", ""), "budget": source.get("budget", ""), "difficulty": source.get("difficulty", 1), "tier": source.get("tier", "B"), "product_context": source.get("product_context", ""), }, ensure_ascii=False, ) lang_name = "Thai" if language == "th" else "English" prompt = ( f"Create ONE new, realistic customer persona that is a fresh incarnation of an existing one.\n" f"Language: {lang_name} (all text in {lang_name})\n" f"{kit_note}\n" f"LOCK (keep exactly these — they drive the training): pains[], objections[], " f"negotiation_levers[], tolerance, special, recontact, goal, decision_timeline, " f"budget, difficulty, tier, product_context.\n" f"VARY (make DIFFERENT so it's not a copy): name, profession, age_group, location, " f"background, income, lifestyle, personality, communication_style, opener, and any " f"surface small-talk. Keep it consistent with the locked traits (a customer with the " f"same pain would believably have a different name/job/life).\n" f"Output exactly one JSON object for the persona.\n" ) result = self.llm.complete_json( PERSONA_SYSTEM, prompt, temperature=0.9, max_tokens=3000 ) variant = result if isinstance(result, dict) else {} # Accept either a single persona object or a {"personas": [...]} container. if isinstance(variant.get("personas"), list) and variant["personas"]: variant = variant["personas"][0] if not isinstance(variant, dict) or not variant.get("name"): raise ValueError("variant generator returned no persona") # Lock the core traits regardless of what the LLM chose to change. for locked in ("pains", "objections", "negotiation_levers", "tolerance", "special", "recontact", "goal", "decision_timeline", "budget", "difficulty", "tier", "product_context"): if locked in source: variant[locked] = source.get(locked) variant.setdefault("initiation_mode", source.get("initiation_mode", "customer")) variant.setdefault("channel", source.get("channel", "social")) variant.setdefault("pains", source.get("pains", [])) return variant