"""Conversation playground, LLM refinement, audio effects, export/import, speak, MCP.""" from __future__ import annotations import asyncio import base64 import contextlib import io import json import re import threading import time import uuid import wave from pathlib import Path import requests from typing import Optional from fastapi import APIRouter, File, Form, HTTPException, Request, UploadFile from fastapi.responses import Response, StreamingResponse from core.config import _load_settings, _save_settings, _clean_preview_backend from core.constants import _VOICES_DIR_DEFAULT, _MAX_UPLOAD_BYTES from core.registry import _registry_get, TEMP_DIR from core.validation import _copy_limited from core.audio import _to_wav_16k from core.voice import ( _AUDIO_EXTS, _UPLOAD_EXTS, _PICTURE_EXTS, _find_voice_audio, _load_meta, _active_voices_dir, _voice_audio_files, _is_internal_voice_file, ) from core.tts_helpers import _preview_request_audio from routes.stt import _transcribe_audio, _clean_stt_backend router = APIRouter() # ── STT hallucination filter ────────────────────────────────────────────────── # Whisper commonly hallucinates these phrases on silence/noise. # Treat them as "no speech detected" rather than passing them to the LLM. _HALLUCINATIONS: frozenset[str] = frozenset([ "reich", "danke", "danke schön", "danke schoen", "vielen dank", "thank you", "thank you.", "thanks", "thanks.", "you", "you.", "copyright", "abonnieren", "untertitel", "subscribe", "subscribing", ]) def _is_hallucination(text: str) -> bool: t = text.strip().lower().rstrip(".!?,;:-").strip() return len(t) <= 2 or t in _HALLUCINATIONS # ── Sentence-boundary helpers for pipelined TTS ─────────────────────────────── _SENT_RE = re.compile(r'(?<=[.!?])\s+') _MIN_SENTENCE = 30 # min chars in buffer before we split def _sentence_split(buf: str) -> int: """Return the index after the first sentence boundary, or -1.""" if len(buf) < _MIN_SENTENCE: return -1 for m in _SENT_RE.finditer(buf): if m.end() >= _MIN_SENTENCE: return m.end() return -1 # ── LLM helpers ─────────────────────────────────────────────────────────────── def _rewrite_with_persona_sync(text: str, persona: str, llm_url: str, model: str = "") -> str: """Inline synchronous persona rewrite; raises RuntimeError on failure.""" settings = _load_settings() system = ( f"Rephrase the user's text as if spoken by this character: {persona}\n" "Keep the same meaning but adapt vocabulary, tone, and style to the character. " "Return ONLY the rephrased text — no quotes, no explanation." ) payload: dict = { "messages": [ {"role": "system", "content": system}, {"role": "user", "content": text}, ], "temperature": 0.3, "max_tokens": 512, } if model: payload["model"] = model resp = requests.post( f"{llm_url.rstrip('/')}/chat/completions", json=payload, headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=60, ) resp.raise_for_status() _msg = resp.json()["choices"][0]["message"] result = (_msg.get("content") or _msg.get("reasoning_content") or "").strip() if result.startswith('"') and result.endswith('"'): result = result[1:-1].strip() return result def _resolve_speak_voice(settings: dict, client_id: str, explicit_voice: str) -> str: if explicit_voice: return explicit_voice bindings: dict = settings.get("client_voice_bindings") or {} if client_id and client_id in bindings: return bindings[client_id] return settings.get("captures_default_voice") or "" # ── Routes ──────────────────────────────────────────────────────────────────── @router.post("/api/refine-text") async def refine_text(request: Request): """Clean up raw STT transcription using a local OpenAI-compatible LLM.""" data = await request.json() settings = _load_settings() text: str = (data.get("text") or "").strip() llm_url: str = (data.get("llm_url") or settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/") model: str = (data.get("model") or settings.get("refine_model") or settings.get("llm_model") or "").strip() toggles: dict = data.get("toggles") or {} if not text: raise HTTPException(400, "No text to refine") rules = [] if toggles.get("fillers", True): rules.append("Remove filler words (um, uh, like, you know, basically, literally, I mean, so, right, etc.)") if toggles.get("repetitions", True): rules.append("Remove repeated words and false starts (e.g. 'the the dog' → 'the dog', 'I was- I was going' → 'I was going')") if toggles.get("corrections", True): rules.append("Remove self-corrections and restarts, keeping only the final intended phrasing") if toggles.get("punctuation", True): rules.append("Fix punctuation, capitalisation, and sentence boundaries") if not rules: return {"text": text, "original": text} system = ( "You are a transcription cleanup assistant. " "Apply ONLY the following rules to the user's text. " "Return ONLY the cleaned text — no explanations, no quotes, no markdown:\n" + "\n".join(f"- {r}" for r in rules) ) payload: dict = { "messages": [ {"role": "system", "content": system}, {"role": "user", "content": text}, ], "temperature": 0.1, "max_tokens": 2048, } if model: payload["model"] = model try: resp = requests.post( f"{llm_url}/chat/completions", json=payload, headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=60, ) resp.raise_for_status() _msg = resp.json()["choices"][0]["message"] refined = (_msg.get("content") or _msg.get("reasoning_content") or "").strip() if refined.startswith('"') and refined.endswith('"'): refined = refined[1:-1].strip() return {"text": refined, "original": text} except Exception as e: raise HTTPException(502, f"LLM refinement failed: {e}") @router.post("/api/rewrite-with-persona") async def rewrite_with_persona(request: Request): """Rewrite user text in a voice persona's character using a local LLM.""" data = await request.json() settings = _load_settings() text: str = (data.get("text") or "").strip() persona: str = (data.get("persona") or "").strip() llm_url: str = (data.get("llm_url") or settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/") model: str = (data.get("model") or settings.get("llm_model") or "").strip() mode: str = (data.get("mode") or "rewrite").strip() if not persona: raise HTTPException(400, "No persona defined for this voice") if not text and mode != "compose": raise HTTPException(400, "No text provided") if mode == "compose": system = ( f"You are a voice assistant with this character: {persona}\n" "Write a single natural utterance in this character's voice about the topic given. " "Return ONLY the utterance — no quotes, no explanation." ) user_msg = text or "Introduce yourself briefly." temp = 0.9 else: system = ( f"Rephrase the user's text as if spoken by this character: {persona}\n" "Keep the same meaning but adapt vocabulary, tone, and style to the character. " "Return ONLY the rephrased text — no quotes, no explanation." ) user_msg = text temp = 0.3 payload: dict = { "messages": [ {"role": "system", "content": system}, {"role": "user", "content": user_msg}, ], "temperature": temp, "max_tokens": 512, } if model: payload["model"] = model try: resp = requests.post( f"{llm_url}/chat/completions", json=payload, headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600, ) resp.raise_for_status() _msg = resp.json()["choices"][0]["message"] result = (_msg.get("content") or _msg.get("reasoning_content") or "").strip() if result.startswith('"') and result.endswith('"'): result = result[1:-1].strip() return {"text": result, "original": text, "persona": persona} except Exception as e: raise HTTPException(502, f"LLM persona rewrite failed: {e}") @router.post("/api/analyze-characters") async def analyze_characters(request: Request): """Analyze a script with an LLM and return per-character voice descriptions. Used by the Script Rehearser to auto-design voices that match each role. Returns: {"characters": [{name, gender, language, age, description}, ...]} """ data = await request.json() script: str = (data.get("script") or "").strip() names: list = data.get("names") or [] language: str = (data.get("language") or "").strip() _settings = _load_settings() llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/") model: str = (data.get("model") or _settings.get("llm_model") or "").strip() if not script: raise HTTPException(400, "No script provided") if not names: raise HTTPException(400, "No character names provided") # Truncate script to leave room for output within typical model context windows. # German/non-English text tokenises at ~2.5–3 chars/token, so 6000 chars ≈ 2000–2400 tokens. if len(script) > 6000: script = script[:6000] lang_hint = f" The script language is {language}." if language else "" def _build_system() -> str: return ( "You are a casting director and TTS voice-design expert. " "Your job is to read a script, understand each character deeply from their " "dialogue, role, and context, then write a voice description that a " "text-to-speech model can use to generate a matching voice.\n" f"{lang_hint}\n" "For EVERY character in the provided list output:\n" "- name: exact name as given\n" "- gender: M, F, or N\n" "- language: spoken language of this character\n" "- age: estimated age range (e.g. 20s, 40s, elderly)\n" "- description: 2–3 sentences covering pitch (high/mid/low), pace, " "timbre, accent/dialect, emotional default, and any distinctive speech trait " "that fits the character's personality and role.\n" "Characters with few lines: infer from their role name and context.\n" "Respond with STRICT JSON only — no markdown, no explanation:\n" '{"characters":[{"name":"NAME","gender":"M|F|N","language":"LANG",' '"age":"30s","description":"voice description"}]}\n' "Use the exact character names provided.\n/no-think" ) def _call_llm(batch: list[str]) -> list[dict]: """Call the LLM for one batch of character names; return list of character dicts.""" user_msg = ( "Character names: " + ", ".join(str(n) for n in batch) + "\n\n" "Script:\n" + script ) # 150 tokens per character output, capped at 3072 to stay within 16K context mt = min(3072, max(512, len(batch) * 150)) payload: dict = { "messages": [ {"role": "system", "content": _build_system()}, {"role": "user", "content": user_msg}, ], "temperature": 0.4, "max_tokens": mt, } if model: payload["model"] = model resp = requests.post( f"{llm_url}/chat/completions", json=payload, headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600, ) resp.raise_for_status() _msg = resp.json()["choices"][0]["message"] raw = (_msg.get("content") or _msg.get("reasoning_content") or "").strip() content = re.sub(r".*?", "", raw, flags=re.DOTALL).strip() or raw for candidate in (content, _extract_json_block(content)): if not candidate: continue try: parsed = json.loads(candidate) if isinstance(parsed, dict) and "characters" in parsed: return parsed["characters"] except Exception: continue return [] # Process cast in batches of 10 to stay well within 16 K context BATCH = 10 all_characters: list[dict] = [] try: for i in range(0, len(names), BATCH): batch = names[i:i + BATCH] all_characters.extend(_call_llm(batch)) except Exception as e: raise HTTPException(502, f"LLM character analysis failed: {e}") if not all_characters: raise HTTPException(502, "LLM did not return valid character JSON") return {"characters": all_characters} @router.post("/api/match-characters-voices") async def match_characters_voices(request: Request): """Pick the best EXISTING library voice for each character (instead of designing new ones). Body: {script, names:[...], voices:[{id, gender, language, tags, description}], llm_url, model, language} Returns: {"assignments": [{name, voice_id, reason}]} """ data = await request.json() script: str = (data.get("script") or "").strip() names: list = data.get("names") or [] voices: list = data.get("voices") or [] language: str = (data.get("language") or "").strip() _settings = _load_settings() llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/") model: str = (data.get("model") or _settings.get("llm_model") or "").strip() if not names: raise HTTPException(400, "No character names provided") if not voices: raise HTTPException(400, "No candidate voices provided") if len(script) > 5000: script = script[:5000] def _vline(v: dict) -> str: meta = [str(v.get(k)) for k in ("gender", "language") if v.get(k)] if v.get("tags"): meta.append("tags:" + str(v["tags"])) desc = str(v.get("description") or v.get("name") or "")[:120] return f"- {v.get('id')} [{', '.join(meta)}] {desc}".rstrip() catalogue = "\n".join(_vline(v) for v in voices[:300]) valid_ids = {str(v.get("id")) for v in voices if v.get("id")} system = ( "You are a casting director assigning existing TTS voices to script characters. " f"{('Script language: ' + language + '. ') if language else ''}" "For EACH character, choose the single BEST voice_id from the CATALOGUE. " "RULE 1 — GENDER FIRST: the voice's gender MUST match the character's gender whenever the " "character's gender is clear from the script; only pick a different gender if no same-gender " "voice exists in the catalogue. " "RULE 2 — then match apparent age, language, personality, and the voice's description/tags. " "You MUST choose a voice_id that appears verbatim in the catalogue — never invent one. " "Reuse a voice for two characters only if no better distinct option exists. " "Respond with STRICT JSON only, no markdown:\n" '{"assignments":[{"name":"NAME","voice_id":"ID","reason":"short reason"}]}\n/no-think' ) user = ( "Characters to cast: " + ", ".join(str(n) for n in names) + "\n\n" "CATALOGUE (voice_id [gender, language, tags] description):\n" + catalogue + "\n\n" "Script excerpt:\n" + script ) payload: dict = { "messages": [{"role": "system", "content": system}, {"role": "user", "content": user}], "temperature": 0.3, "max_tokens": min(2048, max(512, len(names) * 60)), } if model: payload["model"] = model import time as _time raw = None last_err: Exception | None = None for attempt in range(3): try: resp = requests.post( f"{llm_url}/chat/completions", json=payload, headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600, ) resp.raise_for_status() _msg = resp.json()["choices"][0]["message"] raw = (_msg.get("content") or _msg.get("reasoning_content") or "").strip() break except requests.exceptions.ConnectionError as e: # llama-swap (and similar) often drop the first request while swapping/loading # the model — wait and retry rather than failing the cast. last_err = e _time.sleep(4 + attempt * 3) except Exception as e: last_err = e break if raw is None: raise HTTPException(502, f"LLM voice matching failed: {last_err}") content = re.sub(r".*?", "", raw, flags=re.DOTALL).strip() or raw assignments: list = [] for cand in (content, _extract_json_block(content)): if not cand: continue try: parsed = json.loads(cand) if isinstance(parsed, dict) and isinstance(parsed.get("assignments"), list): assignments = parsed["assignments"] break except Exception: continue # Keep only assignments that reference a real catalogue voice clean = [a for a in assignments if isinstance(a, dict) and str(a.get("voice_id")) in valid_ids] return {"assignments": clean} def _extract_json_block(text: str) -> str: """Pull the first {...} JSON object out of an LLM response.""" text = re.sub(r".*?", "", text, flags=re.DOTALL).strip() if text.startswith("```"): text = re.sub(r"^```[a-zA-Z]*\n?", "", text) text = re.sub(r"\n?```$", "", text).strip() start = text.find("{") end = text.rfind("}") if start != -1 and end != -1 and end > start: return text[start:end + 1] return "" @router.post("/api/character-sheets") async def character_sheets(request: Request): """Extract actor-facing RPG-style character sheets from a passage. Body: {text, known_characters:[...], language, llm_url, model} The text may contain "[p.N]" page markers so the model can cite sources. Returns: {sheets:[{name, aliases, archetype, physical, alignment, attribute_high, attribute_low, skills, inventory:[...], secret, conflict_style, win_condition, tier:"main"|"supporting", sources:[{page, quote}]}], characters:[names]} Deduced (not explicit) values are marked with a trailing " *". """ data = await request.json() text: str = (data.get("text") or "").strip() known: list = data.get("known_characters") or [] existing: str = (data.get("existing") or "").strip() # partial sheets so far (progressive fill) language: str = (data.get("language") or "").strip() _settings = _load_settings() llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/") model: str = (data.get("model") or _settings.get("llm_model") or "").strip() if not text: raise HTTPException(400, "No text provided") lang_hint = f" The text language is {language}; write the sheet in that language." if language else "" system = ( "You are an expert dramaturge, developmental editor, and tabletop RPG game master building rich " "character sheets passage by passage as a book is read. Extract playable, action-oriented sheets " "an actor can use to immediately know how to PLAY the character.\n" "PROGRESSIVE FILLING: you may be given the sheets built so far. For returning characters, ADD any " "NEW detail this passage reveals and refine vague fields; do not contradict solid earlier facts or " "blank out a field you cannot improve. Add brand-new characters as they appear. Leave a field empty " "if the book genuinely hasn't shown it yet (a later passage can fill it). Extrapolate from dialogue " "and actions when reasonable, and mark any deduced value with a trailing ' *'.\n" f"{lang_hint}\n" "For each character output these fields:\n" "- name, aliases\n" "- archetype: a two-word role summary (e.g. 'Ruthless Scholar')\n" "- physical: age, height, build, hair, eyes, skin, posture, gait, vocal quality. Use ONLY metric system.\n" "- clothing: distinctive clothing, armour, accessories — as observed in the text\n" "- alignment: strict moral code + the one line they will never cross\n" "- moral_alignment_score: integer 0–100. 100 = purely good/heroic, 0 = purely evil/villainous, 50 = neutral/ambiguous\n" "- arc_direction: one of: 'stable-good', 'stable-bad', 'neutral', 'good-to-bad', 'bad-to-good', 'complex'\n" "- arc_note: one sentence explaining the arc or moral position visible so far\n" "- attribute_high / attribute_low: highest and lowest natural attribute (Charisma, Intelligence, Wisdom, Agility…)\n" "- skills: what they are demonstrably good at in the story\n" "- capabilities: combat, magic, social, technical, or other demonstrated abilities\n" "- backstory: origin, formative background and history revealed in the text\n" "- relationships: key allies, family, rivals and enemies — name them and how they relate\n" "- motivation: the inner drive — WHY they pursue what they pursue (distinct from the win condition)\n" "- fears: their deepest fears, phobias or dread\n" "- mannerisms: habitual gestures, tics, body language, habits and quirks\n" "- voice_pattern: speech style — accent, pacing, vocabulary, register and verbal tics (for voice casting)\n" "- inventory: 1-3 defining items/props/clothing (array of short strings)\n" "- secret: dark secret or fatal flaw\n" "- conflict_style: fight, flight, or manipulate — how they act when cornered\n" "- win_condition: the specific event that would make them feel they have won\n" "- tier: 'main' or 'supporting'\n" "- sources: array of {page, quote, line_hint} — the page number from the nearest [p.N] marker, " "a short verbatim quote that supports the sheet (1-5 entries), and a brief label (e.g. 'appearance', 'alignment'). " "Use null page if unknown.\n" "Reuse the EXACT names from the known-characters list for returning characters. " "Respond with STRICT JSON only:\n" '{"sheets":[{"name":"","aliases":"","archetype":"","physical":"","clothing":"",' '"alignment":"","moral_alignment_score":50,"arc_direction":"neutral","arc_note":"",' '"attribute_high":"","attribute_low":"","skills":"","capabilities":"",' '"backstory":"","relationships":"","motivation":"","fears":"","mannerisms":"","voice_pattern":"",' '"inventory":[],"secret":"","conflict_style":"","win_condition":"",' '"tier":"main","sources":[{"page":1,"quote":"","line_hint":""}]}]}\n/no-think' ) user = ( ("Known characters so far: " + ", ".join(str(n) for n in known) + "\n\n" if known else "") + ("Sheets so far (fill gaps / refine; keep solid facts):\n" + existing + "\n\n" if existing else "") + "Passage:\n" + text ) payload: dict = { "messages": [ {"role": "system", "content": system}, {"role": "user", "content": user}, ], "temperature": 0.4, "max_tokens": 4096, } if model: payload["model"] = model try: resp = requests.post( f"{llm_url}/chat/completions", json=payload, headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600, ) resp.raise_for_status() _msg = resp.json()["choices"][0]["message"] raw = (_msg.get("content") or _msg.get("reasoning_content") or "").strip() except Exception as e: raise HTTPException(502, f"LLM character-sheet generation failed: {e}") content = re.sub(r".*?", "", raw, flags=re.DOTALL).strip() or raw sheets = [] for cand in (content, _extract_json_block(content)): if not cand: continue try: parsed = json.loads(cand) if isinstance(parsed, dict) and isinstance(parsed.get("sheets"), list): sheets = parsed["sheets"] break except Exception: continue clean, names = [], [] for s in sheets: if not isinstance(s, dict): continue name = str(s.get("name") or "").strip() if not name: continue inv = s.get("inventory") if isinstance(inv, str): inv = [x.strip() for x in inv.split(",") if x.strip()] elif not isinstance(inv, list): inv = [] src = s.get("sources") if isinstance(s.get("sources"), list) else [] # Clamp moral alignment score try: mas = int(s.get("moral_alignment_score") or 50) mas = max(0, min(100, mas)) except (TypeError, ValueError): mas = 50 arc = str(s.get("arc_direction") or "neutral").strip() if arc not in ("stable-good", "stable-bad", "neutral", "good-to-bad", "bad-to-good", "complex"): arc = "neutral" s.update({ "name": name, "inventory": inv[:3], "tier": "main" if str(s.get("tier") or "").lower().startswith("main") else "supporting", "sources": src[:5], "moral_alignment_score": mas, "arc_direction": arc, }) clean.append(s) names.append(name) return {"sheets": clean, "characters": names} @router.post("/api/character-deep-analysis") async def character_deep_analysis(request: Request): """Run a deep 5-area psychological analysis of a single character. Body: {name, role, goal, summary, text_excerpt, language, llm_url, model} Returns: {analysis: {core_flaw, agency, dialogue_voice, narrative_arc, paradox}} """ data = await request.json() name: str = (data.get("name") or "").strip() role: str = (data.get("role") or "unknown").strip() goal: str = (data.get("goal") or "").strip() summary: str = (data.get("summary") or "").strip() excerpt: str = (data.get("text_excerpt") or "").strip() language: str = (data.get("language") or "").strip() _settings = _load_settings() llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/") model: str = (data.get("model") or _settings.get("llm_model") or "").strip() if not name: raise HTTPException(400, "Character name required") lang_note = f" Write the analysis in {language}." if language else "" system = ( f"You are an expert developmental editor, literary coach, and psychologist specializing in " f"profound character studies.{lang_note} Provide a deep, multi-layered psychological and " f"narrative analysis of the character in exactly this JSON structure:\n" '{"core_flaw":"",' '"agency":"",' '"dialogue_voice":"