Two gaps observed in real casting output: 1. A mid-quote " - " (em-dash pause) was being read as the end of the quotation, handing the rest of the line to a wrong/new speaker. 2. Narration naming whose voice is about to speak (e.g. "Marcians Stimme wirkte nicht mehr so fest") wasn't being used to resolve the following unattributed line away from 'Unknown'. Added as rules 11/12 to the client-side AB_DEFAULT_PROMPT (the prompt actually sent to the LLM) and to normalizeCastingPrompt()'s migration so already-saved custom prompts pick them up automatically, same mechanism as the earlier Doppelpunkt-Regel upgrade. Also added equivalent rules to the server-side fallback prompt in routes/conversation.py for completeness, though the client-sent prompt is what's actually used in normal operation. Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2301 lines
109 KiB
Python
2301 lines
109 KiB
Python
"""Conversation playground, LLM refinement, audio effects, export/import, speak, MCP."""
|
||
from __future__ import annotations
|
||
|
||
import asyncio
|
||
import base64
|
||
import contextlib
|
||
import io
|
||
import json
|
||
import re
|
||
import threading
|
||
import time
|
||
import uuid
|
||
import wave
|
||
from pathlib import Path
|
||
|
||
import requests
|
||
from typing import Optional
|
||
from fastapi import APIRouter, File, Form, HTTPException, Request, UploadFile
|
||
from fastapi.responses import Response, StreamingResponse
|
||
|
||
from core.config import _load_settings, _save_settings, _clean_preview_backend
|
||
from core.constants import _VOICES_DIR_DEFAULT, _MAX_UPLOAD_BYTES
|
||
from core.registry import _registry_get, TEMP_DIR
|
||
from core.validation import _copy_limited
|
||
from core.audio import _to_wav_16k
|
||
from core.voice import (
|
||
_AUDIO_EXTS, _UPLOAD_EXTS, _PICTURE_EXTS,
|
||
_find_voice_audio, _load_meta, _active_voices_dir,
|
||
_voice_audio_files, _is_internal_voice_file,
|
||
)
|
||
from core.voice_index import rebuild_voice_index
|
||
from core.tts_helpers import _preview_request_audio
|
||
from routes.stt import _transcribe_audio, _clean_stt_backend
|
||
|
||
router = APIRouter()
|
||
|
||
# Serializes outbound calls to the audiobook-attribution LLM. A client-side
|
||
# timeout aborts the browser fetch but can't kill the backend's blocking
|
||
# requests.post thread, so a slow/cold model leaves "ghost" requests still
|
||
# occupying the LLM's single processing slot. Without this lock, the next
|
||
# chunk (or the retry-in-halves) fires into that busy slot and the LLM
|
||
# answers with 429s that cascade until the ghosts drain.
|
||
# threading.Lock (not asyncio.Lock) because the streaming endpoint's SSE
|
||
# generator runs in a plain thread (StreamingResponse iterates a sync
|
||
# generator via a threadpool), so both the blocking and streaming endpoints
|
||
# must share ONE lock object to actually serialize on the LLM's single slot —
|
||
# two separate locks would let a stream call and its own blocking fallback
|
||
# fire into that slot concurrently, which is exactly the "ghost request"
|
||
# scenario this lock exists to prevent.
|
||
_attribution_llm_lock = threading.Lock()
|
||
|
||
|
||
@contextlib.contextmanager
|
||
def _watchdog_close(resp, timeout_seconds):
|
||
"""Force-close `resp`'s connection if the wrapped block hasn't finished
|
||
within timeout_seconds.
|
||
|
||
requests' own `timeout=` on a stream=True call only guards the connect
|
||
+ first byte — NOT the gaps between subsequent body reads. If the LLM
|
||
backend goes silent mid-stream (no more chunks, connection left open),
|
||
the blocking socket recv() inside iter_lines() can hang forever, and
|
||
since that's a native blocking call rather than a Python-level `yield`
|
||
point, neither a wall-clock check in the loop body nor GeneratorExit from
|
||
a disconnected client can interrupt it — both only take effect at the
|
||
next bytecode boundary, which never arrives. Observed in production: a
|
||
stuck stream held the shared attribution lock for 11+ minutes, silently
|
||
starving every other passage. Closing the connection from a separate
|
||
watchdog thread forces the blocked recv() to raise, unblocking the
|
||
generator so the normal except/finally cleanup (incl. releasing the
|
||
lock) actually runs.
|
||
"""
|
||
done = threading.Event()
|
||
|
||
def _kill():
|
||
if not done.wait(timeout_seconds):
|
||
try:
|
||
resp.close()
|
||
except Exception:
|
||
pass
|
||
|
||
t = threading.Thread(target=_kill, daemon=True)
|
||
t.start()
|
||
try:
|
||
yield
|
||
finally:
|
||
done.set()
|
||
|
||
# ── STT hallucination filter ──────────────────────────────────────────────────
|
||
# Whisper commonly hallucinates these phrases on silence/noise.
|
||
# Treat them as "no speech detected" rather than passing them to the LLM.
|
||
_HALLUCINATIONS: frozenset[str] = frozenset([
|
||
"reich", "danke", "danke schön", "danke schoen", "vielen dank",
|
||
"thank you", "thank you.", "thanks", "thanks.", "you", "you.",
|
||
"copyright", "abonnieren", "untertitel", "subscribe", "subscribing",
|
||
])
|
||
|
||
def _is_hallucination(text: str) -> bool:
|
||
t = text.strip().lower().rstrip(".!?,;:-").strip()
|
||
return len(t) <= 2 or t in _HALLUCINATIONS
|
||
|
||
# ── Sentence-boundary helpers for pipelined TTS ───────────────────────────────
|
||
|
||
_SENT_RE = re.compile(r'(?<=[.!?])\s+')
|
||
_MIN_SENTENCE = 30 # min chars in buffer before we split
|
||
|
||
def _sentence_split(buf: str) -> int:
|
||
"""Return the index after the first sentence boundary, or -1."""
|
||
if len(buf) < _MIN_SENTENCE:
|
||
return -1
|
||
for m in _SENT_RE.finditer(buf):
|
||
if m.end() >= _MIN_SENTENCE:
|
||
return m.end()
|
||
return -1
|
||
|
||
# ── LLM helpers ───────────────────────────────────────────────────────────────
|
||
|
||
def _reasoning_text(obj: dict) -> str:
|
||
"""Extract chain-of-thought text from a message or delta dict. Most
|
||
OpenAI-compatible backends use `reasoning_content`, but some vLLM builds
|
||
(observed: vllm-0.23.1rc1) use a plain `reasoning` key instead — check both."""
|
||
return obj.get("reasoning_content") or obj.get("reasoning") or ""
|
||
|
||
|
||
def _rewrite_with_persona_sync(text: str, persona: str, llm_url: str, model: str = "") -> str:
|
||
"""Inline synchronous persona rewrite; raises RuntimeError on failure."""
|
||
settings = _load_settings()
|
||
system = (
|
||
f"Rephrase the user's text as if spoken by this character: {persona}\n"
|
||
"Keep the same meaning but adapt vocabulary, tone, and style to the character. "
|
||
"Return ONLY the rephrased text — no quotes, no explanation."
|
||
)
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": text},
|
||
],
|
||
"temperature": 0.3,
|
||
"max_tokens": 512,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
resp = requests.post(
|
||
f"{llm_url.rstrip('/')}/chat/completions",
|
||
json=payload,
|
||
headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"},
|
||
timeout=60,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
result = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
if result.startswith('"') and result.endswith('"'):
|
||
result = result[1:-1].strip()
|
||
return result
|
||
|
||
|
||
def _request_timeout_seconds(value, default: float = 600.0, minimum: float = 5.0, maximum: float = 600.0) -> float:
|
||
"""Clamp caller-provided LLM timeouts so UI recovery cannot hang indefinitely."""
|
||
try:
|
||
timeout = float(value)
|
||
except Exception:
|
||
timeout = default
|
||
return max(minimum, min(maximum, timeout))
|
||
|
||
|
||
def _is_router_model_alias(model: str) -> bool:
|
||
return bool(re.match(r"^\s*auto[-_ ]?router", model or "", re.I))
|
||
|
||
|
||
def _response_error_text(resp) -> str:
|
||
try:
|
||
text = resp.text or ""
|
||
except Exception:
|
||
text = ""
|
||
return text.strip()[:2000] or getattr(resp, "reason", "") or "Unknown upstream error"
|
||
|
||
|
||
def _llm_chat_completion_urls(llm_url: str) -> list[str]:
|
||
base = (llm_url or "http://localhost:11434/v1").rstrip("/")
|
||
urls: list[str] = []
|
||
if not re.search(r"/v\d+(?:\.\d+)?$", base):
|
||
urls.append(f"{base}/v1/chat/completions")
|
||
urls.append(f"{base}/chat/completions")
|
||
return list(dict.fromkeys(urls))
|
||
|
||
|
||
def _post_llm_chat_completion(llm_url: str, payload: dict, headers: dict, timeout: float):
|
||
last_resp = None
|
||
last_exc = None
|
||
for url in _llm_chat_completion_urls(llm_url):
|
||
try:
|
||
resp = requests.post(url, json=payload, headers=headers, timeout=timeout)
|
||
last_resp = resp
|
||
if resp.status_code in (404, 405) and url != _llm_chat_completion_urls(llm_url)[-1]:
|
||
continue
|
||
return resp
|
||
except Exception as exc:
|
||
last_exc = exc
|
||
if last_resp is not None:
|
||
return last_resp
|
||
if last_exc:
|
||
raise last_exc
|
||
raise RuntimeError("LLM request failed")
|
||
|
||
|
||
def _fallback_attribute_response(text: str) -> dict:
|
||
"""Deterministic fallback for attribution when an upstream LLM fails."""
|
||
quote_re = re.compile(
|
||
r"»([^«]+)«|«([^»]+)»|„([^“”]+)[“”]|“([^”]+)”|\"([^\"]+)\"|「([^」]+)」|『([^』]+)』",
|
||
re.DOTALL,
|
||
)
|
||
segments = []
|
||
pos = 0
|
||
for match in quote_re.finditer(text):
|
||
if match.start() > pos:
|
||
narr = text[pos:match.start()].strip()
|
||
if narr:
|
||
segments.append({"speaker": "Narrator", "type": "narration", "text": narr, "emotion": ""})
|
||
spoken = next((g for g in match.groups() if g), "")
|
||
if spoken.strip():
|
||
segments.append({"speaker": "Unknown", "type": "dialogue", "text": spoken.strip(), "emotion": ""})
|
||
pos = match.end()
|
||
tail = text[pos:].strip()
|
||
if tail:
|
||
segments.append({"speaker": "Narrator", "type": "narration", "text": tail, "emotion": ""})
|
||
if not segments:
|
||
segments = [{"speaker": "Narrator", "type": "narration", "text": text, "emotion": ""}]
|
||
chars = []
|
||
for seg in segments:
|
||
sp = seg.get("speaker")
|
||
if seg.get("type") == "dialogue" and sp and sp not in ("Narrator", "Unknown") and sp not in chars:
|
||
chars.append(sp)
|
||
return {"segments": segments, "characters": chars, "fallback": True}
|
||
|
||
|
||
def _resolve_speak_voice(settings: dict, client_id: str, explicit_voice: str) -> str:
|
||
if explicit_voice:
|
||
return explicit_voice
|
||
bindings: dict = settings.get("client_voice_bindings") or {}
|
||
if client_id and client_id in bindings:
|
||
return bindings[client_id]
|
||
return settings.get("captures_default_voice") or ""
|
||
|
||
|
||
# ── Routes ────────────────────────────────────────────────────────────────────
|
||
|
||
@router.post("/api/refine-text")
|
||
async def refine_text(request: Request):
|
||
"""Clean up raw STT transcription using a local OpenAI-compatible LLM."""
|
||
data = await request.json()
|
||
settings = _load_settings()
|
||
text: str = (data.get("text") or "").strip()
|
||
llm_url: str = (data.get("llm_url") or settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or settings.get("refine_model") or settings.get("llm_model") or "").strip()
|
||
toggles: dict = data.get("toggles") or {}
|
||
|
||
if not text:
|
||
raise HTTPException(400, "No text to refine")
|
||
|
||
rules = []
|
||
if toggles.get("fillers", True):
|
||
rules.append("Remove filler words (um, uh, like, you know, basically, literally, I mean, so, right, etc.)")
|
||
if toggles.get("repetitions", True):
|
||
rules.append("Remove repeated words and false starts (e.g. 'the the dog' → 'the dog', 'I was- I was going' → 'I was going')")
|
||
if toggles.get("corrections", True):
|
||
rules.append("Remove self-corrections and restarts, keeping only the final intended phrasing")
|
||
if toggles.get("punctuation", True):
|
||
rules.append("Fix punctuation, capitalisation, and sentence boundaries")
|
||
if not rules:
|
||
return {"text": text, "original": text}
|
||
|
||
system = (
|
||
"You are a transcription cleanup assistant. "
|
||
"Apply ONLY the following rules to the user's text. "
|
||
"Return ONLY the cleaned text — no explanations, no quotes, no markdown:\n"
|
||
+ "\n".join(f"- {r}" for r in rules)
|
||
)
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": text},
|
||
],
|
||
"temperature": 0.1,
|
||
"max_tokens": 2048,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
|
||
try:
|
||
resp = requests.post(
|
||
f"{llm_url}/chat/completions",
|
||
json=payload,
|
||
headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"},
|
||
timeout=60,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
refined = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
if refined.startswith('"') and refined.endswith('"'):
|
||
refined = refined[1:-1].strip()
|
||
return {"text": refined, "original": text}
|
||
except Exception as e:
|
||
raise HTTPException(502, f"LLM refinement failed: {e}")
|
||
|
||
|
||
@router.post("/api/rewrite-with-persona")
|
||
async def rewrite_with_persona(request: Request):
|
||
"""Rewrite user text in a voice persona's character using a local LLM."""
|
||
data = await request.json()
|
||
settings = _load_settings()
|
||
text: str = (data.get("text") or "").strip()
|
||
persona: str = (data.get("persona") or "").strip()
|
||
llm_url: str = (data.get("llm_url") or settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or settings.get("llm_model") or "").strip()
|
||
mode: str = (data.get("mode") or "rewrite").strip()
|
||
|
||
if not persona:
|
||
raise HTTPException(400, "No persona defined for this voice")
|
||
if not text and mode != "compose":
|
||
raise HTTPException(400, "No text provided")
|
||
|
||
if mode == "compose":
|
||
system = (
|
||
f"You are a voice assistant with this character: {persona}\n"
|
||
"Write a single natural utterance in this character's voice about the topic given. "
|
||
"Return ONLY the utterance — no quotes, no explanation."
|
||
)
|
||
user_msg = text or "Introduce yourself briefly."
|
||
temp = 0.9
|
||
else:
|
||
system = (
|
||
f"Rephrase the user's text as if spoken by this character: {persona}\n"
|
||
"Keep the same meaning but adapt vocabulary, tone, and style to the character. "
|
||
"Return ONLY the rephrased text — no quotes, no explanation."
|
||
)
|
||
user_msg = text
|
||
temp = 0.3
|
||
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": user_msg},
|
||
],
|
||
"temperature": temp,
|
||
"max_tokens": 512,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
|
||
try:
|
||
resp = requests.post(
|
||
f"{llm_url}/chat/completions",
|
||
json=payload,
|
||
headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"},
|
||
timeout=600,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
result = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
if result.startswith('"') and result.endswith('"'):
|
||
result = result[1:-1].strip()
|
||
return {"text": result, "original": text, "persona": persona}
|
||
except Exception as e:
|
||
raise HTTPException(502, f"LLM persona rewrite failed: {e}")
|
||
|
||
|
||
@router.post("/api/analyze-characters")
|
||
async def analyze_characters(request: Request):
|
||
"""Analyze a script with an LLM and return per-character voice descriptions.
|
||
|
||
Used by the Script Rehearser to auto-design voices that match each role.
|
||
Returns: {"characters": [{name, gender, language, age, description}, ...]}
|
||
"""
|
||
data = await request.json()
|
||
script: str = (data.get("script") or "").strip()
|
||
names: list = data.get("names") or []
|
||
language: str = (data.get("language") or "").strip()
|
||
_settings = _load_settings()
|
||
llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or _settings.get("llm_model") or "").strip()
|
||
|
||
if not script:
|
||
raise HTTPException(400, "No script provided")
|
||
if not names:
|
||
raise HTTPException(400, "No character names provided")
|
||
|
||
# Truncate script to leave room for output within typical model context windows.
|
||
# German/non-English text tokenises at ~2.5–3 chars/token, so 6000 chars ≈ 2000–2400 tokens.
|
||
if len(script) > 6000:
|
||
script = script[:6000]
|
||
|
||
lang_hint = f" The script language is {language}." if language else ""
|
||
|
||
def _build_system() -> str:
|
||
return (
|
||
"You are a casting director and TTS voice-design expert. "
|
||
"Your job is to read a script, understand each character deeply from their "
|
||
"dialogue, role, and context, then write a voice description that a "
|
||
"text-to-speech model can use to generate a matching voice.\n"
|
||
f"{lang_hint}\n"
|
||
"For EVERY character in the provided list output:\n"
|
||
"- name: exact name as given\n"
|
||
"- gender: M, F, or N\n"
|
||
"- language: spoken language of this character\n"
|
||
"- age: estimated age range (e.g. 20s, 40s, elderly)\n"
|
||
"- description: 2–3 sentences covering pitch (high/mid/low), pace, "
|
||
"timbre, accent/dialect, emotional default, and any distinctive speech trait "
|
||
"that fits the character's personality and role.\n"
|
||
"Characters with few lines: infer from their role name and context.\n"
|
||
"Respond with STRICT JSON only — no markdown, no explanation:\n"
|
||
'{"characters":[{"name":"NAME","gender":"M|F|N","language":"LANG",'
|
||
'"age":"30s","description":"voice description"}]}\n'
|
||
"Use the exact character names provided.\n/no-think"
|
||
)
|
||
|
||
def _call_llm(batch: list[str]) -> list[dict]:
|
||
"""Call the LLM for one batch of character names; return list of character dicts."""
|
||
user_msg = (
|
||
"Character names: " + ", ".join(str(n) for n in batch) + "\n\n"
|
||
"Script:\n" + script
|
||
)
|
||
# 150 tokens per character output, capped at 3072 to stay within 16K context
|
||
mt = min(3072, max(512, len(batch) * 150))
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": _build_system()},
|
||
{"role": "user", "content": user_msg},
|
||
],
|
||
"temperature": 0.4,
|
||
"max_tokens": mt,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
resp = requests.post(
|
||
f"{llm_url}/chat/completions",
|
||
json=payload,
|
||
headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"},
|
||
timeout=600,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
raw = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
content = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL).strip() or raw
|
||
for candidate in (content, _extract_json_block(content)):
|
||
if not candidate:
|
||
continue
|
||
try:
|
||
parsed = json.loads(candidate)
|
||
if isinstance(parsed, dict) and "characters" in parsed:
|
||
return parsed["characters"]
|
||
except Exception:
|
||
continue
|
||
return []
|
||
|
||
# Process cast in batches of 10 to stay well within 16 K context
|
||
BATCH = 10
|
||
all_characters: list[dict] = []
|
||
try:
|
||
for i in range(0, len(names), BATCH):
|
||
batch = names[i:i + BATCH]
|
||
all_characters.extend(_call_llm(batch))
|
||
except Exception as e:
|
||
raise HTTPException(502, f"LLM character analysis failed: {e}")
|
||
|
||
if not all_characters:
|
||
raise HTTPException(502, "LLM did not return valid character JSON")
|
||
|
||
return {"characters": all_characters}
|
||
|
||
|
||
@router.post("/api/match-characters-voices")
|
||
async def match_characters_voices(request: Request):
|
||
"""Pick the best EXISTING library voice for each character (instead of designing new ones).
|
||
|
||
Body: {script, names:[...], voices:[{id, gender, language, tags, description}], llm_url, model, language}
|
||
Returns: {"assignments": [{name, voice_id, reason}]}
|
||
"""
|
||
data = await request.json()
|
||
script: str = (data.get("script") or "").strip()
|
||
names: list = data.get("names") or []
|
||
voices: list = data.get("voices") or []
|
||
language: str = (data.get("language") or "").strip()
|
||
_settings = _load_settings()
|
||
llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or _settings.get("llm_model") or "").strip()
|
||
|
||
if not names:
|
||
raise HTTPException(400, "No character names provided")
|
||
if not voices:
|
||
raise HTTPException(400, "No candidate voices provided")
|
||
if len(script) > 5000:
|
||
script = script[:5000]
|
||
|
||
def _vline(v: dict) -> str:
|
||
meta = [str(v.get(k)) for k in ("gender", "language") if v.get(k)]
|
||
if v.get("tags"):
|
||
meta.append("tags:" + str(v["tags"]))
|
||
desc = str(v.get("description") or v.get("name") or "")[:120]
|
||
return f"- {v.get('id')} [{', '.join(meta)}] {desc}".rstrip()
|
||
|
||
catalogue = "\n".join(_vline(v) for v in voices[:300])
|
||
valid_ids = {str(v.get("id")) for v in voices if v.get("id")}
|
||
|
||
system = (
|
||
"You are a casting director assigning existing TTS voices to script characters. "
|
||
f"{('Script language: ' + language + '. ') if language else ''}"
|
||
"For EACH character, choose the single BEST voice_id from the CATALOGUE. "
|
||
"RULE 1 — GENDER FIRST: the voice's gender MUST match the character's gender whenever the "
|
||
"character's gender is clear from the script; only pick a different gender if no same-gender "
|
||
"voice exists in the catalogue. "
|
||
"RULE 2 — then match apparent age, language, personality, and the voice's description/tags. "
|
||
"You MUST choose a voice_id that appears verbatim in the catalogue — never invent one. "
|
||
"Reuse a voice for two characters only if no better distinct option exists. "
|
||
"Respond with STRICT JSON only, no markdown:\n"
|
||
'{"assignments":[{"name":"NAME","voice_id":"ID","reason":"short reason"}]}\n/no-think'
|
||
)
|
||
user = (
|
||
"Characters to cast: " + ", ".join(str(n) for n in names) + "\n\n"
|
||
"CATALOGUE (voice_id [gender, language, tags] description):\n" + catalogue + "\n\n"
|
||
"Script excerpt:\n" + script
|
||
)
|
||
payload: dict = {
|
||
"messages": [{"role": "system", "content": system}, {"role": "user", "content": user}],
|
||
"temperature": 0.3,
|
||
"max_tokens": min(2048, max(512, len(names) * 60)),
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
import time as _time
|
||
raw = None
|
||
last_err: Exception | None = None
|
||
for attempt in range(3):
|
||
try:
|
||
resp = requests.post(
|
||
f"{llm_url}/chat/completions", json=payload,
|
||
headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
raw = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
break
|
||
except requests.exceptions.ConnectionError as e:
|
||
# llama-swap (and similar) often drop the first request while swapping/loading
|
||
# the model — wait and retry rather than failing the cast.
|
||
last_err = e
|
||
_time.sleep(4 + attempt * 3)
|
||
except Exception as e:
|
||
last_err = e
|
||
break
|
||
if raw is None:
|
||
raise HTTPException(502, f"LLM voice matching failed: {last_err}")
|
||
|
||
content = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL).strip() or raw
|
||
assignments: list = []
|
||
for cand in (content, _extract_json_block(content)):
|
||
if not cand:
|
||
continue
|
||
try:
|
||
parsed = json.loads(cand)
|
||
if isinstance(parsed, dict) and isinstance(parsed.get("assignments"), list):
|
||
assignments = parsed["assignments"]
|
||
break
|
||
except Exception:
|
||
continue
|
||
# Keep only assignments that reference a real catalogue voice
|
||
clean = [a for a in assignments if isinstance(a, dict) and str(a.get("voice_id")) in valid_ids]
|
||
return {"assignments": clean}
|
||
|
||
|
||
def _extract_json_block(text: str) -> str:
|
||
"""Pull the first {...} JSON object out of an LLM response."""
|
||
text = re.sub(r"<think>.*?</think>", "", text, flags=re.DOTALL).strip()
|
||
if text.startswith("```"):
|
||
text = re.sub(r"^```[a-zA-Z]*\n?", "", text)
|
||
text = re.sub(r"\n?```$", "", text).strip()
|
||
start = text.find("{")
|
||
end = text.rfind("}")
|
||
if start != -1 and end != -1 and end > start:
|
||
return text[start:end + 1]
|
||
return ""
|
||
|
||
|
||
def _charsheets_prepare(data: dict) -> dict:
|
||
"""Resolve settings and build the chat payload for one character-sheets
|
||
extraction request. Shared by the blocking endpoint and the streaming
|
||
(watch-it-fill-out) endpoint so the prompt can never drift between them."""
|
||
text: str = (data.get("text") or "").strip()
|
||
known: list = data.get("known_characters") or []
|
||
target_mode: bool = bool(data.get("target_mode"))
|
||
existing: str = (data.get("existing") or "").strip() # partial sheets so far (progressive fill)
|
||
language: str = (data.get("language") or "").strip()
|
||
_settings = _load_settings()
|
||
llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or _settings.get("llm_model") or "").strip()
|
||
timeout_seconds = _request_timeout_seconds(data.get("timeout_seconds"), 600.0)
|
||
if not text:
|
||
raise HTTPException(400, "No text provided")
|
||
|
||
lang_instruction = (
|
||
f"⚠️ SPRACHE / LANGUAGE — OBLIGATORISCH / MANDATORY:\n"
|
||
f"The source text is in {language}.\n"
|
||
f"YOU MUST write EVERY descriptive field value in {language}. This is non-negotiable.\n"
|
||
f"Fields that MUST be in {language}: physical, clothing, alignment, arc_note, skills, "
|
||
f"capabilities, backstory, relationships, motivation, fears, mannerisms, voice_pattern, secret, "
|
||
f"conflict_style, win_condition, archetype, aliases, first_name, last_name, full_name, title.\n"
|
||
f"voice_design_prompt and image_prompt should be concise English tool prompts for generation tools.\n"
|
||
f"JSON field KEYS stay in English. Character names stay exactly as they appear in the text.\n"
|
||
f"If you write any descriptive value in English instead of {language}, your response is WRONG.\n\n"
|
||
) if language else ""
|
||
target_instruction = (
|
||
"CAST TARGET MODE:\n"
|
||
"- The known-characters list is the already recognized cast roster from the audiobook pass. "
|
||
"Treat it as TARGET PROFILES to fill, not as a list of aliases and not as proof that those names are the same person.\n"
|
||
"- Prefer outputting sheets for those target names when this passage reveals usable profile details. "
|
||
"Do not output a blank sheet just because a target is listed.\n"
|
||
"- If the passage proves that a target has another name/title/nickname, keep ONE profile using the best known target name "
|
||
"and put only the proven alternate form in aliases/title/full_name.\n"
|
||
"- You may add a brand-new character only when the passage clearly introduces one that is not in the target roster.\n"
|
||
"- NEVER copy the known-characters roster into one character's aliases, relationships, title, or description fields.\n\n"
|
||
) if target_mode and known else ""
|
||
system = (
|
||
f"{lang_instruction}"
|
||
f"{target_instruction}"
|
||
"You are an expert dramaturge, developmental editor, and tabletop RPG game master building rich "
|
||
"character sheets passage by passage as a book is read. Extract playable, action-oriented sheets "
|
||
"an actor can use to immediately know how to PLAY the character.\n"
|
||
"PROGRESSIVE FILLING: you may be given the sheets built so far. For returning characters, ADD any "
|
||
"NEW detail this passage reveals and refine vague fields; do not contradict solid earlier facts or "
|
||
"blank out a field you cannot improve. Add brand-new characters as they appear. Leave a field empty "
|
||
"if the book genuinely hasn't shown it yet (a later passage can fill it). Extrapolate from dialogue "
|
||
"and actions when reasonable, and mark any deduced value with a trailing ' *'.\n"
|
||
"For each character output these fields:\n"
|
||
"- name: canonical display name for this one character. Use the real personal name if known; otherwise use the most stable role/title.\n"
|
||
"- aliases: ONLY alternate names, roles, epithets, mistranscriptions, and titles proven to refer to the SAME character, comma-separated (max 6 items; e.g. 'Henker, Vampir, Zerwas der Henker'). Leave empty when uncertain.\n"
|
||
"- first_name, last_name, full_name, title: split the character identity when known. Leave unknown parts empty. Put noble/office/role labels in title (e.g. 'Henker', 'Vampir', 'Graf').\n"
|
||
"- archetype: a two-word role summary (e.g. 'Ruthless Scholar')\n"
|
||
"- gender: 'male', 'female', or 'nonbinary' — as apparent from the text (pronouns, roles, physical description). Leave empty if genuinely indeterminable.\n"
|
||
"- physical: age, height, build, hair, eyes, skin, posture, gait, vocal quality. Use ONLY metric system.\n"
|
||
"- clothing: distinctive clothing, armour, accessories — as observed in the text\n"
|
||
"- alignment: strict moral code + the one line they will never cross\n"
|
||
"- moral_alignment_score: integer 0–100. 100 = purely good/heroic, 0 = purely evil/villainous, 50 = neutral/ambiguous\n"
|
||
"- arc_direction: one of: 'stable-good', 'stable-bad', 'neutral', 'good-to-bad', 'bad-to-good', 'complex'\n"
|
||
"- arc_note: one sentence explaining the arc or moral position visible so far\n"
|
||
"- attribute_high / attribute_low: highest and lowest natural attribute (Charisma, Intelligence, Wisdom, Agility…)\n"
|
||
"- skills: what they are demonstrably good at in the story\n"
|
||
"- capabilities: combat, magic, social, technical, or other demonstrated abilities\n"
|
||
"- backstory: origin, formative background and history revealed in the text\n"
|
||
"- relationships: key allies, family, rivals and enemies — name them and how they relate\n"
|
||
"- motivation: the inner drive — WHY they pursue what they pursue (distinct from the win condition)\n"
|
||
"- fears: their deepest fears, phobias or dread\n"
|
||
"- mannerisms: habitual gestures, tics, body language, habits and quirks\n"
|
||
"- voice_pattern: speech style — accent, pacing, vocabulary, register and verbal tics (for voice casting)\n"
|
||
"- voice_design_prompt: concise English Qwen voice-design prompt (15-45 words). Include age impression, gender/androgyny if inferable, pitch, timbre, pace, accent/register, emotional baseline and suitability for audiobook dialogue. Do NOT mention plot spoilers.\n"
|
||
"- image_prompt: detailed English image-generation prompt for this character. Include face, age impression, build, hair/eyes/skin if known, clothing, posture, props, mood, genre/style, and visible symbols. Mark inferred traits with '*'.\n"
|
||
"- inventory: 1-3 defining items/props/clothing (array of short strings)\n"
|
||
"- secret: dark secret or fatal flaw\n"
|
||
"- conflict_style: fight, flight, or manipulate — how they act when cornered\n"
|
||
"- win_condition: the specific event that would make them feel they have won\n"
|
||
"- tier: 'main' or 'supporting'\n"
|
||
"- sources: array of {page, quote, line_hint} — the page number from the nearest [p.N] marker, "
|
||
"a short verbatim quote that supports the sheet (1-12 entries), and a brief label (e.g. 'physical', 'clothing', 'relationships', 'motivation'). "
|
||
"Use null page if unknown. line_hint MUST name the supported field when possible: physical, clothing, relationships, motivation, fears, mannerisms, voice_pattern, backstory, alignment, skills, capabilities, secret, conflict_style, or win_condition.\n"
|
||
"IDENTITY MERGING: A character may appear under multiple names in the book (first name, last name, title, role, alias, nickname). "
|
||
"Examples: 'Zerwas', 'Henker', and 'Vampir' may all refer to ONE profile if context shows they are the same person. "
|
||
"Do NOT create separate sheets for aliases/titles of the same person; put the alternate forms in aliases/title/full_name and keep one canonical name.\n"
|
||
"Reuse the EXACT names from the known-characters list for returning characters when they are the canonical name or an alias of this character. "
|
||
"Do not merge characters merely because their names appear near each other, in the known-character list, or in relationships.\n"
|
||
"Respond with STRICT JSON only:\n"
|
||
'{"sheets":[{"name":"","aliases":"","first_name":"","last_name":"","full_name":"","title":"","archetype":"","gender":"","physical":"","clothing":"",'
|
||
'"alignment":"","moral_alignment_score":50,"arc_direction":"neutral","arc_note":"",'
|
||
'"attribute_high":"","attribute_low":"","skills":"","capabilities":"",'
|
||
'"backstory":"","relationships":"","motivation":"","fears":"","mannerisms":"","voice_pattern":"","voice_design_prompt":"","image_prompt":"",'
|
||
'"inventory":[],"secret":"","conflict_style":"","win_condition":"",'
|
||
'"tier":"main","sources":[{"page":1,"quote":"","line_hint":""}]}]}\n/no-think'
|
||
)
|
||
lang_reminder = (
|
||
f"⚠️ WICHTIG: Alle beschreibenden Feldwerte MÜSSEN auf {language} geschrieben werden. "
|
||
f"Kein einziges beschreibendes Feld darf auf Englisch sein. Nur JSON-Schlüssel bleiben Englisch.\n\n"
|
||
) if language else ""
|
||
user = (
|
||
("Known characters so far: " + ", ".join(str(n) for n in known) + "\n\n" if known else "")
|
||
+ ("Sheets so far (fill gaps / refine; keep solid facts):\n" + existing + "\n\n" if existing else "")
|
||
+ lang_reminder
|
||
+ "Passage:\n" + text
|
||
)
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": user},
|
||
],
|
||
"temperature": 0.4,
|
||
"max_tokens": 4096,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
return {
|
||
"payload": payload, "llm_url": llm_url, "model": model,
|
||
"api_key": _settings.get("llm_api_key") or "sk-dummy-key",
|
||
"timeout_seconds": timeout_seconds,
|
||
}
|
||
|
||
|
||
def _charsheets_parse(raw: str) -> dict:
|
||
"""Turn the LLM's raw answer into normalised {sheets, characters}."""
|
||
content = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL).strip() or raw
|
||
sheets = []
|
||
for cand in (content, _extract_json_block(content)):
|
||
if not cand:
|
||
continue
|
||
try:
|
||
parsed = json.loads(cand)
|
||
if isinstance(parsed, dict) and isinstance(parsed.get("sheets"), list):
|
||
sheets = parsed["sheets"]
|
||
break
|
||
except Exception:
|
||
continue
|
||
clean, names = [], []
|
||
alias_splitter = re.compile(r"[,;/|]|\baka\b|\baka\.\b|\balias(?:es)?\b|\bgenannt\b|\bnamens\b|\bcalled\b|\bknown as\b", re.I)
|
||
for s in sheets:
|
||
if not isinstance(s, dict):
|
||
continue
|
||
name = str(s.get("name") or "").strip()
|
||
if not name:
|
||
continue
|
||
alias_raw = str(s.get("aliases") or "").strip()
|
||
alias_parts = [p.strip() for p in alias_splitter.split(alias_raw) if p.strip()]
|
||
alias_parts = [p for p in alias_parts if len(p) <= 80 and not re.match(r"^(needs?:|complete$)", p, re.I)]
|
||
if len(alias_raw) > 500 or len(alias_parts) > 12:
|
||
aliases = ""
|
||
else:
|
||
aliases = ", ".join(alias_parts[:6])
|
||
inv = s.get("inventory")
|
||
if isinstance(inv, str):
|
||
inv = [x.strip() for x in inv.split(",") if x.strip()]
|
||
elif not isinstance(inv, list):
|
||
inv = []
|
||
src_raw = s.get("sources") if isinstance(s.get("sources"), list) else []
|
||
src = []
|
||
for item in src_raw:
|
||
if not isinstance(item, dict):
|
||
continue
|
||
quote = str(item.get("quote") or "").strip()
|
||
if not quote:
|
||
continue
|
||
try:
|
||
page = int(item.get("page")) if item.get("page") is not None else None
|
||
except (TypeError, ValueError):
|
||
page = None
|
||
src.append({
|
||
"page": page,
|
||
"quote": quote[:240],
|
||
"line_hint": str(item.get("line_hint") or "").strip()[:60],
|
||
})
|
||
# Clamp moral alignment score
|
||
try:
|
||
mas = int(s.get("moral_alignment_score") or 50)
|
||
mas = max(0, min(100, mas))
|
||
except (TypeError, ValueError):
|
||
mas = 50
|
||
arc = str(s.get("arc_direction") or "neutral").strip()
|
||
if arc not in ("stable-good", "stable-bad", "neutral", "good-to-bad", "bad-to-good", "complex"):
|
||
arc = "neutral"
|
||
gender = str(s.get("gender") or "").strip().lower()
|
||
if gender not in ("male", "female", "nonbinary"):
|
||
gender = ""
|
||
s.update({
|
||
"name": name, "aliases": aliases, "inventory": inv[:3],
|
||
"tier": "main" if str(s.get("tier") or "").lower().startswith("main") else "supporting",
|
||
"sources": src[:12],
|
||
"moral_alignment_score": mas,
|
||
"arc_direction": arc,
|
||
"gender": gender,
|
||
})
|
||
clean.append(s)
|
||
names.append(name)
|
||
return {"sheets": clean, "characters": names}
|
||
|
||
|
||
@router.post("/api/character-sheets")
|
||
async def character_sheets(request: Request):
|
||
"""Extract actor-facing RPG-style character sheets from a passage.
|
||
|
||
Body: {text, known_characters:[...], language, llm_url, model}
|
||
The text may contain "[p.N]" page markers so the model can cite sources.
|
||
Returns: {sheets:[{name, aliases, first_name, last_name, full_name, title, archetype, physical, alignment,
|
||
attribute_high, attribute_low, skills, inventory:[...], secret,
|
||
conflict_style, win_condition, tier:"main"|"supporting",
|
||
sources:[{page, quote}]}], characters:[names]}
|
||
Deduced (not explicit) values are marked with a trailing " *".
|
||
"""
|
||
data = await request.json()
|
||
prep = _charsheets_prepare(data)
|
||
try:
|
||
if not await asyncio.to_thread(_attribution_llm_lock.acquire, True, prep["timeout_seconds"]):
|
||
raise HTTPException(503, "Attribution engine busy")
|
||
try:
|
||
resp = await asyncio.to_thread(
|
||
_post_llm_chat_completion,
|
||
prep["llm_url"], prep["payload"],
|
||
{"Authorization": f"Bearer {prep['api_key']}"}, prep["timeout_seconds"],
|
||
)
|
||
finally:
|
||
_attribution_llm_lock.release()
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
raw = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
except HTTPException:
|
||
raise
|
||
except Exception as e:
|
||
raise HTTPException(502, f"LLM character-sheet generation failed: {e}")
|
||
return _charsheets_parse(raw)
|
||
|
||
|
||
@router.post("/api/character-sheets/stream")
|
||
async def character_sheets_stream(request: Request):
|
||
"""Same job as /api/character-sheets, but streams the LLM's raw output
|
||
(its JSON answer being written field by field) as SSE `{"t": "..."}`
|
||
events, ending with `{"done": true, "result": {...}}` — lets the UI show
|
||
the sheet actually filling out passage by passage instead of just a
|
||
progress bar. On upstream failure it emits `{"error": "..."}` and the
|
||
client falls back to the blocking endpoint."""
|
||
data = await request.json()
|
||
prep = _charsheets_prepare(data)
|
||
payload = dict(prep["payload"])
|
||
payload["stream"] = True
|
||
headers = {"Authorization": f"Bearer {prep['api_key']}"}
|
||
|
||
def gen():
|
||
if not _attribution_llm_lock.acquire(timeout=prep["timeout_seconds"]):
|
||
yield 'data: {"error": "Attribution engine busy"}\n\n'
|
||
return
|
||
upstream = None
|
||
try:
|
||
urls = _llm_chat_completion_urls(prep["llm_url"])
|
||
last_err = None
|
||
for u in urls:
|
||
try:
|
||
upstream = requests.post(
|
||
u, json=payload, headers=headers,
|
||
timeout=(15, prep["timeout_seconds"]), stream=True,
|
||
)
|
||
if upstream.status_code in (404, 405) and u != urls[-1]:
|
||
upstream.close()
|
||
upstream = None
|
||
continue
|
||
break
|
||
except Exception as exc:
|
||
last_err = exc
|
||
upstream = None
|
||
if upstream is None:
|
||
print(f"[character-sheets/stream] LLM connect failed: {last_err}")
|
||
yield f'data: {json.dumps({"error": str(last_err or "LLM connect failed")})}\n\n'
|
||
return
|
||
if upstream.status_code >= 400:
|
||
print(
|
||
f"[character-sheets/stream] LLM failed ({upstream.status_code}) for model "
|
||
f"{payload.get('model') or '(default)'}: {_response_error_text(upstream)[:300]}"
|
||
)
|
||
yield f'data: {json.dumps({"error": f"HTTP {upstream.status_code}: {_response_error_text(upstream)[:300]}"})}\n\n'
|
||
return
|
||
raw_parts = []
|
||
upstream.encoding = "utf-8"
|
||
# Two layers against a stuck stream: the wall-clock check catches
|
||
# a model that keeps trickling chunks past its budget, and
|
||
# _watchdog_close catches the connection going fully silent (which
|
||
# the in-loop check can't see, since a blocked socket read never
|
||
# reaches it — see _watchdog_close's docstring for why).
|
||
deadline = time.monotonic() + prep["timeout_seconds"]
|
||
with _watchdog_close(upstream, prep["timeout_seconds"]):
|
||
for line in upstream.iter_lines(decode_unicode=True):
|
||
if time.monotonic() > deadline:
|
||
print(f"[character-sheets/stream] wall-clock deadline hit after {prep['timeout_seconds']}s, aborting stream")
|
||
yield 'data: {"error": "LLM stream exceeded timeout"}\n\n'
|
||
return
|
||
if not line or not line.startswith("data:"):
|
||
continue
|
||
chunk = line[5:].strip()
|
||
if chunk == "[DONE]":
|
||
break
|
||
try:
|
||
delta = json.loads(chunk)["choices"][0]["delta"]
|
||
except Exception:
|
||
continue
|
||
t = _reasoning_text(delta) or delta.get("content") or ""
|
||
if delta.get("content"):
|
||
raw_parts.append(delta["content"])
|
||
if t:
|
||
yield f'data: {json.dumps({"t": t})}\n\n'
|
||
result = _charsheets_parse("".join(raw_parts))
|
||
yield f'data: {json.dumps({"done": True, "result": result})}\n\n'
|
||
except GeneratorExit:
|
||
raise
|
||
except Exception as e:
|
||
yield f'data: {json.dumps({"error": str(e)})}\n\n'
|
||
finally:
|
||
try:
|
||
if upstream is not None:
|
||
upstream.close()
|
||
except Exception:
|
||
pass
|
||
_attribution_llm_lock.release()
|
||
|
||
return StreamingResponse(gen(), media_type="text/event-stream",
|
||
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|
||
|
||
|
||
@router.post("/api/character-deep-analysis")
|
||
async def character_deep_analysis(request: Request):
|
||
"""Run a deep 5-area psychological analysis of a single character.
|
||
|
||
Body: {name, role, goal, summary, text_excerpt, language, llm_url, model}
|
||
Returns: {analysis: {core_flaw, agency, dialogue_voice, narrative_arc, paradox}}
|
||
"""
|
||
data = await request.json()
|
||
name: str = (data.get("name") or "").strip()
|
||
role: str = (data.get("role") or "unknown").strip()
|
||
goal: str = (data.get("goal") or "").strip()
|
||
summary: str = (data.get("summary") or "").strip()
|
||
excerpt: str = (data.get("text_excerpt") or "").strip()
|
||
language: str = (data.get("language") or "").strip()
|
||
_settings = _load_settings()
|
||
llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or _settings.get("llm_model") or "").strip()
|
||
|
||
if not name:
|
||
raise HTTPException(400, "Character name required")
|
||
|
||
lang_note = f" Write the analysis in {language}." if language else ""
|
||
|
||
system = (
|
||
f"You are an expert developmental editor, literary coach, and psychologist specializing in "
|
||
f"profound character studies.{lang_note} Provide a deep, multi-layered psychological and "
|
||
f"narrative analysis of the character in exactly this JSON structure:\n"
|
||
'{"core_flaw":"<analysis of primary flaw, its psychological origin, and conflict with their desire>",'
|
||
'"agency":"<are they active or reactive? where does their momentum stall? what harder choices could force growth?>",'
|
||
'"dialogue_voice":"<speech pattern analysis: how does their dialogue reflect background, anxieties, hidden motives? how do they sound distinct?>",'
|
||
'"narrative_arc":"<how does their worldview shift? what is the lie they believe vs the truth they must accept?>",'
|
||
'"paradox":"<3 specific contradictions in behaviour/personality that make them feel authentic and unpredictable>"}\n'
|
||
"Be ruthlessly comprehensive and honest. Use concrete examples from the text. "
|
||
"Map where character development might break down or feel cliché. "
|
||
"Respond with STRICT JSON only — no markdown, no explanation./no-think"
|
||
)
|
||
user = (
|
||
f'CHARACTER: "{name}"\n'
|
||
f"Role in story: {role}\n"
|
||
f"Core goal: {goal or 'not specified'}\n"
|
||
f"Background/key actions: {summary or 'not specified'}\n\n"
|
||
+ (f"Text excerpt for analysis:\n{excerpt[:4000]}\n\n" if excerpt else "")
|
||
+ "Provide the deep psychological analysis now."
|
||
)
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": user},
|
||
],
|
||
"temperature": 0.6,
|
||
"max_tokens": 2048,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
try:
|
||
resp = requests.post(
|
||
f"{llm_url}/chat/completions", json=payload,
|
||
headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
raw = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
except Exception as e:
|
||
raise HTTPException(502, f"LLM deep analysis failed: {e}")
|
||
|
||
content = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL).strip() or raw
|
||
analysis = {}
|
||
for cand in (content, _extract_json_block(content)):
|
||
if not cand:
|
||
continue
|
||
try:
|
||
parsed = json.loads(cand)
|
||
if isinstance(parsed, dict) and "core_flaw" in parsed:
|
||
analysis = parsed
|
||
break
|
||
except Exception:
|
||
continue
|
||
if not analysis:
|
||
analysis = {"core_flaw": raw, "agency": "", "dialogue_voice": "", "narrative_arc": "", "paradox": ""}
|
||
return {"name": name, "analysis": analysis}
|
||
|
||
|
||
@router.post("/api/character-generate-prompts")
|
||
async def character_generate_prompts(request: Request):
|
||
"""Turn an already-extracted character sheet into four ready-to-use external
|
||
prompts, in one LLM call over the character's full profile (rather than
|
||
generating them incrementally passage-by-passage, where the model only
|
||
sees a fraction of the character at a time).
|
||
|
||
Body: {name, book, sheet: {...character sheet fields...}, language, llm_url, model}
|
||
Returns: {voice_design_prompt, image_prompt, silly_tavern_prompt, concept_art_prompt}
|
||
"""
|
||
data = await request.json()
|
||
name: str = (data.get("name") or "").strip()
|
||
book: str = (data.get("book") or "").strip()
|
||
sheet: dict = data.get("sheet") or {}
|
||
language: str = (data.get("language") or "").strip()
|
||
_settings = _load_settings()
|
||
llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or _settings.get("llm_model") or "").strip()
|
||
if not name:
|
||
raise HTTPException(400, "Character name required")
|
||
|
||
def _f(key: str) -> str:
|
||
v = sheet.get(key)
|
||
if isinstance(v, list):
|
||
return ", ".join(str(x) for x in v if x)
|
||
return str(v or "").strip()
|
||
|
||
profile = "\n".join(f"{label}: {v}" for label, v in [
|
||
("Name", name), ("Aliases/titles", _f("aliases") or _f("title")),
|
||
("Archetype", _f("archetype")), ("Physical", _f("physical")), ("Clothing", _f("clothing")),
|
||
("Mannerisms", _f("mannerisms")), ("Voice/speech pattern", _f("voice_pattern")),
|
||
("Backstory", _f("backstory")), ("Motivation", _f("motivation")), ("Fears", _f("fears")),
|
||
("Relationships", _f("relationships")), ("Conflict style", _f("conflict_style")),
|
||
("Secret/flaw", _f("secret")), ("Inventory", _f("inventory")),
|
||
] if v)
|
||
|
||
lang_note = f" Write every prompt in {language} EXCEPT where told to use English." if language else ""
|
||
preamble = (
|
||
"You are a prompt engineer who turns a fiction character's profile into ready-to-paste prompts for "
|
||
"other tools. Use ONLY details present in the profile below; mark anything you must reasonably infer "
|
||
f"with a trailing '*'. Never invent plot spoilers not implied by the profile.{lang_note}\n\n"
|
||
)
|
||
user = f"BOOK: {book or 'unspecified'}\n\nCHARACTER PROFILE:\n{profile or name}\n\n"
|
||
|
||
# One independent call per field (not one four-field JSON object, nor even
|
||
# the two-field pairing this used briefly) — the UI now has a Generate
|
||
# button per prompt box, so a click on just one must not also burn tokens
|
||
# regenerating the other three. Singling them out also further shrinks
|
||
# each response, since a four-field JSON object reliably truncated the
|
||
# LAST fields regardless of the token ceiling.
|
||
all_groups = {
|
||
"voice_design_prompt": (
|
||
preamble
|
||
+ "Produce exactly this field:\n"
|
||
"- voice_design_prompt: an English prompt for Qwen3 TTS Voice Design (15-45 words, one paragraph, "
|
||
"no markdown). Cover: apparent age, gender/androgyny if inferable, pitch, timbre/texture, pace, "
|
||
"accent or register, emotional baseline, and suitability for audiobook dialogue delivery. Do not "
|
||
"mention plot events — describe only how the voice should SOUND.\n\n"
|
||
'Respond with STRICT JSON only: {"voice_design_prompt":""}/no-think'
|
||
),
|
||
"image_prompt": (
|
||
preamble
|
||
+ "Produce exactly this field:\n"
|
||
"- image_prompt: a detailed English image-generation prompt for a character profile picture/portrait. "
|
||
"Include face and expression typical of this character, age impression, build, hair/eyes/skin if known, "
|
||
"clothing, signature tools/weapons/props, an environment typical for them, mood, and an art style "
|
||
"(e.g. 'detailed digital painting, dramatic lighting'). One dense paragraph, comma-separated descriptors "
|
||
"are fine.\n\n"
|
||
'Respond with STRICT JSON only: {"image_prompt":""}/no-think'
|
||
),
|
||
"silly_tavern_prompt": (
|
||
preamble
|
||
+ "Produce exactly this field:\n"
|
||
"- silly_tavern_prompt: character-card content for SillyTavern, formatted as labelled sections on their "
|
||
"own lines: 'Description:' (physical + personality summary), 'Personality:' (a compact trait list), "
|
||
"'Scenario:' (the situation/setting they're typically found in), 'First message:' (one in-character "
|
||
"greeting line in their own voice/speech pattern), and 'Example dialogue:' (2-3 short in-character "
|
||
"lines showing their manner of speech). Keep each section a few lines at most.\n\n"
|
||
'Respond with STRICT JSON only: {"silly_tavern_prompt":""}/no-think'
|
||
),
|
||
"concept_art_prompt": (
|
||
preamble
|
||
+ "Produce exactly this field:\n"
|
||
"- concept_art_prompt: an English prompt for a character CONCEPT SHEET (not a single portrait) — "
|
||
"a turnaround/reference sheet with multiple views and expressions: front view, side or back view, "
|
||
"2-3 facial expressions, and a close-up of a signature prop/costume detail, all on one clean sheet, "
|
||
"in a character-design-sheet art style (e.g. 'character turnaround, model sheet, flat lighting, "
|
||
"white background').\n\n"
|
||
'Respond with STRICT JSON only: {"concept_art_prompt":""}/no-think'
|
||
),
|
||
}
|
||
requested = [f for f in (data.get("fields") or []) if f in all_groups]
|
||
field_groups = [((f,), all_groups[f]) for f in (requested or all_groups.keys())]
|
||
|
||
def _call_group(fields: tuple, system: str) -> dict:
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": user + f"Generate the {' and '.join(fields)} now."},
|
||
],
|
||
"temperature": 0.7,
|
||
"max_tokens": 1536,
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
try:
|
||
resp = requests.post(
|
||
f"{llm_url}/chat/completions", json=payload,
|
||
headers={"Authorization": f"Bearer {_settings.get('llm_api_key') or 'sk-dummy-key'}"}, timeout=600,
|
||
)
|
||
resp.raise_for_status()
|
||
_msg = resp.json()["choices"][0]["message"]
|
||
raw = (_msg.get("content") or _reasoning_text(_msg) or "").strip()
|
||
except Exception as e:
|
||
print(f"[character-generate-prompts] LLM call failed for {fields}: {e}")
|
||
return {}
|
||
|
||
content = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL).strip() or raw
|
||
block = _extract_json_block(content)
|
||
# If the model still got cut off mid-string, closing the dangling string +
|
||
# object often salvages the field that DID complete.
|
||
candidates = [content, block]
|
||
if block:
|
||
candidates.extend([block + "\"}", block + "}"])
|
||
for cand in candidates:
|
||
if not cand:
|
||
continue
|
||
try:
|
||
parsed = json.loads(cand)
|
||
if isinstance(parsed, dict) and fields[0] in parsed:
|
||
return parsed
|
||
except Exception:
|
||
continue
|
||
return {}
|
||
|
||
results = await asyncio.gather(*[
|
||
asyncio.to_thread(_call_group, fields, system) for fields, system in field_groups
|
||
])
|
||
out = {}
|
||
for (fields, _), result in zip(field_groups, results):
|
||
for f in fields:
|
||
out[f] = str(result.get(f) or "").strip()
|
||
if not any(out.values()):
|
||
# A silent all-empty response looked identical to success in the UI —
|
||
# fail loudly so the client can show a real error instead.
|
||
raise HTTPException(502, "LLM returned no parseable prompts — try again (or a different model)")
|
||
return out
|
||
|
||
|
||
def _attribution_prepare(data: dict) -> dict:
|
||
"""Resolve settings and build the chat payload for one attribution request.
|
||
Shared by the blocking endpoint and the streaming (watch-the-LLM-think)
|
||
endpoint so the prompt logic can never drift between the two."""
|
||
text: str = (data.get("text") or "").strip()
|
||
known: list = data.get("known_characters") or []
|
||
recent: str = (data.get("recent") or "").strip() # last few attributed lines, for continuity
|
||
language: str = (data.get("language") or "").strip()
|
||
_settings = _load_settings()
|
||
llm_url: str = (data.get("llm_url") or _settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
model: str = (data.get("model") or _settings.get("llm_model") or "").strip()
|
||
timeout_seconds = _request_timeout_seconds(data.get("timeout_seconds"), 600.0)
|
||
if not text:
|
||
raise HTTPException(400, "No text provided")
|
||
|
||
lang_hint = f" The text language is {language}; keep names and wording in that language." if language else ""
|
||
|
||
prompt_override = data.get("audiobook_prompt")
|
||
if not isinstance(prompt_override, str):
|
||
prompt_override = data.get("prompt")
|
||
base_prompt = (prompt_override if isinstance(prompt_override, str) else "") or _settings.get("audiobook_prompt") or ""
|
||
# A user-saved custom prompt can predate (or have replaced) the deduction
|
||
# rules that cut down false Unknown/Narrator attributions — the client-side
|
||
# prompt migration only upgrades prompts still containing the original
|
||
# anchor text. Guarantee the rules ride along regardless of what prompt is
|
||
# in play, so attribution quality never silently depends on prompt history.
|
||
if base_prompt.strip() and "Doppelpunkt-Regel" not in base_prompt:
|
||
base_prompt += (
|
||
"\n\nANALYSE-REGELN FÜR DIE ZUORDNUNG DES SPRECHERS (zusätzlich, immer anwenden):\n"
|
||
"1. Pronomen (er/sie) IMMER zum zuletzt genannten Charakter passenden Geschlechts auflösen.\n"
|
||
"2. Doppelpunkt-Regel: Endet ein Erzählersatz mit \":\", spricht dessen Subjekt das folgende Zitat "
|
||
"(z.B. \"Dann richtete er sich auf und rief in die Runde:\" → der zuvor genannte Charakter spricht).\n"
|
||
"3. Nachgestellte Zuordnung: Der Erzählersatz NACH einem Zitat verrät oft den Sprecher — auch bei "
|
||
"unpersönlicher Formel (\"»Was machst du da?« ertönte es über ihm. Karyla war herübergekommen.\" → Karyla sprach).\n"
|
||
"4. Adressaten-Regel: \"X wandte sich an Y\" / \"X sah Y an\" → X spricht das nächste Zitat, Y antwortet.\n"
|
||
"5. Ping-Pong: Zwei Personen im Gespräch wechseln sich strikt ab, auch über viele Zitate ohne Tags. "
|
||
"In einer Zwei-Personen-Szene ist 'Unknown' fast immer falsch.\n"
|
||
"6. Rollenbezeichnungen sind gültige Sprecher ('Ork', 'Der Fremde', 'Nachbar', 'Wächter') — nutze sie statt 'Unknown'.\n"
|
||
"7. 'Unknown' NUR, wenn eine Zuordnung trotz aller Regeln absolut unmöglich ist.\n"
|
||
"8. Gedankenstrich-Pause-Regel: Ein \" - \" MITTEN in einem Zitat ist eine Sprechpause DESSELBEN Sprechers, "
|
||
"kein Zitatende — die Rede desselben Charakters geht danach unverändert weiter, bis das tatsächliche "
|
||
"schließende Anführungszeichen erscheint.\n"
|
||
"9. Stimm-Ankündigung: Erwähnt ein Erzählersatz kurz vor einer noch nicht zugeordneten Zeile explizit die "
|
||
"Stimme oder das (beginnende) Sprechen einer bestimmten Person (z.B. \"Marcians Stimme wirkte nicht mehr so "
|
||
"fest\", \"X setzte zum Sprechen an\"), gehört diese Zeile dieser Person — nicht 'Unknown'."
|
||
)
|
||
if not base_prompt.strip():
|
||
base_prompt = (
|
||
"You attribute dialogue in prose fiction for a multi-voice audiobook. "
|
||
"Split the passage into consecutive segments in reading order. For each segment output:\n"
|
||
"- speaker: 'Narrator' for narration/description, or the character's name for spoken dialogue. "
|
||
"Use the exact English word 'Unknown' ONLY as an absolute last resort.\n"
|
||
"- type: 'narration' or 'dialogue'\n"
|
||
"- text: the verbatim spoken words for dialogue (WITHOUT the surrounding quotation marks), or the verbatim prose for narration\n"
|
||
"- emotion: for dialogue, one or two words in the SAME LANGUAGE as the text (e.g. for German: wütend, traurig, flüsternd); '' for narration\n"
|
||
f"{lang_hint}\n"
|
||
"QUOTATION STYLES — books mark speech in many ways; treat ALL of these as spoken dialogue:\n"
|
||
" English straight \"...\" and curly “...”; German »...« (guillemets pointing inward) and „...“; "
|
||
"French «...» (pointing outward); single ‘...’; CJK 「...」 『...』; and em-dash speech where a line "
|
||
"starts with — or – (Spanish/French/Polish style).\n"
|
||
"PARAGRAPH/CHAPTER STRUCTURE: a blank line marks a real paragraph break or a chapter/scene start. "
|
||
"A very short standalone line right before a blank line (e.g. 'Chapter 1', 'Prologue', a bare number) "
|
||
"is a chapter heading — always narration/'Narrator', never dialogue. Keep blank lines as part of the "
|
||
"surrounding narration segment or their own narration segment; never invent dialogue from them and "
|
||
"never drop them from the reconstructed text.\n"
|
||
"German guillemets are the MOST IMPORTANT to detect: »Was schaust du dir an?« is a spoken line.\n"
|
||
"ATTRIBUTING THE SPEAKER (this is the hard, important part — be decisive):\n"
|
||
"1. If there is a dialogue tag ('sagte Riskan', 'fragte sie', 'Peter said'), use it. Resolve pronouns "
|
||
"(er/sie/he/she) to the actual name from nearby context.\n"
|
||
"2. UNTAGGED lines: use **conversational turn-taking**. In a two-person exchange the speaker ALTERNATES "
|
||
"every line — if Riskan just spoke, the next untagged quote is the other person, then back to Riskan, and so on.\n"
|
||
"3. Use the scene context, action beats around a quote (the person doing the action usually speaks), the "
|
||
"'Recent dialogue' below (continue the same conversation/alternation across the passage boundary), and the "
|
||
"known-characters list. Reuse the EXACT known names.\n"
|
||
"4. Only output 'Unknown' if the speaker is genuinely indeterminable even after applying turn-taking and "
|
||
"context — this should be rare. Prefer the most likely named character over 'Unknown'.\n"
|
||
"RULES:\n"
|
||
"- Put dialogue tags and action beats in a NARRATION segment, never inside the dialogue text.\n"
|
||
"- GRAMMAR CHECK FOR NARRATION: if a fragment is a reporting clause about speech "
|
||
"(finite speech verb + speaker/pronoun/name, e.g. 'murmelte er mit erstickter Stimme', "
|
||
"', entgegnete Marcian kalt', 'asked Peter quietly'), it is narration, never dialogue.\n"
|
||
"- Action beats are narration: a character looks, walks, laughs, stays silent, raises a hand, "
|
||
"turns away, etc. Only the actual quoted words are dialogue.\n"
|
||
"- If a quote is interrupted by a tag (»Die Pause«, sagte Peter, »ist vorbei.«), stitch the spoken parts "
|
||
"into ONE dialogue segment ('Die Pause ist vorbei.') with the tag as a separate narration segment.\n"
|
||
"- Strip the quotation marks/guillemets from dialogue text. Keep every word otherwise, in order.\n"
|
||
"- DO NOT hallucinate, summarize, or alter the text. The combined text of your segments MUST exactly match the original passage, word for word, except for dropped quotation marks.\n"
|
||
"- MID-QUOTE DASH RULE: a \" - \" (em-dash/hyphen used as a pause) in the MIDDLE of a quotation does NOT "
|
||
"end it — the SAME speaker's line continues unchanged after the dash, until the actual closing quotation "
|
||
"mark appears. Do not split it into narration or a new speaker at the dash.\n"
|
||
"- VOICE-ANNOUNCEMENT RULE: if narration explicitly mentions a specific character's voice or that they "
|
||
"are about to speak, shortly before an unattributed line (e.g. \"Marcian's voice sounded...\", "
|
||
"\"X began to say\"), attribute that line to that character instead of 'Unknown'."
|
||
)
|
||
else:
|
||
if lang_hint:
|
||
base_prompt += f"\n\n{lang_hint}"
|
||
|
||
# Only the streaming ("watch it think") call asks for a reasoning preamble —
|
||
# it costs extra tokens/latency, which the blocking/fallback path can't
|
||
# afford when it's already the retry-in-halves path for a slow model.
|
||
if data.get("want_reasoning"):
|
||
system = (
|
||
f"{base_prompt}\n\n"
|
||
"First, in a <think>...</think> block, briefly reason (2-4 short sentences) about who speaks each "
|
||
"untagged or ambiguous quote in this passage — mention the names/pronouns you're resolving. Keep it short.\n"
|
||
"Then, after </think>, respond with a single, valid JSON object containing a 'segments' array. "
|
||
"Nothing else outside the <think> block and the JSON object:\n"
|
||
'{"segments":[{"speaker":"Narrator","type":"narration","text":"...","emotion":""}]}'
|
||
)
|
||
else:
|
||
system = (
|
||
f"{base_prompt}\n\n"
|
||
"You MUST respond with a single, valid JSON object containing a 'segments' array.\n"
|
||
"Do NOT output any reasoning, chain of thought, or conversational text. Output ONLY the raw JSON object:\n"
|
||
'{"segments":[{"speaker":"Narrator","type":"narration","text":"...","emotion":""}]}'
|
||
)
|
||
user = (
|
||
("Known characters so far: " + ", ".join(str(n) for n in known) + "\n\n" if known else "")
|
||
+ ("Recent dialogue (the immediately preceding lines — continue the same conversation/turn-taking):\n" + recent + "\n\n" if recent else "")
|
||
+ "Passage to attribute:\n" + text + "\n\n"
|
||
+ "Return ONLY valid JSON. Do not include markdown blocks or any other text."
|
||
)
|
||
payload: dict = {
|
||
"messages": [
|
||
{"role": "system", "content": system},
|
||
{"role": "user", "content": user},
|
||
],
|
||
"temperature": 0.2,
|
||
# Dialogue-dense passages need much more output than input: every quoted
|
||
# line becomes a full JSON object ({"speaker":...,"type":...,"text":...,
|
||
# "emotion":...}), so budgeting off raw character count alone (as a
|
||
# tighter cap here once did) truncates mid-JSON on exchange-heavy scenes
|
||
# — which the parser can't always repair, degrading the whole chunk to
|
||
# naive quote-splitting with every speaker labelled "Unknown".
|
||
"max_tokens": min(8192, max(2048, len(text) + 1000)) + (250 if data.get("want_reasoning") else 0),
|
||
}
|
||
if model:
|
||
payload["model"] = model
|
||
return {
|
||
"payload": payload, "llm_url": llm_url, "model": model,
|
||
"api_key": _settings.get("llm_api_key") or "sk-dummy-key",
|
||
"timeout_seconds": timeout_seconds, "text": text,
|
||
"fallback_model": (_settings.get("llm_model") or "").strip(),
|
||
}
|
||
|
||
|
||
@router.post("/api/attribute-dialogue")
|
||
async def attribute_dialogue(request: Request):
|
||
"""Split a prose passage into attributed segments for a multi-voice audiobook.
|
||
|
||
Body: {text, known_characters:[...], language, llm_url, model}
|
||
Returns: {segments:[{speaker, type:"narration"|"dialogue", text, emotion}], characters:[names]}
|
||
The frontend calls this per chunk, passing the running character roster so the
|
||
same speaker keeps the same name across the whole book.
|
||
"""
|
||
data = await request.json()
|
||
prep = _attribution_prepare(data)
|
||
payload, llm_url, model = prep["payload"], prep["llm_url"], prep["model"]
|
||
timeout_seconds, text = prep["timeout_seconds"], prep["text"]
|
||
api_key = prep["api_key"]
|
||
try:
|
||
if not await asyncio.to_thread(_attribution_llm_lock.acquire, True, timeout_seconds):
|
||
return _fallback_attribute_response(text)
|
||
try:
|
||
resp = await asyncio.to_thread(
|
||
_post_llm_chat_completion,
|
||
llm_url, payload,
|
||
{"Authorization": f"Bearer {api_key}"}, timeout_seconds,
|
||
)
|
||
fallback_model = prep["fallback_model"]
|
||
if (
|
||
resp.status_code >= 400
|
||
and _is_router_model_alias(model)
|
||
and fallback_model
|
||
and fallback_model != model
|
||
and not _is_router_model_alias(fallback_model)
|
||
):
|
||
retry_payload = dict(payload)
|
||
retry_payload["model"] = fallback_model
|
||
resp = await asyncio.to_thread(
|
||
_post_llm_chat_completion,
|
||
llm_url, retry_payload,
|
||
{"Authorization": f"Bearer {api_key}"}, timeout_seconds,
|
||
)
|
||
payload = retry_payload
|
||
finally:
|
||
_attribution_llm_lock.release()
|
||
if resp.status_code >= 400:
|
||
print(
|
||
f"[attribute-dialogue] LLM failed ({resp.status_code}) for model "
|
||
f"{payload.get('model') or '(default)'}: {_response_error_text(resp)}"
|
||
)
|
||
return _fallback_attribute_response(text)
|
||
msg = resp.json()["choices"][0]["message"]
|
||
raw = (msg.get("content") or _reasoning_text(msg) or "").strip()
|
||
except HTTPException:
|
||
raise
|
||
except Exception as e:
|
||
print(f"[attribute-dialogue] LLM failed, using deterministic fallback: {e}")
|
||
return _fallback_attribute_response(text)
|
||
|
||
return _attribution_parse(raw, text)
|
||
|
||
|
||
def _attribution_parse(raw: str, text: str) -> dict:
|
||
"""Turn the LLM's raw answer into normalised {segments, characters}."""
|
||
content = re.sub(r"<think>.*?</think>", "", raw, flags=re.DOTALL).strip() or raw
|
||
segments = []
|
||
|
||
block = _extract_json_block(content)
|
||
candidates = [content, block]
|
||
if block:
|
||
# If the JSON was truncated by token limits, appending ]} to the last complete object often salvages it.
|
||
candidates.append(block + "]}")
|
||
|
||
for cand in candidates:
|
||
if not cand:
|
||
continue
|
||
try:
|
||
parsed = json.loads(cand)
|
||
if isinstance(parsed, dict) and isinstance(parsed.get("segments"), list):
|
||
segments = parsed["segments"]
|
||
break
|
||
except Exception:
|
||
continue
|
||
|
||
if not segments:
|
||
# If the LLM completely failed to output JSON despite the strict system prompt,
|
||
# it means it dropped into a conversational/reasoning hallucination.
|
||
# Do NOT try to parse its reasoning as a script. Abort and preserve the book text.
|
||
segments = [{"speaker": "Narrator", "type": "narration", "text": text, "emotion": ""}]
|
||
|
||
# Normalise + collect speaker roster
|
||
clean, chars = [], []
|
||
for seg in segments:
|
||
if not isinstance(seg, dict):
|
||
continue
|
||
t = str(seg.get("text") or "").strip()
|
||
if not t:
|
||
continue
|
||
sp = str(seg.get("speaker") or "Narrator").strip() or "Narrator"
|
||
typ = "dialogue" if str(seg.get("type") or "").lower().startswith("dial") else "narration"
|
||
if typ == "narration":
|
||
sp = "Narrator"
|
||
emo = str(seg.get("emotion") or "").strip()
|
||
clean.append({"speaker": sp, "type": typ, "text": t, "emotion": emo})
|
||
if typ == "dialogue" and sp.lower() != "narrator" and sp not in chars:
|
||
chars.append(sp)
|
||
return {"segments": clean, "characters": chars}
|
||
|
||
|
||
@router.post("/api/attribute-dialogue/stream")
|
||
async def attribute_dialogue_stream(request: Request):
|
||
"""Same job as /api/attribute-dialogue, but streams the LLM's live output
|
||
(reasoning + the JSON being written) as SSE `{"t": "..."}` events so the UI
|
||
can show what the model is thinking, ending with `{"done": true, "result"}`.
|
||
On upstream failure it emits `{"error": "..."}` — the client then falls
|
||
back to the blocking endpoint."""
|
||
data = await request.json()
|
||
prep = _attribution_prepare(data)
|
||
payload = dict(prep["payload"])
|
||
payload["stream"] = True
|
||
headers = {"Authorization": f"Bearer {prep['api_key']}"}
|
||
|
||
def gen():
|
||
if not _attribution_llm_lock.acquire(timeout=prep["timeout_seconds"]):
|
||
yield 'data: {"error": "Attribution engine busy"}\n\n'
|
||
return
|
||
upstream = None
|
||
try:
|
||
urls = _llm_chat_completion_urls(prep["llm_url"])
|
||
last_err = None
|
||
for u in urls:
|
||
try:
|
||
upstream = requests.post(
|
||
u, json=payload, headers=headers,
|
||
timeout=(15, prep["timeout_seconds"]), stream=True,
|
||
)
|
||
if upstream.status_code in (404, 405) and u != urls[-1]:
|
||
upstream.close()
|
||
upstream = None
|
||
continue
|
||
break
|
||
except Exception as exc:
|
||
last_err = exc
|
||
upstream = None
|
||
if upstream is None:
|
||
print(f"[attribute-dialogue/stream] LLM connect failed: {last_err}")
|
||
yield f'data: {json.dumps({"error": str(last_err or "LLM connect failed")})}\n\n'
|
||
return
|
||
if upstream.status_code >= 400:
|
||
print(
|
||
f"[attribute-dialogue/stream] LLM failed ({upstream.status_code}) for model "
|
||
f"{payload.get('model') or '(default)'}: {_response_error_text(upstream)[:300]}"
|
||
)
|
||
yield f'data: {json.dumps({"error": f"HTTP {upstream.status_code}: {_response_error_text(upstream)[:300]}"})}\n\n'
|
||
return
|
||
raw_parts = []
|
||
# The LLM's SSE response usually omits a charset on its Content-Type,
|
||
# so requests falls back to Latin-1 (the old HTTP default) instead of
|
||
# UTF-8 when decode_unicode=True guesses the encoding — every umlaut
|
||
# then comes out mojibake'd ("Häfen" -> "Häfen"). Force UTF-8.
|
||
upstream.encoding = "utf-8"
|
||
# Two layers against a stuck stream: the wall-clock check catches
|
||
# a model that keeps trickling chunks past its budget, and
|
||
# _watchdog_close catches the connection going fully silent (which
|
||
# the in-loop check can't see, since a blocked socket read never
|
||
# reaches it — see _watchdog_close's docstring for why). Observed
|
||
# in production without the watchdog: a stuck stream held the
|
||
# shared lock for 11+ minutes, silently starving every other
|
||
# passage/request.
|
||
deadline = time.monotonic() + prep["timeout_seconds"]
|
||
with _watchdog_close(upstream, prep["timeout_seconds"]):
|
||
for line in upstream.iter_lines(decode_unicode=True):
|
||
if time.monotonic() > deadline:
|
||
print(f"[attribute-dialogue/stream] wall-clock deadline hit after {prep['timeout_seconds']}s, aborting stream")
|
||
yield 'data: {"error": "LLM stream exceeded timeout"}\n\n'
|
||
return
|
||
if not line or not line.startswith("data:"):
|
||
continue
|
||
chunk = line[5:].strip()
|
||
if chunk == "[DONE]":
|
||
break
|
||
try:
|
||
delta = json.loads(chunk)["choices"][0]["delta"]
|
||
except Exception:
|
||
continue
|
||
# Reasoning models emit thoughts via reasoning_content (or, on
|
||
# some vLLM builds, a plain "reasoning" key — see _reasoning_text);
|
||
# the answer JSON arrives via content. Forward both to the
|
||
# viewer, but only content counts toward the parseable answer.
|
||
t = _reasoning_text(delta) or delta.get("content") or ""
|
||
if delta.get("content"):
|
||
raw_parts.append(delta["content"])
|
||
if t:
|
||
yield f'data: {json.dumps({"t": t})}\n\n'
|
||
result = _attribution_parse("".join(raw_parts), prep["text"])
|
||
yield f'data: {json.dumps({"done": True, "result": result})}\n\n'
|
||
except GeneratorExit:
|
||
raise
|
||
except Exception as e:
|
||
yield f'data: {json.dumps({"error": str(e)})}\n\n'
|
||
finally:
|
||
try:
|
||
if upstream is not None:
|
||
upstream.close()
|
||
except Exception:
|
||
pass
|
||
_attribution_llm_lock.release()
|
||
|
||
return StreamingResponse(gen(), media_type="text/event-stream",
|
||
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|
||
|
||
|
||
# ── Audio effects ─────────────────────────────────────────────────────────────
|
||
|
||
def _apply_audio_effects(audio_bytes: bytes, effects: list) -> bytes:
|
||
try:
|
||
from pedalboard import Pedalboard, Reverb, Chorus, Delay, Compressor, Gain, HighpassFilter, LowpassFilter, PitchShift # type: ignore
|
||
import numpy as np # type: ignore
|
||
except ImportError:
|
||
raise RuntimeError("pedalboard is not installed — run: pip install pedalboard numpy")
|
||
|
||
with io.BytesIO(audio_bytes) as buf:
|
||
with wave.open(buf, "rb") as wf:
|
||
n_channels = wf.getnchannels()
|
||
sample_rate = wf.getframerate()
|
||
n_frames = wf.getnframes()
|
||
raw = wf.readframes(n_frames)
|
||
sampwidth = wf.getsampwidth()
|
||
|
||
import numpy as np # noqa: F811
|
||
dtype = {1: np.int8, 2: np.int16, 4: np.int32}.get(sampwidth, np.int16)
|
||
samples = np.frombuffer(raw, dtype=dtype).astype(np.float32) / float(np.iinfo(dtype).max)
|
||
samples = samples.reshape(1, -1) if n_channels == 1 else samples.reshape(-1, n_channels).T
|
||
|
||
board = []
|
||
for fx in effects:
|
||
t = fx.get("type", "")
|
||
p = fx.get("params", {})
|
||
if t == "reverb":
|
||
board.append(Reverb(
|
||
room_size=float(p.get("room_size", 0.35)),
|
||
damping=float(p.get("damping", 0.5)),
|
||
wet_level=float(p.get("wet", 0.25)),
|
||
dry_level=float(p.get("dry", 0.8)),
|
||
))
|
||
elif t == "chorus":
|
||
board.append(Chorus(
|
||
rate_hz=float(p.get("rate_hz", 1.0)),
|
||
depth=float(p.get("depth", 0.25)),
|
||
mix=float(p.get("mix", 0.5)),
|
||
))
|
||
elif t == "delay":
|
||
board.append(Delay(
|
||
delay_seconds=float(p.get("delay_s", 0.25)),
|
||
feedback=float(p.get("feedback", 0.3)),
|
||
mix=float(p.get("mix", 0.4)),
|
||
))
|
||
elif t == "compressor":
|
||
board.append(Compressor(
|
||
threshold_db=float(p.get("threshold_db", -20.0)),
|
||
ratio=float(p.get("ratio", 4.0)),
|
||
attack_ms=float(p.get("attack_ms", 10.0)),
|
||
release_ms=float(p.get("release_ms", 100.0)),
|
||
))
|
||
elif t == "gain":
|
||
board.append(Gain(gain_db=float(p.get("gain_db", 0.0))))
|
||
elif t == "highpass":
|
||
board.append(HighpassFilter(cutoff_frequency_hz=float(p.get("cutoff_hz", 80.0))))
|
||
elif t == "lowpass":
|
||
board.append(LowpassFilter(cutoff_frequency_hz=float(p.get("cutoff_hz", 8000.0))))
|
||
elif t == "pitch_shift":
|
||
board.append(PitchShift(semitones=float(p.get("semitones", 0.0))))
|
||
|
||
if board:
|
||
samples = Pedalboard(board)(samples, sample_rate)
|
||
|
||
out = np.clip(samples, -1.0, 1.0)
|
||
pcm = ((out[0] if out.shape[0] == 1 else out.T.reshape(-1)) * 32767).astype(np.int16).tobytes()
|
||
buf_out = io.BytesIO()
|
||
with wave.open(buf_out, "wb") as wf:
|
||
wf.setnchannels(n_channels)
|
||
wf.setsampwidth(2)
|
||
wf.setframerate(sample_rate)
|
||
wf.writeframes(pcm)
|
||
return buf_out.getvalue()
|
||
|
||
|
||
@router.post("/api/audio/effects")
|
||
async def audio_effects(request: Request):
|
||
"""Apply an effects chain to a WAV file."""
|
||
form = await request.form()
|
||
audio_file = form.get("audio")
|
||
effects_json = str(form.get("effects") or "[]")
|
||
|
||
if audio_file is None:
|
||
raise HTTPException(400, "No audio file provided")
|
||
audio_bytes = await audio_file.read()
|
||
try:
|
||
effects = json.loads(effects_json)
|
||
except Exception:
|
||
raise HTTPException(400, "Invalid effects JSON")
|
||
|
||
try:
|
||
result = await asyncio.to_thread(_apply_audio_effects, audio_bytes, effects)
|
||
return Response(content=result, media_type="audio/wav")
|
||
except RuntimeError as e:
|
||
raise HTTPException(501, str(e))
|
||
except Exception as e:
|
||
raise HTTPException(500, f"Effects processing failed: {e}")
|
||
|
||
|
||
# ── Voices export / import ────────────────────────────────────────────────────
|
||
|
||
_EXPORT_SKIP_KEYS = {
|
||
"groq_api_key", "whisper_api_key", "tts_api_key", "voice_design_api_key",
|
||
"elevenlabs_api_key", "llm_api_key", "engine_api_keys",
|
||
}
|
||
_IMPORT_ALLOWED_SUFFIXES = set(_AUDIO_EXTS + [".reference.txt", ".meta.json", ".jpg", ".jpeg", ".png", ".webp"])
|
||
|
||
|
||
@router.get("/api/voices/export")
|
||
async def voices_export():
|
||
"""Export all voices + non-sensitive settings as a ZIP archive."""
|
||
import zipfile
|
||
from datetime import datetime
|
||
settings = _load_settings()
|
||
scan_dir = Path(settings.get("voices_scan_dir", _VOICES_DIR_DEFAULT))
|
||
|
||
buf = io.BytesIO()
|
||
with zipfile.ZipFile(buf, "w", zipfile.ZIP_DEFLATED) as zf:
|
||
safe_settings = {k: v for k, v in settings.items() if k not in _EXPORT_SKIP_KEYS}
|
||
zf.writestr("settings.json", json.dumps(safe_settings, indent=2))
|
||
if scan_dir.exists():
|
||
for f in scan_dir.rglob("*"):
|
||
if f.is_file():
|
||
try:
|
||
zf.write(f, str(f.relative_to(scan_dir.parent)))
|
||
except Exception:
|
||
pass
|
||
buf.seek(0)
|
||
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
||
return Response(
|
||
content=buf.read(),
|
||
media_type="application/zip",
|
||
headers={"Content-Disposition": f'attachment; filename="voices_export_{ts}.zip"'},
|
||
)
|
||
|
||
|
||
@router.post("/api/voices/import")
|
||
async def voices_import(file: UploadFile = File(...)):
|
||
"""Import voices from a ZIP archive (skips settings.json and unsafe paths)."""
|
||
import zipfile
|
||
settings = _load_settings()
|
||
scan_dir = Path(settings.get("voices_scan_dir", _VOICES_DIR_DEFAULT))
|
||
scan_dir.mkdir(parents=True, exist_ok=True)
|
||
|
||
content = await file.read()
|
||
if len(content) > _MAX_UPLOAD_BYTES:
|
||
raise HTTPException(413, "ZIP file too large")
|
||
|
||
try:
|
||
imported = 0
|
||
with zipfile.ZipFile(io.BytesIO(content)) as zf:
|
||
for info in zf.infolist():
|
||
if info.is_dir():
|
||
continue
|
||
parts = Path(info.filename).parts
|
||
if any(p in ("..", "") for p in parts) or Path(info.filename).name == "settings.json":
|
||
continue
|
||
if Path(info.filename).suffix.lower() not in _IMPORT_ALLOWED_SUFFIXES:
|
||
continue
|
||
rel = parts[1:] if len(parts) > 1 else parts
|
||
dest = scan_dir / Path(*rel)
|
||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||
dest.write_bytes(zf.read(info))
|
||
imported += 1
|
||
except zipfile.BadZipFile:
|
||
raise HTTPException(400, "Not a valid ZIP file")
|
||
except Exception as e:
|
||
raise HTTPException(500, f"Import failed: {e}")
|
||
|
||
await asyncio.to_thread(rebuild_voice_index, settings)
|
||
return {"ok": True, "imported": imported}
|
||
|
||
|
||
# ── /speak REST endpoint ──────────────────────────────────────────────────────
|
||
|
||
@router.post("/speak")
|
||
async def speak(request: Request):
|
||
data = await request.json()
|
||
text: str = str(data.get("text") or "").strip()
|
||
if not text:
|
||
raise HTTPException(400, "text is required")
|
||
|
||
explicit_voice: str = str(data.get("voice") or data.get("profile_id") or data.get("profile") or "").strip()
|
||
backend: str = _clean_preview_backend(str(data.get("backend") or "voice_clone"))
|
||
apply_persona: bool = bool(data.get("apply_persona") or data.get("personality"))
|
||
client_id: str = request.headers.get("X-Voice-Creator-Client-Id", "").strip()
|
||
|
||
settings = _load_settings()
|
||
voice = _resolve_speak_voice(settings, client_id, explicit_voice)
|
||
if not voice:
|
||
raise HTTPException(400, "No voice specified and no default voice configured")
|
||
|
||
if apply_persona:
|
||
scan_dir = Path(settings.get("voices_scan_dir", _VOICES_DIR_DEFAULT))
|
||
wav = _find_voice_audio(voice, scan_dir)
|
||
if wav:
|
||
persona = _load_meta(wav).get("persona", "")
|
||
if persona:
|
||
llm_url = (settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
llm_model = settings.get("llm_model") or ""
|
||
try:
|
||
text = await asyncio.to_thread(_rewrite_with_persona_sync, text, persona, llm_url, llm_model)
|
||
except Exception as e:
|
||
raise HTTPException(502, f"Persona rewrite failed: {e}")
|
||
|
||
instruct = str(data.get("instruct") or "")
|
||
try:
|
||
audio, media_type = await asyncio.to_thread(
|
||
_preview_request_audio, text, voice, settings, instruct, backend
|
||
)
|
||
except Exception as e:
|
||
raise HTTPException(502, f"TTS error: {e}")
|
||
|
||
return Response(content=audio, media_type=media_type)
|
||
|
||
|
||
@router.get("/speak/bindings")
|
||
async def get_speak_bindings():
|
||
s = _load_settings()
|
||
return {"bindings": s.get("client_voice_bindings") or {}}
|
||
|
||
|
||
@router.put("/speak/bindings/{client_id}")
|
||
async def put_speak_binding(client_id: str, request: Request):
|
||
data = await request.json()
|
||
voice = str(data.get("voice") or "").strip()
|
||
if not voice:
|
||
raise HTTPException(400, "voice is required")
|
||
s = _load_settings()
|
||
bindings: dict = dict(s.get("client_voice_bindings") or {})
|
||
bindings[client_id] = voice
|
||
s["client_voice_bindings"] = bindings
|
||
_save_settings(s)
|
||
return {"ok": True, "client_id": client_id, "voice": voice}
|
||
|
||
|
||
@router.delete("/speak/bindings/{client_id}")
|
||
async def delete_speak_binding(client_id: str):
|
||
s = _load_settings()
|
||
bindings: dict = dict(s.get("client_voice_bindings") or {})
|
||
if client_id not in bindings:
|
||
raise HTTPException(404, f"No binding for client '{client_id}'")
|
||
del bindings[client_id]
|
||
s["client_voice_bindings"] = bindings
|
||
_save_settings(s)
|
||
return {"ok": True, "client_id": client_id}
|
||
|
||
|
||
# ── MCP JSON-RPC 2.0 server ───────────────────────────────────────────────────
|
||
|
||
_MCP_SERVER_INFO = {"name": "tts-voice-creator", "version": "1.0.0"}
|
||
|
||
_MCP_TOOLS = [
|
||
{
|
||
"name": "speak",
|
||
"description": "Generate speech audio from text using a cloned voice. Returns a data URI with the WAV audio.",
|
||
"inputSchema": {
|
||
"type": "object",
|
||
"properties": {
|
||
"text": {"type": "string", "description": "Text to synthesize"},
|
||
"voice": {"type": "string", "description": "Voice ID (optional — uses per-client binding or default if omitted)"},
|
||
"apply_persona": {"type": "boolean", "description": "Rewrite text through the voice's persona before synthesis"},
|
||
},
|
||
"required": ["text"],
|
||
},
|
||
},
|
||
{
|
||
"name": "transcribe",
|
||
"description": "Transcribe base64-encoded WAV audio to text.",
|
||
"inputSchema": {
|
||
"type": "object",
|
||
"properties": {
|
||
"audio_base64": {"type": "string", "description": "Base64-encoded audio bytes (WAV preferred)"},
|
||
},
|
||
"required": ["audio_base64"],
|
||
},
|
||
},
|
||
{
|
||
"name": "list_captures",
|
||
"description": "List the 20 most recently generated audio files.",
|
||
"inputSchema": {"type": "object", "properties": {}},
|
||
},
|
||
{
|
||
"name": "list_profiles",
|
||
"description": "List all available voice profiles with their language, persona, and enabled state.",
|
||
"inputSchema": {"type": "object", "properties": {}},
|
||
},
|
||
]
|
||
|
||
|
||
async def _mcp_tool_speak(args: dict, client_id: str) -> dict:
|
||
text = str(args.get("text") or "").strip()
|
||
if not text:
|
||
raise ValueError("text is required")
|
||
explicit_voice = str(args.get("voice") or "").strip()
|
||
apply_persona = bool(args.get("apply_persona", False))
|
||
settings = _load_settings()
|
||
voice = _resolve_speak_voice(settings, client_id, explicit_voice)
|
||
if not voice:
|
||
raise ValueError("No voice specified and no default voice configured")
|
||
if apply_persona:
|
||
scan_dir = Path(settings.get("voices_scan_dir", _VOICES_DIR_DEFAULT))
|
||
wav = _find_voice_audio(voice, scan_dir)
|
||
if wav:
|
||
persona = _load_meta(wav).get("persona", "")
|
||
if persona:
|
||
llm_url = (settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
llm_model = settings.get("llm_model") or ""
|
||
text = await asyncio.to_thread(_rewrite_with_persona_sync, text, persona, llm_url, llm_model)
|
||
audio, _media_type = await asyncio.to_thread(
|
||
_preview_request_audio, text, voice, settings, "", "voice_clone"
|
||
)
|
||
audio_b64 = base64.b64encode(audio).decode()
|
||
return {
|
||
"content": [
|
||
{"type": "text", "text": f"Generated {len(audio)} bytes of speech audio for voice '{voice}'."},
|
||
{"type": "resource", "resource": {"uri": f"data:audio/wav;base64,{audio_b64}", "mimeType": "audio/wav"}},
|
||
]
|
||
}
|
||
|
||
|
||
async def _mcp_tool_transcribe(args: dict) -> dict:
|
||
raw = args.get("audio_base64") or ""
|
||
try:
|
||
audio_bytes = base64.b64decode(raw)
|
||
except Exception:
|
||
raise ValueError("audio_base64 is not valid base64")
|
||
tmp = TEMP_DIR / f"{uuid.uuid4().hex}_mcp_transcribe.wav"
|
||
tmp.write_bytes(audio_bytes)
|
||
settings = _load_settings()
|
||
try:
|
||
text, used_backend = await asyncio.to_thread(_transcribe_audio, tmp, settings, "configured")
|
||
finally:
|
||
try:
|
||
tmp.unlink(missing_ok=True)
|
||
except Exception:
|
||
pass
|
||
return {"content": [{"type": "text", "text": text}], "backend": used_backend}
|
||
|
||
|
||
async def _mcp_tool_list_captures() -> dict:
|
||
settings = _load_settings()
|
||
out_dir = _active_voices_dir(settings)
|
||
if not out_dir.exists():
|
||
return {"content": [{"type": "text", "text": "[]"}]}
|
||
files = sorted(
|
||
(p for p in out_dir.rglob("*.wav") if not _is_internal_voice_file(p)),
|
||
key=lambda p: p.stat().st_mtime,
|
||
reverse=True,
|
||
)[:20]
|
||
result = [{"name": p.stem, "path": str(p), "mtime": p.stat().st_mtime} for p in files]
|
||
return {"content": [{"type": "text", "text": json.dumps(result)}]}
|
||
|
||
|
||
async def _mcp_tool_list_profiles() -> dict:
|
||
settings = _load_settings()
|
||
scan_dir = Path(settings.get("voices_scan_dir", _VOICES_DIR_DEFAULT))
|
||
profiles = []
|
||
seen: set[str] = set()
|
||
for audio in sorted(_voice_audio_files(scan_dir), key=lambda p: p.stem.lower()):
|
||
if audio.stem in seen:
|
||
continue
|
||
seen.add(audio.stem)
|
||
meta = _load_meta(audio)
|
||
parts = audio.stem.split("_", 2)
|
||
lang = parts[0].upper() if parts else ""
|
||
profiles.append({
|
||
"id": audio.stem,
|
||
"name": audio.stem,
|
||
"lang": lang,
|
||
"persona": meta.get("persona", ""),
|
||
"enabled": meta.get("enabled", True),
|
||
})
|
||
return {"content": [{"type": "text", "text": json.dumps(profiles)}]}
|
||
|
||
|
||
def _mcp_error_response(code: int, message: str, rpc_id) -> Response:
|
||
import logging
|
||
body = {"jsonrpc": "2.0", "error": {"code": code, "message": message}, "id": rpc_id}
|
||
return Response(content=json.dumps(body), media_type="application/json")
|
||
|
||
|
||
@router.post("/mcp")
|
||
async def mcp_jsonrpc(request: Request):
|
||
import logging
|
||
logger = logging.getLogger("uvicorn.error")
|
||
try:
|
||
body = await request.json()
|
||
except Exception:
|
||
return _mcp_error_response(-32700, "Parse error", None)
|
||
|
||
rpc_id = body.get("id")
|
||
method = body.get("method", "")
|
||
params = body.get("params") or {}
|
||
client_id = request.headers.get("X-Voice-Creator-Client-Id", "").strip()
|
||
|
||
try:
|
||
if method == "initialize":
|
||
result = {
|
||
"protocolVersion": "2024-11-05",
|
||
"serverInfo": _MCP_SERVER_INFO,
|
||
"capabilities": {"tools": {}},
|
||
}
|
||
elif method == "notifications/initialized":
|
||
return Response(status_code=204)
|
||
elif method == "tools/list":
|
||
result = {"tools": _MCP_TOOLS}
|
||
elif method == "tools/call":
|
||
tool_name = str(params.get("name") or "")
|
||
tool_args = params.get("arguments") or {}
|
||
if tool_name == "speak":
|
||
result = await _mcp_tool_speak(tool_args, client_id)
|
||
elif tool_name == "transcribe":
|
||
result = await _mcp_tool_transcribe(tool_args)
|
||
elif tool_name == "list_captures":
|
||
result = await _mcp_tool_list_captures()
|
||
elif tool_name == "list_profiles":
|
||
result = await _mcp_tool_list_profiles()
|
||
else:
|
||
return _mcp_error_response(-32601, f"Unknown tool: {tool_name}", rpc_id)
|
||
else:
|
||
return _mcp_error_response(-32601, f"Method not found: {method}", rpc_id)
|
||
except ValueError as e:
|
||
return _mcp_error_response(-32602, str(e), rpc_id)
|
||
except Exception as e:
|
||
logger.exception("MCP tool error in method %s", method)
|
||
return _mcp_error_response(-32000, f"Server error: {e}", rpc_id)
|
||
|
||
return Response(
|
||
content=json.dumps({"jsonrpc": "2.0", "result": result, "id": rpc_id}),
|
||
media_type="application/json",
|
||
)
|
||
|
||
|
||
@router.get("/mcp")
|
||
async def mcp_sse(request: Request):
|
||
"""SSE keep-alive stream (satisfies MCP spec GET /mcp requirement)."""
|
||
from typing import AsyncGenerator
|
||
|
||
async def _keepalive() -> AsyncGenerator[str, None]:
|
||
yield ": mcp-sse-ready\n\n"
|
||
while True:
|
||
await asyncio.sleep(15)
|
||
yield ": ping\n\n"
|
||
return StreamingResponse(
|
||
_keepalive(),
|
||
media_type="text/event-stream",
|
||
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"},
|
||
)
|
||
|
||
|
||
# ── Conversation playground ───────────────────────────────────────────────────
|
||
|
||
@router.get("/api/conversation/llm-models")
|
||
async def conversation_llm_models(url: str = "", api_key: str = ""):
|
||
"""List models from a local LLM endpoint (Ollama / vLLM / LM Studio)."""
|
||
from core.validation import _validate_http_url
|
||
settings = _load_settings()
|
||
base = (url or settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
key_to_use = api_key or settings.get("llm_api_key") or "sk-dummy-key"
|
||
try:
|
||
base = _validate_http_url(base, allow_private=True)
|
||
r = requests.get(f"{base}/models", timeout=5, headers={"Authorization": f"Bearer {key_to_use}"})
|
||
|
||
# Fallback for gateways like LiteLLM that might prefer /models over /v1/models
|
||
if r.status_code != 200 and base.endswith("/v1"):
|
||
fallback_base = base[:-3]
|
||
r2 = requests.get(f"{fallback_base}/models", timeout=5, headers={"Authorization": f"Bearer {key_to_use}"})
|
||
if r2.status_code == 200:
|
||
r = r2
|
||
|
||
if r.status_code == 200:
|
||
payload = r.json()
|
||
data = payload.get("data", []) if isinstance(payload, dict) else []
|
||
models = [
|
||
str(item["id"]) if isinstance(item, dict) and item.get("id") else str(item)
|
||
for item in data if item
|
||
]
|
||
return {"models": models, "url": base}
|
||
else:
|
||
return {"models": [], "url": base, "error": r.text}
|
||
except Exception as e:
|
||
return {"models": [], "url": base, "error": str(e)}
|
||
|
||
|
||
def _make_tts_task(
|
||
text: str,
|
||
voice: str,
|
||
settings: dict,
|
||
backend: str,
|
||
sem: "asyncio.Semaphore | None",
|
||
) -> "asyncio.Task":
|
||
if sem is None:
|
||
return asyncio.create_task(
|
||
asyncio.to_thread(_preview_request_audio, text, voice, settings, "", backend)
|
||
)
|
||
|
||
async def _guarded() -> tuple[bytes, str]:
|
||
async with sem:
|
||
return await asyncio.to_thread(_preview_request_audio, text, voice, settings, "", backend)
|
||
|
||
return asyncio.create_task(_guarded())
|
||
|
||
|
||
@router.post("/api/conversation/turn")
|
||
async def conversation_turn(
|
||
audio: Optional[UploadFile] = None,
|
||
text: str = Form(""),
|
||
stt_backend: str = Form("configured"),
|
||
llm_url: str = Form(""),
|
||
llm_model: str = Form(""),
|
||
tts_backend: str = Form("voice_clone"),
|
||
tts_voice: str = Form(""),
|
||
system_prompt: str = Form("You are a helpful voice assistant. Keep replies short and conversational."),
|
||
history: str = Form("[]"),
|
||
):
|
||
"""Stream a full conversation turn (STT → LLM → TTS) as Server-Sent Events.
|
||
Pass either an audio file (runs STT first) or a plain text string (skips STT).
|
||
"""
|
||
direct_text = text.strip()
|
||
if not direct_text and (audio is None or not getattr(audio, "filename", None)):
|
||
raise HTTPException(400, "Provide either an audio file or a text field")
|
||
|
||
settings = _load_settings()
|
||
eff_llm_url = (llm_url or settings.get("llm_url") or "http://localhost:11434/v1").rstrip("/")
|
||
|
||
# Prepare audio temp files only when an audio upload was provided
|
||
_tmp: Path | None = None
|
||
_wav: Path | None = None
|
||
if not direct_text and audio is not None:
|
||
suffix = Path(audio.filename or "audio.webm").suffix.lower() or ".webm"
|
||
if suffix not in _UPLOAD_EXTS:
|
||
suffix = ".webm"
|
||
_tmp = TEMP_DIR / f"{uuid.uuid4().hex}_conv{suffix}"
|
||
try:
|
||
with _tmp.open("wb") as f:
|
||
_copy_limited(audio.file, f, _MAX_UPLOAD_BYTES)
|
||
_wav = _to_wav_16k(_tmp)
|
||
except Exception as e:
|
||
_tmp.unlink(missing_ok=True)
|
||
raise HTTPException(400, f"Audio upload failed: {e}")
|
||
|
||
try:
|
||
hist = json.loads(history) if history else []
|
||
if not isinstance(hist, list):
|
||
hist = []
|
||
except Exception:
|
||
hist = []
|
||
|
||
stt_be = _clean_stt_backend(stt_backend)
|
||
tts_be = _clean_preview_backend(tts_backend)
|
||
|
||
async def generate():
|
||
t0 = time.monotonic()
|
||
stt_ms = llm_ttft_ms = llm_total_ms = tts_ms = None
|
||
transcript = llm_text = ""
|
||
|
||
def sse(obj: dict) -> str:
|
||
return f"data: {json.dumps(obj)}\n\n"
|
||
|
||
# 1. STT — skipped when caller sends direct text
|
||
if direct_text:
|
||
transcript = direct_text
|
||
yield sse({"type": "transcript", "text": transcript, "stt_ms": None})
|
||
else:
|
||
try:
|
||
t_stt = time.monotonic()
|
||
transcript, _ = await asyncio.to_thread(_transcribe_audio, _wav, settings, stt_be)
|
||
stt_ms = int((time.monotonic() - t_stt) * 1000)
|
||
yield sse({"type": "transcript", "text": transcript, "stt_ms": stt_ms})
|
||
except Exception as e:
|
||
yield sse({"type": "error", "stage": "stt", "message": str(e)})
|
||
return
|
||
finally:
|
||
for p in filter(None, {_tmp, _wav}):
|
||
with contextlib.suppress(Exception):
|
||
p.unlink(missing_ok=True)
|
||
|
||
if not transcript.strip() or _is_hallucination(transcript):
|
||
yield sse({"type": "error", "stage": "stt", "message": "No speech detected."})
|
||
return
|
||
|
||
# 2. LLM stream — runs in a thread; tokens arrive via asyncio.Queue
|
||
# so the event loop is never blocked and TTS can start on sentence 1
|
||
# while the LLM is still generating sentences 2, 3, …
|
||
messages = [{"role": "system", "content": system_prompt}]
|
||
messages.extend(hist[-20:])
|
||
messages.append({"role": "user", "content": transcript})
|
||
llm_payload: dict = {"messages": messages, "stream": True, "temperature": 0.7, "max_tokens": 512}
|
||
eff_llm_model = llm_model or settings.get("llm_model") or ""
|
||
if eff_llm_model:
|
||
llm_payload["model"] = eff_llm_model
|
||
|
||
_loop = asyncio.get_event_loop()
|
||
_token_q: asyncio.Queue[tuple[str, str] | None] = asyncio.Queue()
|
||
|
||
def _llm_thread() -> None:
|
||
try:
|
||
resp = requests.post(
|
||
f"{eff_llm_url}/chat/completions", json=llm_payload,
|
||
headers={"Authorization": f"Bearer {settings.get('llm_api_key') or 'sk-dummy-key'}"}, stream=True, timeout=120,
|
||
)
|
||
resp.raise_for_status()
|
||
for raw_line in resp.iter_lines():
|
||
if not raw_line:
|
||
continue
|
||
line = raw_line.decode("utf-8") if isinstance(raw_line, bytes) else str(raw_line)
|
||
if not line.startswith("data:"):
|
||
continue
|
||
payload_str = line[5:].strip()
|
||
if payload_str == "[DONE]":
|
||
break
|
||
try:
|
||
obj = json.loads(payload_str)
|
||
d_obj = ((obj.get("choices") or [{}])[0].get("delta") or {})
|
||
reasoning = _reasoning_text(d_obj)
|
||
content = d_obj.get("content") or ""
|
||
if not content and not reasoning and isinstance(obj.get("message"), dict):
|
||
content = obj["message"].get("content") or ""
|
||
if reasoning:
|
||
_loop.call_soon_threadsafe(_token_q.put_nowait, ("reasoning", reasoning))
|
||
if content:
|
||
_loop.call_soon_threadsafe(_token_q.put_nowait, ("content", content))
|
||
except Exception:
|
||
continue
|
||
except Exception as exc:
|
||
_loop.call_soon_threadsafe(_token_q.put_nowait, ("error", str(exc)))
|
||
finally:
|
||
_loop.call_soon_threadsafe(_token_q.put_nowait, None)
|
||
|
||
threading.Thread(target=_llm_thread, daemon=True).start()
|
||
|
||
# Some models (e.g. Qwen3 in "thinking" mode) don't use a separate
|
||
# reasoning_content field — they emit a <think>...</think> block inline
|
||
# inside content, split across arbitrary delta chunks. Many chat
|
||
# templates also inject the opening "<think>" as a fixed prompt prefix,
|
||
# so it never appears in the generated stream at all — only the closing
|
||
# "</think>" does. This tracks both explicit-tag and implicit-open
|
||
# styles so the answer stays clean for chat history/TTS and the
|
||
# reasoning can be shown in a collapsible "thinking" panel instead.
|
||
_THINK_SNIFF_LIMIT = 300 # chars buffered before assuming "not a reasoning response"
|
||
_think_state = {"buf": "", "resolved": False, "in_think": False}
|
||
|
||
def _explicit_tag_split(text: str) -> tuple[str, str]:
|
||
# Assumes _think_state["resolved"] is True — scans for <think>/</think>
|
||
# tags that may be split across delta chunks.
|
||
_think_state["buf"] += text
|
||
thinking_parts: list[str] = []
|
||
answer_parts: list[str] = []
|
||
buf = _think_state["buf"]
|
||
while True:
|
||
tag = "</think>" if _think_state["in_think"] else "<think>"
|
||
idx = buf.find(tag)
|
||
if idx == -1:
|
||
break
|
||
before = buf[:idx]
|
||
(thinking_parts if _think_state["in_think"] else answer_parts).append(before)
|
||
buf = buf[idx + len(tag):]
|
||
_think_state["in_think"] = not _think_state["in_think"]
|
||
# Hold back a short suffix that could be the start of a split tag
|
||
hold = 0
|
||
for cut in range(1, min(8, len(buf)) + 1):
|
||
suffix = buf[-cut:]
|
||
if "<think>".startswith(suffix) or "</think>".startswith(suffix):
|
||
hold = cut
|
||
flush = buf[:len(buf) - hold] if hold else buf
|
||
_think_state["buf"] = buf[len(buf) - hold:] if hold else ""
|
||
(thinking_parts if _think_state["in_think"] else answer_parts).append(flush)
|
||
return "".join(thinking_parts), "".join(answer_parts)
|
||
|
||
def _split_think(text: str) -> tuple[str, str]:
|
||
if _think_state["resolved"]:
|
||
return _explicit_tag_split(text)
|
||
_think_state["buf"] += text
|
||
buf = _think_state["buf"]
|
||
open_idx = buf.find("<think>")
|
||
close_idx = buf.find("</think>")
|
||
if close_idx != -1 and (open_idx == -1 or close_idx < open_idx):
|
||
# No opening tag before this close — the model (or its chat
|
||
# template) started inside a think block implicitly.
|
||
thinking_text = buf[:close_idx]
|
||
rest = buf[close_idx + len("</think>"):]
|
||
_think_state["resolved"] = True
|
||
_think_state["in_think"] = False
|
||
_think_state["buf"] = ""
|
||
_, rest_answer = _explicit_tag_split(rest) if rest else ("", "")
|
||
return thinking_text, rest_answer
|
||
if open_idx != -1:
|
||
before = buf[:open_idx]
|
||
rest = buf[open_idx:]
|
||
_think_state["resolved"] = True
|
||
_think_state["in_think"] = False
|
||
_think_state["buf"] = ""
|
||
thinking_rest, answer_rest = _explicit_tag_split(rest) if rest else ("", "")
|
||
return thinking_rest, before + answer_rest
|
||
if len(buf) >= _THINK_SNIFF_LIMIT:
|
||
# No think tag within the sniff window — treat as a plain,
|
||
# non-reasoning response from here on.
|
||
_think_state["resolved"] = True
|
||
_think_state["in_think"] = False
|
||
_think_state["buf"] = ""
|
||
return "", buf
|
||
return "", "" # still buffering — undecided
|
||
|
||
def _finish_split_think() -> tuple[str, str]:
|
||
if not _think_state["buf"]:
|
||
return "", ""
|
||
leftover = _think_state["buf"]
|
||
_think_state["buf"] = ""
|
||
if not _think_state["resolved"]:
|
||
return "", leftover # stream ended before we saw any think tag
|
||
return (leftover, "") if _think_state["in_think"] else ("", leftover)
|
||
|
||
# Streaming backend (8023) has per-process voice state — serialize to prevent
|
||
# concurrent requests from clobbering each other's voice context.
|
||
tts_sem: asyncio.Semaphore | None = asyncio.Semaphore(1) if tts_be == "streaming" else None
|
||
|
||
t_llm = time.monotonic()
|
||
llm_ttft_ms: int | None = None
|
||
ttft_done = False
|
||
sent_buf = ""
|
||
tts_tasks: list[asyncio.Task] = []
|
||
tts_texts: list[str] = [] # sentence text corresponding to each task
|
||
tts_first_start: float | None = None
|
||
|
||
try:
|
||
while True:
|
||
item = await _token_q.get()
|
||
if item is None:
|
||
break
|
||
kind, text = item
|
||
if kind == "error":
|
||
yield sse({"type": "error", "stage": "llm", "message": text})
|
||
return
|
||
if not ttft_done:
|
||
llm_ttft_ms = int((time.monotonic() - t_llm) * 1000)
|
||
ttft_done = True
|
||
thinking_piece, answer_piece = (text, "") if kind == "reasoning" else _split_think(text)
|
||
if thinking_piece:
|
||
yield sse({"type": "thinking", "delta": thinking_piece})
|
||
if not answer_piece:
|
||
continue
|
||
llm_text += answer_piece
|
||
sent_buf += answer_piece
|
||
yield sse({"type": "token", "delta": answer_piece})
|
||
# Fire TTS on sentence boundary — runs concurrently with LLM
|
||
split = _sentence_split(sent_buf)
|
||
if split > 0:
|
||
chunk_text = sent_buf[:split].strip()
|
||
sent_buf = sent_buf[split:]
|
||
if chunk_text:
|
||
if tts_first_start is None:
|
||
tts_first_start = time.monotonic()
|
||
tts_texts.append(chunk_text)
|
||
tts_tasks.append(_make_tts_task(chunk_text, tts_voice, settings, tts_be, tts_sem))
|
||
except Exception as exc:
|
||
yield sse({"type": "error", "stage": "llm", "message": str(exc)})
|
||
return
|
||
|
||
# Flush any text still held back (partial tag, or sniff buffer never resolved)
|
||
_thinking_tail, _answer_tail = _finish_split_think()
|
||
if _thinking_tail:
|
||
yield sse({"type": "thinking", "delta": _thinking_tail})
|
||
if _answer_tail:
|
||
llm_text += _answer_tail
|
||
sent_buf += _answer_tail
|
||
yield sse({"type": "token", "delta": _answer_tail})
|
||
|
||
# Flush any remaining text as a final TTS task
|
||
if sent_buf.strip():
|
||
if tts_first_start is None:
|
||
tts_first_start = time.monotonic()
|
||
tts_texts.append(sent_buf.strip())
|
||
tts_tasks.append(_make_tts_task(sent_buf.strip(), tts_voice, settings, tts_be, tts_sem))
|
||
|
||
llm_total_ms = int((time.monotonic() - t_llm) * 1000)
|
||
yield sse({"type": "llm_done", "text": llm_text,
|
||
"llm_ttft_ms": llm_ttft_ms, "llm_total_ms": llm_total_ms})
|
||
|
||
if not llm_text.strip():
|
||
yield sse({"type": "error", "stage": "llm",
|
||
"message": "LLM returned empty response. "
|
||
"If using a Qwen3 model, try adding /no-think to your system prompt "
|
||
"or pick a non-thinking model in the Language Model dropdown."})
|
||
return
|
||
|
||
# 3. Stream audio chunks in order — each chunk's TTS ran concurrently
|
||
# with LLM generation, so first audio arrives much sooner than
|
||
# waiting for the full response.
|
||
tts_start = tts_first_start or time.monotonic()
|
||
for i, task in enumerate(tts_tasks):
|
||
try:
|
||
audio_bytes, mime = await task
|
||
except Exception as exc:
|
||
yield sse({"type": "error", "stage": "tts", "message": str(exc)})
|
||
return
|
||
if i == 0:
|
||
tts_ms = int((time.monotonic() - tts_start) * 1000)
|
||
sentence_text = tts_texts[i] if i < len(tts_texts) else ""
|
||
yield sse({"type": "audio", "b64": base64.b64encode(audio_bytes).decode(),
|
||
"mime": mime, "text": sentence_text})
|
||
|
||
total_ms = int((time.monotonic() - t0) * 1000)
|
||
yield sse({"type": "stats", "stt_ms": stt_ms, "llm_ttft_ms": llm_ttft_ms,
|
||
"llm_total_ms": llm_total_ms, "tts_ms": tts_ms, "total_ms": total_ms})
|
||
yield sse({"type": "done"})
|
||
|
||
return StreamingResponse(generate(), media_type="text/event-stream",
|
||
headers={"Cache-Control": "no-cache", "X-Accel-Buffering": "no"})
|