Security fixes: - Block /proc /sys /dev /run /boot in /api/browse-dirs (path traversal) - Verify yt-dlp output stays inside TEMP_DIR before registration - Remove Access-Control-Allow-Origin: * from /api/proxy-audio - TTL-based temp file registry (default 2h) to prevent disk fill Performance: - Cache settings + routing rules in memory (mtime-checked); eliminates per-request disk reads on every TTS call UI: - Add container name (optional) field to Docker stack TTS/STT engine cards (Qwen3 Voice Clone, Voice Design, Custom Voice, Streaming, NVIDIA Magpie, Parakeet) — enables Stop/Start/Restart buttons on all engine cards, matching the existing Other Local TTS/STT cards Refactor — backend: - server.py: 5560 lines → 43-line entry point - core/ package: constants, registry, validation, docker_client, config, routing, audio, voice, presets, tts_helpers - routes/ package: admin, settings, library, stt, sources, docker, tts, conversation (FastAPI APIRouter modules) - Dockerfile + docker-compose.yml updated to include core/ and routes/ Refactor — frontend: - static/app.js: 8744 lines → 16 modules in static/js/ utils, voice-inspector, voice-sources, integrations, routing, settings, voice-clone, voice-library, tts-preview, benchmark, stt, init, engines, ai-backends, generation, conversation - static/loader.js updated to load modules sequentially Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
409 lines
15 KiB
Python
409 lines
15 KiB
Python
"""Voice meta helpers, file utilities, and benchmark helpers."""
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import re
|
|
import shutil
|
|
import time
|
|
import uuid
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
|
|
import requests
|
|
from fastapi import HTTPException
|
|
|
|
from core.constants import (
|
|
_VOICES_DIR_DEFAULT, _OUTPUT_DIR_DEFAULT,
|
|
_VOICE_TARGET_DBFS, _VOICE_PEAK_DBFS,
|
|
_BENCHMARK_SENTENCES, _MAX_TTS_OUTPUT_SECONDS,
|
|
)
|
|
from core.audio import _duration, _export_normalized_wav
|
|
|
|
# ── File type constants ───────────────────────────────────────────────────────
|
|
|
|
_PICTURE_EXTS = [".jpg", ".jpeg", ".png", ".webp"]
|
|
_SOUND_ASSET_DIRS = {"sound", "sounds", "sfx", "effects", "sound_effects", "route_sounds"}
|
|
_SOUND_ASSET_PREFIXES = ("computerbeep", "beep", "ding", "chime", "notification")
|
|
_AUDIO_EXTS = [".wav", ".mp3", ".m4a", ".flac", ".ogg", ".opus"]
|
|
_UPLOAD_EXTS = _AUDIO_EXTS + [".mp4", ".mkv", ".webm", ".mov", ".avi"]
|
|
_AUDIO_MIME = {
|
|
".wav": "audio/wav",
|
|
".mp3": "audio/mpeg",
|
|
".m4a": "audio/mp4",
|
|
".flac": "audio/flac",
|
|
".ogg": "audio/ogg",
|
|
".opus": "audio/ogg",
|
|
}
|
|
|
|
# Default flag (ISO country code) for each language code
|
|
_LANG_FLAG_DEFAULT = {
|
|
"EN": "GB", "DE": "DE", "ZH": "CN", "FR": "FR", "ES": "ES",
|
|
"JA": "JP", "KO": "KR", "IT": "IT", "PT": "BR", "RU": "RU",
|
|
"AR": "SA", "PL": "PL", "NL": "NL", "SV": "SE", "TR": "TR", "HI": "IN",
|
|
}
|
|
|
|
|
|
# ── Voice audio discovery helpers ─────────────────────────────────────────────
|
|
|
|
def _is_internal_voice_file(path: Path) -> bool:
|
|
stem = path.stem.lower()
|
|
name = path.name.lower()
|
|
return (
|
|
stem.endswith(".original")
|
|
or ".normalized.tmp" in stem
|
|
or name.startswith(".")
|
|
or name.endswith(".bak")
|
|
)
|
|
|
|
|
|
def _is_sound_asset_file(path: Path) -> bool:
|
|
stem = path.stem.lower()
|
|
parent_names = {part.lower() for part in path.parts}
|
|
return (
|
|
any(part in _SOUND_ASSET_DIRS for part in parent_names)
|
|
or stem.startswith(_SOUND_ASSET_PREFIXES)
|
|
)
|
|
|
|
|
|
def _voice_audio_files(root: Path):
|
|
for ext in _AUDIO_EXTS:
|
|
for p in root.rglob(f"*{ext}"):
|
|
if not _is_internal_voice_file(p) and not _is_sound_asset_file(p):
|
|
yield p
|
|
|
|
|
|
def _find_voice_audio(voice_id: str, scan_dir: Path) -> Path | None:
|
|
for ext in _AUDIO_EXTS:
|
|
for p in sorted(scan_dir.rglob(f"{voice_id}{ext}")):
|
|
if not _is_internal_voice_file(p) and not _is_sound_asset_file(p):
|
|
return p
|
|
return None
|
|
|
|
|
|
# ── Voice meta helpers ────────────────────────────────────────────────────────
|
|
|
|
def _meta_path(wav: Path) -> Path:
|
|
return wav.with_suffix(".meta.json")
|
|
|
|
|
|
def _load_meta(wav: Path) -> dict:
|
|
mp = _meta_path(wav)
|
|
if mp.exists():
|
|
try:
|
|
return json.loads(mp.read_text())
|
|
except Exception:
|
|
pass
|
|
# Auto-detect flag and gender from voice_id
|
|
parts = wav.stem.split("_", 2)
|
|
lang = parts[0].upper() if parts else ""
|
|
gender = parts[1].upper() if len(parts) >= 2 and parts[1].upper() in ("F", "M", "N") else ""
|
|
return {
|
|
"note": "",
|
|
"rating": 0,
|
|
"flag": _LANG_FLAG_DEFAULT.get(lang, ""),
|
|
"gender": gender,
|
|
"enabled": True,
|
|
}
|
|
|
|
|
|
def _save_meta(wav: Path, meta: dict) -> None:
|
|
_meta_path(wav).write_text(json.dumps(meta, indent=2))
|
|
|
|
|
|
def _picture_path(wav: Path) -> Path | None:
|
|
for ext in _PICTURE_EXTS:
|
|
p = wav.with_suffix(ext)
|
|
if p.exists():
|
|
return p
|
|
return None
|
|
|
|
|
|
def _picture_mime(path: Path) -> str:
|
|
return {".jpg": "image/jpeg", ".jpeg": "image/jpeg",
|
|
".png": "image/png", ".webp": "image/webp"}.get(path.suffix.lower(), "image/jpeg")
|
|
|
|
|
|
def _hidden_voices_dir(settings: dict) -> Path:
|
|
return Path(settings.get("voices_scan_dir", _VOICES_DIR_DEFAULT)) / "hidden_voices"
|
|
|
|
|
|
def _active_voices_dir(settings: dict) -> Path:
|
|
return Path(settings.get("output_dir", _OUTPUT_DIR_DEFAULT))
|
|
|
|
|
|
def _backup_path(audio: Path) -> Path:
|
|
return audio.with_name(f".{audio.stem}.original{audio.suffix}.bak")
|
|
|
|
|
|
def _legacy_backup_path(audio: Path) -> Path:
|
|
return audio.with_name(f"{audio.stem}.original{audio.suffix}")
|
|
|
|
|
|
def _backup_candidates(audio: Path, meta: dict | None = None) -> list[Path]:
|
|
candidates = [_backup_path(audio), _legacy_backup_path(audio)]
|
|
if meta and meta.get("original_backup"):
|
|
candidates.insert(0, Path(str(meta["original_backup"])))
|
|
|
|
seen: set[Path] = set()
|
|
unique = []
|
|
for candidate in candidates:
|
|
try:
|
|
key = candidate.resolve() if candidate.exists() else candidate
|
|
except Exception:
|
|
key = candidate
|
|
if key not in seen:
|
|
seen.add(key)
|
|
unique.append(candidate)
|
|
return unique
|
|
|
|
|
|
def _backup_audio_suffix(backup: Path, voice_id: str) -> str | None:
|
|
name = backup.name
|
|
for ext in _AUDIO_EXTS:
|
|
if name == f".{voice_id}.original{ext}.bak" or name == f"{voice_id}.original{ext}":
|
|
return ext
|
|
return None
|
|
|
|
|
|
def _remove_audio_variants(parent: Path, voice_id: str, keep: Path | None = None) -> None:
|
|
keep_resolved = keep.resolve() if keep and keep.exists() else None
|
|
for ext in _AUDIO_EXTS:
|
|
p = parent / f"{voice_id}{ext}"
|
|
if p.exists() and not _is_internal_voice_file(p) and (keep_resolved is None or p.resolve() != keep_resolved):
|
|
p.unlink()
|
|
|
|
|
|
def _voice_package_paths(audio: Path) -> list[Path]:
|
|
parent = audio.parent
|
|
paths = []
|
|
for sfx in _AUDIO_EXTS + [".reference.txt", ".meta.json"] + _PICTURE_EXTS:
|
|
p = parent / f"{audio.stem}{sfx}"
|
|
if p.exists():
|
|
paths.append(p)
|
|
paths.extend(
|
|
p for p in _backup_candidates(audio, _load_meta(audio))
|
|
if p.exists() and p.parent.resolve() == parent.resolve()
|
|
)
|
|
return paths
|
|
|
|
|
|
def _remove_voice_package(audio: Path, keep: set[Path] | None = None) -> None:
|
|
keep_resolved = {p.resolve() for p in (keep or set()) if p.exists()}
|
|
for p in _voice_package_paths(audio):
|
|
if p.exists() and p.resolve() not in keep_resolved:
|
|
p.unlink()
|
|
|
|
|
|
def _backup_original_voice(audio: Path) -> Path | None:
|
|
if not audio.exists() or audio.suffix.lower() not in _AUDIO_EXTS:
|
|
return None
|
|
backup = _backup_path(audio)
|
|
if not backup.exists():
|
|
legacy = _legacy_backup_path(audio)
|
|
if legacy.exists():
|
|
shutil.move(str(legacy), str(backup))
|
|
else:
|
|
shutil.copy2(str(audio), str(backup))
|
|
return backup
|
|
|
|
|
|
def _voice_audio_from_request(data: dict, scan_dir: Path) -> Path | None:
|
|
from core.validation import _safe_child_path
|
|
requested_path = data.get("path")
|
|
if requested_path:
|
|
p = _safe_child_path(scan_dir, Path(requested_path))
|
|
if not p.exists() or not p.is_file() or p.suffix.lower() not in _AUDIO_EXTS:
|
|
raise HTTPException(404, "Voice file not found")
|
|
return p
|
|
|
|
voice_id = data.get("voice_id", "")
|
|
if not voice_id:
|
|
raise HTTPException(400, "voice_id or path is required")
|
|
return _find_voice_audio(voice_id, scan_dir)
|
|
|
|
|
|
def _move_voice_package(audio: Path, target_dir: Path) -> Path:
|
|
if audio.parent.resolve() == target_dir.resolve():
|
|
return audio
|
|
|
|
target_dir.mkdir(parents=True, exist_ok=True)
|
|
paths = _voice_package_paths(audio)
|
|
for src in paths:
|
|
dest = target_dir / src.name
|
|
if dest.exists() and dest.resolve() != src.resolve():
|
|
raise HTTPException(409, f"Target file already exists: {dest}")
|
|
|
|
moved_audio = target_dir / audio.name
|
|
for src in paths:
|
|
dest = target_dir / src.name
|
|
if dest.resolve() != src.resolve():
|
|
shutil.move(str(src), str(dest))
|
|
return moved_audio
|
|
|
|
|
|
def _read_reference_text(audio: Path) -> tuple[bool, str]:
|
|
ref = audio.with_suffix(".reference.txt")
|
|
if not ref.exists():
|
|
return False, ""
|
|
return True, ref.read_text(encoding="utf-8").strip()
|
|
|
|
|
|
# ── Voice health and entry ────────────────────────────────────────────────────
|
|
|
|
def _voice_health(
|
|
p: Path,
|
|
*,
|
|
meta: dict | None = None,
|
|
duration: float | None = None,
|
|
transcript: str | None = None,
|
|
) -> dict:
|
|
if transcript is None:
|
|
_has_ref, transcript = _read_reference_text(p)
|
|
if duration is None:
|
|
try:
|
|
duration = float(_duration(p))
|
|
except Exception:
|
|
duration = 0.0
|
|
else:
|
|
duration = float(duration)
|
|
if meta is None:
|
|
meta = _load_meta(p)
|
|
words = re.findall(r"\b[\w'-]+\b", transcript, flags=re.UNICODE)
|
|
word_count = len(words)
|
|
words_per_sec = (word_count / duration) if duration > 0 else 0.0
|
|
warnings: list[str] = []
|
|
if not transcript:
|
|
warnings.append("Missing reference transcript")
|
|
if duration > 25:
|
|
warnings.append("Reference audio is longer than the recommended 10-20 seconds")
|
|
if duration > 0 and word_count and words_per_sec > 4.5:
|
|
warnings.append("Reference transcript is too long for the audio; re-transcribe or shorten it")
|
|
from core.audio import _loudness_info as _li
|
|
loudness = meta.get("loudness", {})
|
|
peak = loudness.get("peak_dbfs")
|
|
if peak is not None and float(peak) > -0.1:
|
|
warnings.append("Reference audio is clipping or too loud")
|
|
return {
|
|
"ok": not warnings,
|
|
"warnings": warnings,
|
|
"duration": round(duration, 2) if duration else None,
|
|
"word_count": word_count,
|
|
"words_per_sec": round(words_per_sec, 2) if words_per_sec else 0,
|
|
"loudness": loudness,
|
|
}
|
|
|
|
|
|
def _voice_entry(p: Path) -> dict:
|
|
meta = _load_meta(p)
|
|
parts = p.stem.split("_", 2)
|
|
lang = parts[0].upper() if parts else ""
|
|
try:
|
|
duration = round(_duration(p), 2)
|
|
except Exception:
|
|
duration = None
|
|
has_ref, transcript = _read_reference_text(p)
|
|
loudness = meta.get("loudness", {})
|
|
benchmark = meta.get("benchmark", {})
|
|
if not meta.get("flag") and lang in _LANG_FLAG_DEFAULT:
|
|
meta["flag"] = _LANG_FLAG_DEFAULT[lang]
|
|
health = _voice_health(p, meta=meta, duration=duration, transcript=transcript)
|
|
return {
|
|
"id": p.stem,
|
|
"path": str(p),
|
|
"file_type": p.suffix.lower().lstrip("."),
|
|
"duration": duration,
|
|
"loudness": loudness,
|
|
"benchmark": benchmark,
|
|
"has_ref": has_ref,
|
|
"transcript": transcript,
|
|
"has_picture": _picture_path(p) is not None,
|
|
"health": health,
|
|
"lang": lang,
|
|
**meta,
|
|
}
|
|
|
|
|
|
# ── Benchmark helpers ─────────────────────────────────────────────────────────
|
|
|
|
def _benchmark_advice(audio: Path, elapsed_sec: float | None, audio_sec: float | None,
|
|
clipped: bool = False, error: str = "") -> tuple[bool, list[str]]:
|
|
advice: list[str] = []
|
|
try:
|
|
ref_duration = float(_duration(audio))
|
|
except Exception:
|
|
ref_duration = 0.0
|
|
|
|
name = audio.stem.lower()
|
|
if ref_duration >= 40 or "privat_" in name or "privat-" in name:
|
|
advice.append("Avoid for real-time assistants; long clone samples often benchmark slowly. Trim or remake as a 10-20 second voice.")
|
|
elif ref_duration > 25:
|
|
advice.append("Reference is longer than recommended. Trim to a clean 10-20 second sample.")
|
|
if clipped:
|
|
advice.append("Generated output hit the max-duration guard. Re-transcribe the reference exactly or remake this clone.")
|
|
if elapsed_sec is not None and elapsed_sec > 12:
|
|
advice.append("Slow synthesis. Prefer a shorter optimized voice for Open WebUI or Home Assistant.")
|
|
if audio_sec and elapsed_sec:
|
|
rtf = elapsed_sec / max(audio_sec, 0.01)
|
|
if rtf > 2.0:
|
|
advice.append("High real-time factor. Use a shorter reference, normalize volume, and remove silence/noise.")
|
|
if error:
|
|
advice.append("Benchmark failed. Check that Qwen3-TTS has rescanned this voice and that the reference files are valid.")
|
|
|
|
realtime_ok = not error and not clipped and (elapsed_sec or 999) <= 8 and ref_duration <= 25
|
|
return realtime_ok, advice
|
|
|
|
|
|
def _benchmark_summary(runs: list[dict]) -> dict:
|
|
ok_runs = [r for r in runs if r.get("ok")]
|
|
if not ok_runs:
|
|
return {}
|
|
|
|
def avg(key: str) -> float | None:
|
|
vals = [float(r[key]) for r in ok_runs if r.get(key) is not None]
|
|
return round(sum(vals) / len(vals), 3) if vals else None
|
|
|
|
return {
|
|
"avg_ttfa_ms": avg("ttfa_ms"),
|
|
"avg_total_sec": avg("total_sec"),
|
|
"avg_audio_sec": avg("audio_sec"),
|
|
"avg_rtf": avg("rtf"),
|
|
"avg_speed": avg("speed"),
|
|
}
|
|
|
|
|
|
def _benchmark_voice(audio: Path, settings: dict, sentences: list[tuple[str, str]]) -> dict:
|
|
from core.tts_helpers import _tts_benchmark_request
|
|
runs: list[dict] = []
|
|
for label, text in sentences:
|
|
try:
|
|
runs.append(_tts_benchmark_request(text, audio.stem, settings, label))
|
|
except Exception as e:
|
|
runs.append({"ok": False, "label": label, "text": text, "error": str(e)})
|
|
|
|
ok_runs = [r for r in runs if r.get("ok")]
|
|
summary = _benchmark_summary(runs)
|
|
elapsed = summary.get("avg_total_sec") if summary else None
|
|
audio_sec = summary.get("avg_audio_sec") if summary else None
|
|
errors = [r.get("error", "Benchmark failed") for r in runs if not r.get("ok")]
|
|
realtime_ok, advice = _benchmark_advice(audio, elapsed, audio_sec, error="; ".join(errors))
|
|
if summary.get("avg_rtf") is not None and summary["avg_rtf"] > 2:
|
|
advice.append("RTF is above 2.0. This voice is likely too slow for real-time assistants.")
|
|
|
|
return {
|
|
"ok": bool(ok_runs) and not errors,
|
|
"realtime_ok": realtime_ok and not errors,
|
|
"elapsed_sec": round(elapsed, 2) if elapsed is not None else None,
|
|
"audio_sec": round(audio_sec, 2) if audio_sec is not None else None,
|
|
"rtf": round(summary.get("avg_rtf"), 2) if summary.get("avg_rtf") is not None else None,
|
|
"speed": round(summary.get("avg_speed"), 2) if summary.get("avg_speed") is not None else None,
|
|
"ttfa_ms": round(summary.get("avg_ttfa_ms"), 0) if summary.get("avg_ttfa_ms") is not None else None,
|
|
"bytes": sum(int(r.get("bytes") or 0) for r in ok_runs),
|
|
"clipped": False,
|
|
"benchmarked_at": datetime.now(timezone.utc).isoformat(),
|
|
"text": sentences[0][1] if len(sentences) == 1 else "short / medium / long",
|
|
"runs": runs,
|
|
"summary": summary,
|
|
"advice": advice,
|
|
**({"error": "; ".join(errors)} if errors else {}),
|
|
}
|