Files
avatar/avatar/tts.py
T

364 lines
16 KiB
Python
Raw Blame History

This file contains invisible Unicode characters
This file contains invisible Unicode characters that are indistinguishable to humans but may be processed differently by a computer. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Sintesi vocale: Kokoro in locale (voci italiane) oppure la voce di sistema di macOS."""
from __future__ import annotations
import re
import subprocess
import tempfile
import threading
from pathlib import Path
import numpy as np
KOKORO_VOICES = {"if_sara": "Sara (femminile)", "im_nicola": "Nicola (maschile)"}
OUT_RATE = 24000
def clean_for_speech(text: str) -> str:
"""Toglie Markdown, link ed emoji prima di leggere."""
t = re.sub(r"```[\s\S]*?```", " ", text)
t = re.sub(r"`([^`]+)`", r"\1", t)
t = re.sub(r"!\[[^\]]*\]\([^)]*\)", "", t)
t = re.sub(r"\[([^\]]+)\]\([^)]*\)", r"\1", t)
t = re.sub(r"https?://\S+", "", t)
t = re.sub(r"^#{1,6}\s+", "", t, flags=re.M)
t = re.sub(r"^\s*[-*+]\s+", "", t, flags=re.M)
t = re.sub(r"^\s*\d+\.\s+", "", t, flags=re.M)
t = re.sub(r"\*\*([^*]+)\*\*", r"\1", t)
t = re.sub(r"\*([^*]+)\*", r"\1", t)
t = re.sub(r"_([^_]+)_", r"\1", t)
t = re.sub(r"^\|.*\|$", "", t, flags=re.M)
t = re.sub(r"[*_#>|]", "", t)
t = re.sub(r"[\U0001F000-\U0001FAFF☀-➿️‍]", "", t)
return re.sub(r"\s+", " ", t).strip()
class SentenceSplitter:
"""Riceve testo incrementale e restituisce le frasi complete."""
_RE = re.compile(r'[^.!?\n]+[.!?]+["»”)]?\s+|[^\n]+\n+')
def __init__(self) -> None:
self._buf = ""
def push(self, delta: str) -> list[str]:
self._buf += delta
if self._buf.count("```") % 2 == 1:
return []
out, last = [], 0
for m in self._RE.finditer(self._buf):
out.append(m.group(0))
last = m.end()
self._buf = self._buf[last:]
return [s for s in (clean_for_speech(x) for x in out) if s]
def flush(self) -> list[str]:
rest, self._buf = clean_for_speech(self._buf), ""
return [rest] if rest else []
class KokoroVoice:
"""Kokoro-82M via il pacchetto `kokoro`; fonetica italiana con espeak-ng."""
def __init__(self, voice: str = "if_sara", speed: float = 1.0, on_status=None) -> None:
self.voice = voice
self.speed = speed
self._on_status = on_status or (lambda m: None)
self._pipe = None
self._lock = threading.Lock()
def load(self) -> None:
with self._lock:
if self._pipe is not None:
return
self._on_status("Carico la voce Kokoro…")
try:
from kokoro import KPipeline
self._pipe = KPipeline(lang_code="i", repo_id="hexgrad/Kokoro-82M")
for _ in self._pipe("ciao", voice=self.voice, speed=self.speed):
pass
finally:
self._on_status(None)
def synthesize(self, text: str) -> np.ndarray:
if self._pipe is None:
self.load()
chunks = []
with self._lock:
for _, _, audio in self._pipe(text, voice=self.voice, speed=self.speed):
if audio is not None:
arr = audio.detach().cpu().numpy() if hasattr(audio, "detach") else np.asarray(audio)
chunks.append(arr.astype(np.float32).reshape(-1))
if not chunks:
return np.zeros(0, dtype=np.float32)
return np.concatenate(chunks)
class SystemVoice:
"""Voce di macOS tramite `say`, resa in un file audio e riprodotta dall'app."""
def __init__(self, voice: str = "") -> None:
self.voice = voice
@staticmethod
def list_voices() -> list[str]:
try:
out = subprocess.run(["say", "-v", "?"], capture_output=True, text=True, timeout=10).stdout
except Exception:
return []
names = []
for line in out.splitlines():
if "it_IT" in line:
names.append(line.split(" ")[0].strip())
return sorted(set(names))
def load(self) -> None:
pass
def synthesize(self, text: str) -> np.ndarray:
import soundfile as sf
voice = self.voice or next(iter(v for v in self.list_voices() if v.startswith("Alice")), "") or ""
with tempfile.TemporaryDirectory() as tmp:
path = Path(tmp) / "say.wav"
cmd = ["say", "-o", str(path), "--data-format=LEI16@24000"]
if voice:
cmd += ["-v", voice]
cmd.append(text)
subprocess.run(cmd, check=True, timeout=120)
data, rate = sf.read(str(path), dtype="float32")
if data.ndim > 1:
data = data[:, 0]
if rate != OUT_RATE:
data = np.interp(np.arange(0, data.size, rate / OUT_RATE), np.arange(data.size), data).astype(np.float32)
return data
class ChatterboxVoice:
"""Chatterbox (Resemble AI) in un ambiente separato, avviato come servizio locale."""
PORT = 8792
_proc = None
def __init__(self, exaggeration: float = 0.6, cfg: float = 0.3, ref_audio: str = "", on_status=None) -> None:
self.exaggeration, self.cfg, self.ref_audio = exaggeration, cfg, ref_audio
self.voice = f"chatterbox:{exaggeration}:{cfg}:{ref_audio}"
self._on_status = on_status or (lambda m: None)
@classmethod
def _start(cls) -> None:
import subprocess
from pathlib import Path as _P
if cls._proc is not None and cls._proc.poll() is None:
return
base = _P(__file__).resolve().parent.parent
python = base / ".venv-chatterbox" / "bin" / "python"
if not python.exists():
raise RuntimeError("Chatterbox non installato: esegui `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts 'setuptools<81'`.")
log = open(base / "data" / "chatterbox.log", "a")
cls._proc = subprocess.Popen([str(python), str(base / "tts_chatterbox" / "server.py"), "--port", str(cls.PORT)],
stdout=log, stderr=subprocess.STDOUT, cwd=str(base))
def load(self) -> None:
import requests
import time as _t
self._start()
self._on_status("Avvio Chatterbox (la prima volta scarica il modello, circa 2 GB)…")
try:
for _ in range(600): # fino a 10 minuti (download iniziale)
try:
st = requests.get(f"http://127.0.0.1:{self.PORT}/", timeout=3).json()
if st.get("ready"):
return
if st.get("error"):
raise RuntimeError(st["error"])
except requests.RequestException:
pass
if self._proc is not None and self._proc.poll() is not None:
raise RuntimeError("Il servizio Chatterbox si è chiuso: vedi data/chatterbox.log.")
_t.sleep(1)
raise RuntimeError("Chatterbox non è pronto.")
finally:
self._on_status(None)
REFERENCE_TEXT = ("Ciao, sono la voce del tuo assistente. Oggi ti racconto come è andata la giornata, con calma e con un po' di allegria. "
"Se hai bisogno di qualcosa, basta chiedere: sono qui per aiutarti, e mi fa piacere farlo.")
PRESETS = {"preset:femminile": "if_sara", "preset:maschile": "im_nicola"}
def _reference_path(self) -> str | None:
"""File wav di riferimento: percorso dell'utente, oppure preset generato con Kokoro (una volta)."""
ref = (self.ref_audio or "").strip()
if not ref:
return None
if ref in self.PRESETS:
from pathlib import Path as _P
import soundfile as sf
out = _P(__file__).resolve().parent.parent / "data" / "voices" / f"{ref.split(':')[1]}.wav"
if not out.exists():
out.parent.mkdir(parents=True, exist_ok=True)
self._on_status("Preparo il campione di voce per Chatterbox…")
k = KokoroVoice(self.PRESETS[ref], speed=0.95)
k.load()
sf.write(str(out), k.synthesize(self.REFERENCE_TEXT), OUT_RATE)
self._on_status(None)
return str(out)
return ref
def synthesize(self, text: str) -> np.ndarray:
import requests
r = requests.post(f"http://127.0.0.1:{self.PORT}/tts", json={"text": text, "language": "it", "exaggeration": self.exaggeration,
"cfg": self.cfg, "ref": self._reference_path()}, timeout=600)
if r.status_code != 200:
raise RuntimeError(r.json().get("error", r.text))
sr = int(r.headers.get("X-Sample-Rate", "24000"))
data = np.frombuffer(r.content, dtype=np.float32)
if sr != OUT_RATE:
data = np.interp(np.arange(0, data.size, sr / OUT_RATE), np.arange(data.size), data).astype(np.float32)
return data
@classmethod
def stop(cls) -> None:
if cls._proc is not None and cls._proc.poll() is None:
cls._proc.terminate()
cls._proc = None
class ElevenLabsVoice:
"""ElevenLabs (cloud): voci molto espressive, prima parola in circa mezzo secondo."""
API = "https://api.elevenlabs.io/v1"
def __init__(self, api_key: str, voice_id: str, model: str = "eleven_flash_v2_5", stability: float = 0.45, similarity: float = 0.8, style: float = 0.3) -> None:
self.api_key, self.voice_id, self.model = api_key, voice_id, model
self.stability, self.similarity, self.style = stability, similarity, style
self.voice = f"elevenlabs:{voice_id}:{model}:{stability}:{style}"
@classmethod
def list_voices(cls, api_key: str) -> list[tuple[str, str]]:
import requests
r = requests.get(f"{cls.API}/voices", headers={"xi-api-key": api_key}, timeout=15)
r.raise_for_status()
out = []
for v in r.json().get("voices", []):
labels = v.get("labels") or {}
desc = ", ".join(x for x in (labels.get("gender"), labels.get("accent"), labels.get("language"), v.get("category")) if x)
out.append((v["voice_id"], f"{v['name']}" + (f" ({desc})" if desc else "")))
return out
def load(self) -> None:
if not self.api_key:
raise RuntimeError("ElevenLabs: manca la chiave API (Motore e Voce).")
if not self.voice_id:
raise RuntimeError("ElevenLabs: scegli una voce in Motore e Voce (pulsante «Carica voci»).")
def synthesize(self, text: str) -> np.ndarray:
import requests
r = requests.post(f"{self.API}/text-to-speech/{self.voice_id}", params={"output_format": "pcm_24000"},
headers={"xi-api-key": self.api_key, "Content-Type": "application/json"},
json={"text": text, "model_id": self.model, "language_code": "it",
"voice_settings": {"stability": self.stability, "similarity_boost": self.similarity, "style": self.style, "use_speaker_boost": True}},
timeout=60)
if r.status_code != 200:
try:
detail = r.json().get("detail", {})
msg = detail.get("message") if isinstance(detail, dict) else str(detail)
except Exception:
msg = r.text[:200]
raise RuntimeError(f"ElevenLabs {r.status_code}: {msg}")
return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0
class VoiceboxVoice:
"""Voicebox (app locale): Qwen3-TTS, Chatterbox, Kokoro… con accelerazione MLX e profili vocali clonati."""
API = "http://127.0.0.1:17493"
def __init__(self, profile_id: str, engine: str = "qwen", instruct: str = "", on_status=None) -> None:
self.profile_id, self.engine, self.instruct = profile_id, engine, instruct
self.voice = f"voicebox:{profile_id}:{engine}:{instruct}"
self._on_status = on_status or (lambda m: None)
@classmethod
def alive(cls) -> bool:
import requests
try:
return requests.get(f"{cls.API}/health", timeout=3).ok
except requests.RequestException:
return False
@classmethod
def ensure_running(cls, wait: int = 60) -> None:
import subprocess
import time as _t
if cls.alive():
return
subprocess.run(["open", "-a", "Voicebox"], check=False, timeout=15)
for _ in range(wait):
_t.sleep(1)
if cls.alive():
return
raise RuntimeError("Voicebox non risponde: installalo da github.com/jamiepine/voicebox e aprilo una volta.")
@classmethod
def list_profiles(cls) -> list[tuple[str, str]]:
import requests
cls.ensure_running()
d = requests.get(f"{cls.API}/profiles", timeout=15).json()
d = d if isinstance(d, list) else d.get("profiles", [])
return [(p["id"], f"{p.get('name')} ({p.get('language') or '?'}, {p.get('voice_type') or ''})") for p in d]
def load(self) -> None:
if not self.profile_id:
raise RuntimeError("Voicebox: scegli un profilo vocale in Motore e Voce (pulsante «Carica profili»).")
self._on_status("Avvio Voicebox…")
try:
self.ensure_running()
finally:
self._on_status(None)
def synthesize(self, text: str) -> np.ndarray:
import io
import time as _t
import requests
import soundfile as sf
body = {"profile_id": self.profile_id, "text": text, "language": "it", "engine": self.engine}
if self.engine == "qwen":
body["model_size"] = "1.7B"
if self.instruct and self.engine in ("qwen", "qwen_custom_voice"):
body["instruct"] = self.instruct[:500]
r = requests.post(f"{self.API}/generate", json=body, timeout=120)
if not r.ok:
raise RuntimeError(f"Voicebox {r.status_code}: {r.text[:160]}")
g = r.json()
gid = g["id"]
t0 = _t.time()
# /generate/{id}/status è un flusso SSE: per il polling si usa /history/{id} (JSON).
while g.get("status") in ("generating", "pending", "queued", "processing") and _t.time() - t0 < 300:
_t.sleep(0.25)
h = requests.get(f"{self.API}/history/{gid}", timeout=10)
if h.ok and h.headers.get("Content-Type", "").startswith("application/json"):
g = h.json()
if g.get("status") != "completed":
raise RuntimeError(f"Voicebox: {g.get('status')} {g.get('error') or ''}".strip())
a = requests.get(f"{self.API}/audio/{gid}", timeout=60)
data, sr = sf.read(io.BytesIO(a.content), dtype="float32")
if data.ndim > 1:
data = data[:, 0]
if sr != OUT_RATE:
data = np.interp(np.arange(0, data.size, sr / OUT_RATE), np.arange(data.size), data).astype(np.float32)
return data
def make_voice(settings, on_status=None):
if settings.get("tts_engine") == "voicebox":
return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"),
str(settings.get("voicebox_instruct") or ""), on_status=on_status)
if settings.get("tts_engine") == "elevenlabs":
return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""),
str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"),
float(settings.get("elevenlabs_stability", 0.45)), 0.8, float(settings.get("elevenlabs_style", 0.3)))
if settings.get("tts_engine") == "chatterbox":
return ChatterboxVoice(float(settings.get("chatterbox_exaggeration", 0.6)), float(settings.get("chatterbox_cfg", 0.3)),
str(settings.get("chatterbox_ref", "") or ""), on_status=on_status)
if settings.get("tts_engine") == "system":
return SystemVoice(settings.get("system_voice", ""))
return KokoroVoice(settings.get("kokoro_voice", "if_sara"), on_status=on_status)