Motore voce Voicebox (Qwen3-TTS via MLX, profili clonati, istruzioni di stile)

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
lucianoandClaude Fable 5.1 committed 2026-09-24 10:22:15 +02:00
1 parent 950e70b2c7
commit 50d2319e22
8 files changed
+118 -2

No files matched your search

+1 -1
View File
@@ -234,7 +234,7 @@ class Assistant:
self.settings.save() self.settings.save()
else: else:
eng = self.settings.get("tts_engine") eng = self.settings.get("tts_engine")
label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs") else label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs", "voicebox") else
{"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara")) {"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara"))
try: try:
if cm.get_voice() != label: if cm.get_voice() != label:
+3
View File
@@ -26,6 +26,9 @@ DEFAULTS: dict[str, Any] = {
"kokoro_voice": "if_sara", "kokoro_voice": "if_sara",
"chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica "chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica
"chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo "chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo
"voicebox_profile_id": "",
"voicebox_engine": "qwen", # qwen | chatterbox | chatterbox_turbo | kokoro | luxtts
"voicebox_instruct": "", # istruzione di stile per Qwen (es. "parla in modo caloroso e calmo")
"elevenlabs_voice_id": "", "elevenlabs_voice_id": "",
"elevenlabs_model": "eleven_flash_v2_5", # rapido; eleven_multilingual_v2 = qualità massima "elevenlabs_model": "eleven_flash_v2_5", # rapido; eleven_multilingual_v2 = qualità massima
"elevenlabs_stability": 0.45, "elevenlabs_stability": 0.45,
+30 -1
View File
@@ -99,7 +99,7 @@ class SettingsDialog(QDialog):
self.stack.setCurrentIndex(self.provider.currentIndex()) self.stack.setCurrentIndex(self.provider.currentIndex())
form2 = QFormLayout() form2 = QFormLayout()
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine")) self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
form2.addRow("Motore voce", self.tts_engine) form2.addRow("Motore voce", self.tts_engine)
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice) self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice")) self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
@@ -116,6 +116,16 @@ class SettingsDialog(QDialog):
self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1) self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1)
form2.addRow("Voce Chatterbox", row) form2.addRow("Voce Chatterbox", row)
row = QHBoxLayout() row = QHBoxLayout()
self.vb_profile = QComboBox(); self.vb_profile.setMinimumWidth(220)
self._vb_current = str(s.get("voicebox_profile_id") or "")
self.vb_profile.addItem("(carica i profili da Voicebox)" if not self._vb_current else f"profilo salvato: {self._vb_current[:8]}…", self._vb_current)
row.addWidget(self.vb_profile, 1)
b = QPushButton("Carica profili"); b.clicked.connect(self._vb_profiles); row.addWidget(b)
self.vb_engine = _combo([("qwen", "Qwen3-TTS (consigliato)"), ("kokoro", "Kokoro"), ("chatterbox_turbo", "Chatterbox Turbo"), ("chatterbox", "Chatterbox"), ("luxtts", "LuxTTS")], s.get("voicebox_engine")); row.addWidget(self.vb_engine)
form2.addRow("Voicebox", row)
self.vb_instruct = QLineEdit(s.get("voicebox_instruct") or ""); self.vb_instruct.setPlaceholderText("istruzione di stile per Qwen, es. «parla in modo caloroso e calmo» (facoltativa)")
form2.addRow("", self.vb_instruct)
row = QHBoxLayout()
self.el_key = QLineEdit(); self.el_key.setEchoMode(QLineEdit.EchoMode.Password) self.el_key = QLineEdit(); self.el_key.setEchoMode(QLineEdit.EchoMode.Password)
self.el_key.setPlaceholderText("•••••• (salvata)" if s.get_secret("elevenlabs_api_key") else "chiave API da elevenlabs.io"); row.addWidget(self.el_key, 1) self.el_key.setPlaceholderText("•••••• (salvata)" if s.get_secret("elevenlabs_api_key") else "chiave API da elevenlabs.io"); row.addWidget(self.el_key, 1)
self.el_voice = QComboBox(); self.el_voice.setMinimumWidth(220); row.addWidget(self.el_voice, 1) self.el_voice = QComboBox(); self.el_voice.setMinimumWidth(220); row.addWidget(self.el_voice, 1)
@@ -313,6 +323,15 @@ class SettingsDialog(QDialog):
self._async.emit("wa", (f"Errore: {err}", None)) self._async.emit("wa", (f"Errore: {err}", None))
threading.Thread(target=work, daemon=True).start() threading.Thread(target=work, daemon=True).start()
def _vb_profiles(self) -> None:
def work():
try:
from .tts import VoiceboxVoice
self._async.emit("vb", VoiceboxVoice.list_profiles())
except Exception as err:
self._async.emit("vb_error", str(err)[:120])
threading.Thread(target=work, daemon=True).start()
def _el_voices(self) -> None: def _el_voices(self) -> None:
key = self.el_key.text().strip() or self.settings.get_secret("elevenlabs_api_key") key = self.el_key.text().strip() or self.settings.get_secret("elevenlabs_api_key")
if not key: if not key:
@@ -328,6 +347,15 @@ class SettingsDialog(QDialog):
threading.Thread(target=work, daemon=True).start() threading.Thread(target=work, daemon=True).start()
def _on_async(self, kind: str, payload) -> None: def _on_async(self, kind: str, payload) -> None:
if kind == "vb":
self.vb_profile.clear()
for pid, label in payload:
self.vb_profile.addItem(label, pid)
idx = self.vb_profile.findData(self._vb_current)
self.vb_profile.setCurrentIndex(idx if idx >= 0 else 0)
return
if kind == "vb_error":
self.vb_profile.clear(); self.vb_profile.addItem(f"errore: {payload}", ""); return
if kind == "el": if kind == "el":
self.el_voice.clear() self.el_voice.clear()
for vid, label in payload: for vid, label in payload:
@@ -389,6 +417,7 @@ class SettingsDialog(QDialog):
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(), "claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(), "tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(), "system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
"voicebox_profile_id": self.vb_profile.currentData() or "", "voicebox_engine": self.vb_engine.currentData(), "voicebox_instruct": self.vb_instruct.text().strip(),
"elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(), "elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(),
"elevenlabs_stability": float(self.el_stab.value()), "elevenlabs_style": float(self.el_style.value()), "elevenlabs_stability": float(self.el_stab.value()), "elevenlabs_style": float(self.el_style.value()),
"chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()), "chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()),
+84
View File
@@ -266,7 +266,91 @@ class ElevenLabsVoice:
return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0 return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0
class VoiceboxVoice:
"""Voicebox (app locale): Qwen3-TTS, Chatterbox, Kokoro… con accelerazione MLX e profili vocali clonati."""
API = "http://127.0.0.1:17493"
def __init__(self, profile_id: str, engine: str = "qwen", instruct: str = "", on_status=None) -> None:
self.profile_id, self.engine, self.instruct = profile_id, engine, instruct
self.voice = f"voicebox:{profile_id}:{engine}:{instruct}"
self._on_status = on_status or (lambda m: None)
@classmethod
def alive(cls) -> bool:
import requests
try:
return requests.get(f"{cls.API}/health", timeout=3).ok
except requests.RequestException:
return False
@classmethod
def ensure_running(cls, wait: int = 60) -> None:
import subprocess
import time as _t
if cls.alive():
return
subprocess.run(["open", "-a", "Voicebox"], check=False, timeout=15)
for _ in range(wait):
_t.sleep(1)
if cls.alive():
return
raise RuntimeError("Voicebox non risponde: installalo da github.com/jamiepine/voicebox e aprilo una volta.")
@classmethod
def list_profiles(cls) -> list[tuple[str, str]]:
import requests
cls.ensure_running()
d = requests.get(f"{cls.API}/profiles", timeout=15).json()
d = d if isinstance(d, list) else d.get("profiles", [])
return [(p["id"], f"{p.get('name')} ({p.get('language') or '?'}, {p.get('voice_type') or ''})") for p in d]
def load(self) -> None:
if not self.profile_id:
raise RuntimeError("Voicebox: scegli un profilo vocale in Motore e Voce (pulsante «Carica profili»).")
self._on_status("Avvio Voicebox…")
try:
self.ensure_running()
finally:
self._on_status(None)
def synthesize(self, text: str) -> np.ndarray:
import io
import time as _t
import requests
import soundfile as sf
body = {"profile_id": self.profile_id, "text": text, "language": "it", "engine": self.engine}
if self.engine == "qwen":
body["model_size"] = "1.7B"
if self.instruct and self.engine in ("qwen", "qwen_custom_voice"):
body["instruct"] = self.instruct[:500]
r = requests.post(f"{self.API}/generate", json=body, timeout=120)
if not r.ok:
raise RuntimeError(f"Voicebox {r.status_code}: {r.text[:160]}")
g = r.json()
gid = g["id"]
t0 = _t.time()
# /generate/{id}/status è un flusso SSE: per il polling si usa /history/{id} (JSON).
while g.get("status") in ("generating", "pending", "queued", "processing") and _t.time() - t0 < 300:
_t.sleep(0.25)
h = requests.get(f"{self.API}/history/{gid}", timeout=10)
if h.ok and h.headers.get("Content-Type", "").startswith("application/json"):
g = h.json()
if g.get("status") != "completed":
raise RuntimeError(f"Voicebox: {g.get('status')} {g.get('error') or ''}".strip())
a = requests.get(f"{self.API}/audio/{gid}", timeout=60)
data, sr = sf.read(io.BytesIO(a.content), dtype="float32")
if data.ndim > 1:
data = data[:, 0]
if sr != OUT_RATE:
data = np.interp(np.arange(0, data.size, sr / OUT_RATE), np.arange(data.size), data).astype(np.float32)
return data
def make_voice(settings, on_status=None): def make_voice(settings, on_status=None):
if settings.get("tts_engine") == "voicebox":
return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"),
str(settings.get("voicebox_instruct") or ""), on_status=on_status)
if settings.get("tts_engine") == "elevenlabs": if settings.get("tts_engine") == "elevenlabs":
return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""), return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""),
str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"), str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"),
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.