diff --git a/avatar/assistant.py b/avatar/assistant.py index 82a3c97..5f49d3d 100644 --- a/avatar/assistant.py +++ b/avatar/assistant.py @@ -234,7 +234,7 @@ class Assistant: self.settings.save() else: eng = self.settings.get("tts_engine") - label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs") else + label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs", "voicebox") else {"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara")) try: if cm.get_voice() != label: diff --git a/avatar/settings.py b/avatar/settings.py index 0c0d6df..dd0499b 100644 --- a/avatar/settings.py +++ b/avatar/settings.py @@ -26,6 +26,9 @@ DEFAULTS: dict[str, Any] = { "kokoro_voice": "if_sara", "chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica "chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo + "voicebox_profile_id": "", + "voicebox_engine": "qwen", # qwen | chatterbox | chatterbox_turbo | kokoro | luxtts + "voicebox_instruct": "", # istruzione di stile per Qwen (es. "parla in modo caloroso e calmo") "elevenlabs_voice_id": "", "elevenlabs_model": "eleven_flash_v2_5", # rapido; eleven_multilingual_v2 = qualità massima "elevenlabs_stability": 0.45, diff --git a/avatar/settings_dialog.py b/avatar/settings_dialog.py index 2cf0a46..2efdc4a 100644 --- a/avatar/settings_dialog.py +++ b/avatar/settings_dialog.py @@ -99,7 +99,7 @@ class SettingsDialog(QDialog): self.stack.setCurrentIndex(self.provider.currentIndex()) form2 = QFormLayout() - self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine")) + self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine")) form2.addRow("Motore voce", self.tts_engine) self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice) self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice")) @@ -116,6 +116,16 @@ class SettingsDialog(QDialog): self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1) form2.addRow("Voce Chatterbox", row) row = QHBoxLayout() + self.vb_profile = QComboBox(); self.vb_profile.setMinimumWidth(220) + self._vb_current = str(s.get("voicebox_profile_id") or "") + self.vb_profile.addItem("(carica i profili da Voicebox)" if not self._vb_current else f"profilo salvato: {self._vb_current[:8]}…", self._vb_current) + row.addWidget(self.vb_profile, 1) + b = QPushButton("Carica profili"); b.clicked.connect(self._vb_profiles); row.addWidget(b) + self.vb_engine = _combo([("qwen", "Qwen3-TTS (consigliato)"), ("kokoro", "Kokoro"), ("chatterbox_turbo", "Chatterbox Turbo"), ("chatterbox", "Chatterbox"), ("luxtts", "LuxTTS")], s.get("voicebox_engine")); row.addWidget(self.vb_engine) + form2.addRow("Voicebox", row) + self.vb_instruct = QLineEdit(s.get("voicebox_instruct") or ""); self.vb_instruct.setPlaceholderText("istruzione di stile per Qwen, es. «parla in modo caloroso e calmo» (facoltativa)") + form2.addRow("", self.vb_instruct) + row = QHBoxLayout() self.el_key = QLineEdit(); self.el_key.setEchoMode(QLineEdit.EchoMode.Password) self.el_key.setPlaceholderText("•••••• (salvata)" if s.get_secret("elevenlabs_api_key") else "chiave API da elevenlabs.io"); row.addWidget(self.el_key, 1) self.el_voice = QComboBox(); self.el_voice.setMinimumWidth(220); row.addWidget(self.el_voice, 1) @@ -313,6 +323,15 @@ class SettingsDialog(QDialog): self._async.emit("wa", (f"Errore: {err}", None)) threading.Thread(target=work, daemon=True).start() + def _vb_profiles(self) -> None: + def work(): + try: + from .tts import VoiceboxVoice + self._async.emit("vb", VoiceboxVoice.list_profiles()) + except Exception as err: + self._async.emit("vb_error", str(err)[:120]) + threading.Thread(target=work, daemon=True).start() + def _el_voices(self) -> None: key = self.el_key.text().strip() or self.settings.get_secret("elevenlabs_api_key") if not key: @@ -328,6 +347,15 @@ class SettingsDialog(QDialog): threading.Thread(target=work, daemon=True).start() def _on_async(self, kind: str, payload) -> None: + if kind == "vb": + self.vb_profile.clear() + for pid, label in payload: + self.vb_profile.addItem(label, pid) + idx = self.vb_profile.findData(self._vb_current) + self.vb_profile.setCurrentIndex(idx if idx >= 0 else 0) + return + if kind == "vb_error": + self.vb_profile.clear(); self.vb_profile.addItem(f"errore: {payload}", ""); return if kind == "el": self.el_voice.clear() for vid, label in payload: @@ -389,6 +417,7 @@ class SettingsDialog(QDialog): "claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(), "tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(), "system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(), + "voicebox_profile_id": self.vb_profile.currentData() or "", "voicebox_engine": self.vb_engine.currentData(), "voicebox_instruct": self.vb_instruct.text().strip(), "elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(), "elevenlabs_stability": float(self.el_stab.value()), "elevenlabs_style": float(self.el_style.value()), "chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()), diff --git a/avatar/tts.py b/avatar/tts.py index d795d48..6b2fdbf 100644 --- a/avatar/tts.py +++ b/avatar/tts.py @@ -266,7 +266,91 @@ class ElevenLabsVoice: return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0 +class VoiceboxVoice: + """Voicebox (app locale): Qwen3-TTS, Chatterbox, Kokoro… con accelerazione MLX e profili vocali clonati.""" + + API = "http://127.0.0.1:17493" + + def __init__(self, profile_id: str, engine: str = "qwen", instruct: str = "", on_status=None) -> None: + self.profile_id, self.engine, self.instruct = profile_id, engine, instruct + self.voice = f"voicebox:{profile_id}:{engine}:{instruct}" + self._on_status = on_status or (lambda m: None) + + @classmethod + def alive(cls) -> bool: + import requests + try: + return requests.get(f"{cls.API}/health", timeout=3).ok + except requests.RequestException: + return False + + @classmethod + def ensure_running(cls, wait: int = 60) -> None: + import subprocess + import time as _t + if cls.alive(): + return + subprocess.run(["open", "-a", "Voicebox"], check=False, timeout=15) + for _ in range(wait): + _t.sleep(1) + if cls.alive(): + return + raise RuntimeError("Voicebox non risponde: installalo da github.com/jamiepine/voicebox e aprilo una volta.") + + @classmethod + def list_profiles(cls) -> list[tuple[str, str]]: + import requests + cls.ensure_running() + d = requests.get(f"{cls.API}/profiles", timeout=15).json() + d = d if isinstance(d, list) else d.get("profiles", []) + return [(p["id"], f"{p.get('name')} ({p.get('language') or '?'}, {p.get('voice_type') or ''})") for p in d] + + def load(self) -> None: + if not self.profile_id: + raise RuntimeError("Voicebox: scegli un profilo vocale in Motore e Voce (pulsante «Carica profili»).") + self._on_status("Avvio Voicebox…") + try: + self.ensure_running() + finally: + self._on_status(None) + + def synthesize(self, text: str) -> np.ndarray: + import io + import time as _t + import requests + import soundfile as sf + body = {"profile_id": self.profile_id, "text": text, "language": "it", "engine": self.engine} + if self.engine == "qwen": + body["model_size"] = "1.7B" + if self.instruct and self.engine in ("qwen", "qwen_custom_voice"): + body["instruct"] = self.instruct[:500] + r = requests.post(f"{self.API}/generate", json=body, timeout=120) + if not r.ok: + raise RuntimeError(f"Voicebox {r.status_code}: {r.text[:160]}") + g = r.json() + gid = g["id"] + t0 = _t.time() + # /generate/{id}/status è un flusso SSE: per il polling si usa /history/{id} (JSON). + while g.get("status") in ("generating", "pending", "queued", "processing") and _t.time() - t0 < 300: + _t.sleep(0.25) + h = requests.get(f"{self.API}/history/{gid}", timeout=10) + if h.ok and h.headers.get("Content-Type", "").startswith("application/json"): + g = h.json() + if g.get("status") != "completed": + raise RuntimeError(f"Voicebox: {g.get('status')} {g.get('error') or ''}".strip()) + a = requests.get(f"{self.API}/audio/{gid}", timeout=60) + data, sr = sf.read(io.BytesIO(a.content), dtype="float32") + if data.ndim > 1: + data = data[:, 0] + if sr != OUT_RATE: + data = np.interp(np.arange(0, data.size, sr / OUT_RATE), np.arange(data.size), data).astype(np.float32) + return data + + def make_voice(settings, on_status=None): + if settings.get("tts_engine") == "voicebox": + return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"), + str(settings.get("voicebox_instruct") or ""), on_status=on_status) if settings.get("tts_engine") == "elevenlabs": return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""), str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"), diff --git a/samples/voicebox-chatterbox.wav b/samples/voicebox-chatterbox.wav new file mode 100644 index 0000000..74232c0 Binary files /dev/null and b/samples/voicebox-chatterbox.wav differ diff --git a/samples/voicebox-kokoro.wav b/samples/voicebox-kokoro.wav new file mode 100644 index 0000000..400364b Binary files /dev/null and b/samples/voicebox-kokoro.wav differ diff --git a/samples/voicebox-qwen-istr.wav b/samples/voicebox-qwen-istr.wav new file mode 100644 index 0000000..542ba44 Binary files /dev/null and b/samples/voicebox-qwen-istr.wav differ diff --git a/samples/voicebox-qwen.wav b/samples/voicebox-qwen.wav new file mode 100644 index 0000000..6f96a21 Binary files /dev/null and b/samples/voicebox-qwen.wav differ