From 950e70b2c7f87c8ab5efd3d7d5874ed74da6f211 Mon Sep 17 00:00:00 2001 From: luciano Date: Thu, 24 Sep 2026 09:44:56 +0200 Subject: [PATCH] Motore voce ElevenLabs Co-Authored-By: Claude Fable 5.1 --- README.md | 2 +- avatar/assistant.py | 2 +- avatar/settings.py | 6 ++++- avatar/settings_dialog.py | 43 +++++++++++++++++++++++++++++++++- avatar/tts.py | 49 +++++++++++++++++++++++++++++++++++++++ 5 files changed, 98 insertions(+), 4 deletions(-) diff --git a/README.md b/README.md index 790089a..40518ac 100644 --- a/README.md +++ b/README.md @@ -7,7 +7,7 @@ L'interfaccia è riusata con licenza CC BY-NC 4.0 (uso personale, non commercial ## Cosa fa - **Conversazione a mani libere**: il microfono è sempre in ascolto; quando smetti di parlare, la frase viene trascrizione in locale (Whisper via MLX) e inviata al motore. In alternativa: premi-per-parlare, parola di attivazione "Hey Jarvis" (openwakeword, installabile dal pannello), o il campo di testo. -- **Voce**: tre motori. *Kokoro* in locale con le voci italiane Sara e Nicola, veloce e naturale (predefinito). *Chatterbox* (Resemble AI), espressivo, con controllo dell'enfasi e clonazione da un wav di riferimento: gira in un ambiente Python separato (`.venv-chatterbox`, creato con `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts "setuptools<81"`) come servizio locale avviato dall'app, scarica circa 2 GB al primo uso e su Apple Silicon genera a circa un terzo del tempo reale, quindi con pause di alcuni secondi tra le frasi. *Voce di sistema* di macOS. Le frasi vengono sintetizzate in anticipo mentre l'app pronuncia la precedente. +- **Voce**: tre motori. *Kokoro* in locale con le voci italiane Sara e Nicola, veloce e naturale (predefinito). *Chatterbox* (Resemble AI), espressivo, con controllo dell'enfasi e clonazione da un wav di riferimento: gira in un ambiente Python separato (`.venv-chatterbox`, creato con `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts "setuptools<81"`) come servizio locale avviato dall'app, scarica circa 2 GB al primo uso e su Apple Silicon genera a circa un terzo del tempo reale, quindi con pause di alcuni secondi tra le frasi. *ElevenLabs* (cloud, chiave API da elevenlabs.io): la più espressiva, prima parola in circa mezzo secondo, voci scelte dal tuo account con il pulsante «Carica voci», modello Flash (rapido) o Multilingual v2 (qualità), regolazioni di stabilità e stile; piano gratuito di 10 mila caratteri al mese. *Voce di sistema* di macOS. Le frasi vengono sintetizzate in anticipo mentre l'app pronuncia la precedente. - **Volto**: la bocca segue lo spettro dell'audio (50 forme al secondo) fuso con il testo pronunciato; lo sguardo e le sopracciglia seguono lo stato (ascolta, pensa, parla, dorme). - **Tre motori**, selezionabili dal pulsante "Motore & Voce": - *Claude (Anthropic)*: Claude Opus 5 via API, con ricerca web server-side. Serve una chiave, salvata nel portachiavi di macOS. diff --git a/avatar/assistant.py b/avatar/assistant.py index c14cbf6..82a3c97 100644 --- a/avatar/assistant.py +++ b/avatar/assistant.py @@ -234,7 +234,7 @@ class Assistant: self.settings.save() else: eng = self.settings.get("tts_engine") - label = "Sistema" if eng == "system" else ("Sara" if eng == "chatterbox" else + label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs") else {"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara")) try: if cm.get_voice() != label: diff --git a/avatar/settings.py b/avatar/settings.py index f34186b..0c0d6df 100644 --- a/avatar/settings.py +++ b/avatar/settings.py @@ -26,6 +26,10 @@ DEFAULTS: dict[str, Any] = { "kokoro_voice": "if_sara", "chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica "chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo + "elevenlabs_voice_id": "", + "elevenlabs_model": "eleven_flash_v2_5", # rapido; eleven_multilingual_v2 = qualità massima + "elevenlabs_stability": 0.45, + "elevenlabs_style": 0.3, "chatterbox_ref": "preset:femminile", # preset:femminile | preset:maschile | percorso di un wav da imitare "system_voice": "", # "" = automatica "stt_model": "mlx-community/whisper-small-mlx", @@ -41,7 +45,7 @@ DEFAULTS: dict[str, Any] = { "monitor_escludi": "", # parole/frasi (separate da virgola) che escludono un messaggio dagli avvisi } -SECRET_KEYS = ("anthropic_api_key", "local_api_key", "telegram_api_hash") +SECRET_KEYS = ("anthropic_api_key", "local_api_key", "telegram_api_hash", "elevenlabs_api_key") class Settings: diff --git a/avatar/settings_dialog.py b/avatar/settings_dialog.py index 3c9dd11..2cf0a46 100644 --- a/avatar/settings_dialog.py +++ b/avatar/settings_dialog.py @@ -99,7 +99,7 @@ class SettingsDialog(QDialog): self.stack.setCurrentIndex(self.provider.currentIndex()) form2 = QFormLayout() - self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine")) + self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine")) form2.addRow("Motore voce", self.tts_engine) self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice) self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice")) @@ -115,6 +115,20 @@ class SettingsDialog(QDialog): row.addWidget(QLabel("voce")); row.addWidget(self.cb_voice) self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1) form2.addRow("Voce Chatterbox", row) + row = QHBoxLayout() + self.el_key = QLineEdit(); self.el_key.setEchoMode(QLineEdit.EchoMode.Password) + self.el_key.setPlaceholderText("•••••• (salvata)" if s.get_secret("elevenlabs_api_key") else "chiave API da elevenlabs.io"); row.addWidget(self.el_key, 1) + self.el_voice = QComboBox(); self.el_voice.setMinimumWidth(220); row.addWidget(self.el_voice, 1) + b = QPushButton("Carica voci"); b.clicked.connect(self._el_voices); row.addWidget(b) + form2.addRow("ElevenLabs", row) + row = QHBoxLayout() + self.el_model = _combo([("eleven_flash_v2_5", "Flash v2.5 (rapido)"), ("eleven_multilingual_v2", "Multilingual v2 (qualità)"), ("eleven_v3", "v3 (più espressivo, sperimentale)")], s.get("elevenlabs_model")); row.addWidget(QLabel("modello")); row.addWidget(self.el_model) + self.el_stab = QDoubleSpinBox(); self.el_stab.setRange(0.0, 1.0); self.el_stab.setSingleStep(0.05); self.el_stab.setValue(float(s.get("elevenlabs_stability", 0.45))); row.addWidget(QLabel("stabilità")); row.addWidget(self.el_stab) + self.el_style = QDoubleSpinBox(); self.el_style.setRange(0.0, 1.0); self.el_style.setSingleStep(0.05); self.el_style.setValue(float(s.get("elevenlabs_style", 0.3))); row.addWidget(QLabel("stile")); row.addWidget(self.el_style) + row.addStretch() + form2.addRow("", row) + self._el_current = str(s.get("elevenlabs_voice_id") or "") + self.el_voice.addItem("(carica le voci con la chiave)" if not self._el_current else f"voce salvata: {self._el_current[:10]}…", self._el_current) self.stt_model = _combo(STT_MODELS, s.get("stt_model")); form2.addRow("Riconoscimento vocale", self.stt_model) from .avatar3d import list_models models = list_models() @@ -299,7 +313,30 @@ class SettingsDialog(QDialog): self._async.emit("wa", (f"Errore: {err}", None)) threading.Thread(target=work, daemon=True).start() + def _el_voices(self) -> None: + key = self.el_key.text().strip() or self.settings.get_secret("elevenlabs_api_key") + if not key: + self.el_voice.clear(); self.el_voice.addItem("inserisci prima la chiave API", ""); return + if self.el_key.text().strip(): + self.settings.update({"elevenlabs_api_key": key}) + def work(): + try: + from .tts import ElevenLabsVoice + self._async.emit("el", ElevenLabsVoice.list_voices(key)) + except Exception as err: + self._async.emit("el_error", str(err)[:120]) + threading.Thread(target=work, daemon=True).start() + def _on_async(self, kind: str, payload) -> None: + if kind == "el": + self.el_voice.clear() + for vid, label in payload: + self.el_voice.addItem(label, vid) + idx = self.el_voice.findData(self._el_current) + self.el_voice.setCurrentIndex(idx if idx >= 0 else 0) + return + if kind == "el_error": + self.el_voice.clear(); self.el_voice.addItem(f"errore: {payload}", ""); return if kind == "wa": msg, png = payload self.wa_hint.setText(str(msg)) @@ -352,6 +389,8 @@ class SettingsDialog(QDialog): "claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(), "tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(), "system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(), + "elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(), + "elevenlabs_stability": float(self.el_stab.value()), "elevenlabs_style": float(self.el_style.value()), "chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()), "chatterbox_ref": (self.cb_ref.text().strip() if self.cb_voice.currentData() == "custom" else self.cb_voice.currentData()), "vad_threshold": float(self.vad.value()), @@ -371,6 +410,8 @@ class SettingsDialog(QDialog): values["anthropic_api_key"] = self.api_key.text().strip() if self.local_key.text().strip(): values["local_api_key"] = self.local_key.text().strip() + if self.el_key.text().strip(): + values["elevenlabs_api_key"] = self.el_key.text().strip() self.settings.update(values) self.accept() if self.on_saved: diff --git a/avatar/tts.py b/avatar/tts.py index e890218..d795d48 100644 --- a/avatar/tts.py +++ b/avatar/tts.py @@ -221,7 +221,56 @@ class ChatterboxVoice: cls._proc = None +class ElevenLabsVoice: + """ElevenLabs (cloud): voci molto espressive, prima parola in circa mezzo secondo.""" + + API = "https://api.elevenlabs.io/v1" + + def __init__(self, api_key: str, voice_id: str, model: str = "eleven_flash_v2_5", stability: float = 0.45, similarity: float = 0.8, style: float = 0.3) -> None: + self.api_key, self.voice_id, self.model = api_key, voice_id, model + self.stability, self.similarity, self.style = stability, similarity, style + self.voice = f"elevenlabs:{voice_id}:{model}:{stability}:{style}" + + @classmethod + def list_voices(cls, api_key: str) -> list[tuple[str, str]]: + import requests + r = requests.get(f"{cls.API}/voices", headers={"xi-api-key": api_key}, timeout=15) + r.raise_for_status() + out = [] + for v in r.json().get("voices", []): + labels = v.get("labels") or {} + desc = ", ".join(x for x in (labels.get("gender"), labels.get("accent"), labels.get("language"), v.get("category")) if x) + out.append((v["voice_id"], f"{v['name']}" + (f" ({desc})" if desc else ""))) + return out + + def load(self) -> None: + if not self.api_key: + raise RuntimeError("ElevenLabs: manca la chiave API (Motore e Voce).") + if not self.voice_id: + raise RuntimeError("ElevenLabs: scegli una voce in Motore e Voce (pulsante «Carica voci»).") + + def synthesize(self, text: str) -> np.ndarray: + import requests + r = requests.post(f"{self.API}/text-to-speech/{self.voice_id}", params={"output_format": "pcm_24000"}, + headers={"xi-api-key": self.api_key, "Content-Type": "application/json"}, + json={"text": text, "model_id": self.model, "language_code": "it", + "voice_settings": {"stability": self.stability, "similarity_boost": self.similarity, "style": self.style, "use_speaker_boost": True}}, + timeout=60) + if r.status_code != 200: + try: + detail = r.json().get("detail", {}) + msg = detail.get("message") if isinstance(detail, dict) else str(detail) + except Exception: + msg = r.text[:200] + raise RuntimeError(f"ElevenLabs {r.status_code}: {msg}") + return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0 + + def make_voice(settings, on_status=None): + if settings.get("tts_engine") == "elevenlabs": + return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""), + str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"), + float(settings.get("elevenlabs_stability", 0.45)), 0.8, float(settings.get("elevenlabs_style", 0.3))) if settings.get("tts_engine") == "chatterbox": return ChatterboxVoice(float(settings.get("chatterbox_exaggeration", 0.6)), float(settings.get("chatterbox_cfg", 0.3)), str(settings.get("chatterbox_ref", "") or ""), on_status=on_status)