Motore voce ElevenLabs

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
lucianoandClaude Fable 5.1 committed 2026-09-24 09:44:56 +02:00
1 parent 1400b5b104
commit 950e70b2c7
5 files changed
+98 -4

No files matched your search

+1 -1
View File
@@ -7,7 +7,7 @@ L'interfaccia è riusata con licenza CC BY-NC 4.0 (uso personale, non commercial
## Cosa fa
- **Conversazione a mani libere**: il microfono è sempre in ascolto; quando smetti di parlare, la frase viene trascrizione in locale (Whisper via MLX) e inviata al motore. In alternativa: premi-per-parlare, parola di attivazione "Hey Jarvis" (openwakeword, installabile dal pannello), o il campo di testo.
- **Voce**: tre motori. *Kokoro* in locale con le voci italiane Sara e Nicola, veloce e naturale (predefinito). *Chatterbox* (Resemble AI), espressivo, con controllo dell'enfasi e clonazione da un wav di riferimento: gira in un ambiente Python separato (`.venv-chatterbox`, creato con `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts "setuptools<81"`) come servizio locale avviato dall'app, scarica circa 2 GB al primo uso e su Apple Silicon genera a circa un terzo del tempo reale, quindi con pause di alcuni secondi tra le frasi. *Voce di sistema* di macOS. Le frasi vengono sintetizzate in anticipo mentre l'app pronuncia la precedente.
- **Voce**: tre motori. *Kokoro* in locale con le voci italiane Sara e Nicola, veloce e naturale (predefinito). *Chatterbox* (Resemble AI), espressivo, con controllo dell'enfasi e clonazione da un wav di riferimento: gira in un ambiente Python separato (`.venv-chatterbox`, creato con `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts "setuptools<81"`) come servizio locale avviato dall'app, scarica circa 2 GB al primo uso e su Apple Silicon genera a circa un terzo del tempo reale, quindi con pause di alcuni secondi tra le frasi. *ElevenLabs* (cloud, chiave API da elevenlabs.io): la più espressiva, prima parola in circa mezzo secondo, voci scelte dal tuo account con il pulsante «Carica voci», modello Flash (rapido) o Multilingual v2 (qualità), regolazioni di stabilità e stile; piano gratuito di 10 mila caratteri al mese. *Voce di sistema* di macOS. Le frasi vengono sintetizzate in anticipo mentre l'app pronuncia la precedente.
- **Volto**: la bocca segue lo spettro dell'audio (50 forme al secondo) fuso con il testo pronunciato; lo sguardo e le sopracciglia seguono lo stato (ascolta, pensa, parla, dorme).
- **Tre motori**, selezionabili dal pulsante "Motore & Voce":
- *Claude (Anthropic)*: Claude Opus 5 via API, con ricerca web server-side. Serve una chiave, salvata nel portachiavi di macOS.
+1 -1
View File
@@ -234,7 +234,7 @@ class Assistant:
self.settings.save()
else:
eng = self.settings.get("tts_engine")
label = "Sistema" if eng == "system" else ("Sara" if eng == "chatterbox" else
label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs") else
{"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara"))
try:
if cm.get_voice() != label:
+5 -1
View File
@@ -26,6 +26,10 @@ DEFAULTS: dict[str, Any] = {
"kokoro_voice": "if_sara",
"chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica
"chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo
"elevenlabs_voice_id": "",
"elevenlabs_model": "eleven_flash_v2_5", # rapido; eleven_multilingual_v2 = qualità massima
"elevenlabs_stability": 0.45,
"elevenlabs_style": 0.3,
"chatterbox_ref": "preset:femminile", # preset:femminile | preset:maschile | percorso di un wav da imitare
"system_voice": "", # "" = automatica
"stt_model": "mlx-community/whisper-small-mlx",
@@ -41,7 +45,7 @@ DEFAULTS: dict[str, Any] = {
"monitor_escludi": "", # parole/frasi (separate da virgola) che escludono un messaggio dagli avvisi
}
SECRET_KEYS = ("anthropic_api_key", "local_api_key", "telegram_api_hash")
SECRET_KEYS = ("anthropic_api_key", "local_api_key", "telegram_api_hash", "elevenlabs_api_key")
class Settings:
+42 -1
View File
@@ -99,7 +99,7 @@ class SettingsDialog(QDialog):
self.stack.setCurrentIndex(self.provider.currentIndex())
form2 = QFormLayout()
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
form2.addRow("Motore voce", self.tts_engine)
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
@@ -115,6 +115,20 @@ class SettingsDialog(QDialog):
row.addWidget(QLabel("voce")); row.addWidget(self.cb_voice)
self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1)
form2.addRow("Voce Chatterbox", row)
row = QHBoxLayout()
self.el_key = QLineEdit(); self.el_key.setEchoMode(QLineEdit.EchoMode.Password)
self.el_key.setPlaceholderText("•••••• (salvata)" if s.get_secret("elevenlabs_api_key") else "chiave API da elevenlabs.io"); row.addWidget(self.el_key, 1)
self.el_voice = QComboBox(); self.el_voice.setMinimumWidth(220); row.addWidget(self.el_voice, 1)
b = QPushButton("Carica voci"); b.clicked.connect(self._el_voices); row.addWidget(b)
form2.addRow("ElevenLabs", row)
row = QHBoxLayout()
self.el_model = _combo([("eleven_flash_v2_5", "Flash v2.5 (rapido)"), ("eleven_multilingual_v2", "Multilingual v2 (qualità)"), ("eleven_v3", "v3 (più espressivo, sperimentale)")], s.get("elevenlabs_model")); row.addWidget(QLabel("modello")); row.addWidget(self.el_model)
self.el_stab = QDoubleSpinBox(); self.el_stab.setRange(0.0, 1.0); self.el_stab.setSingleStep(0.05); self.el_stab.setValue(float(s.get("elevenlabs_stability", 0.45))); row.addWidget(QLabel("stabilità")); row.addWidget(self.el_stab)
self.el_style = QDoubleSpinBox(); self.el_style.setRange(0.0, 1.0); self.el_style.setSingleStep(0.05); self.el_style.setValue(float(s.get("elevenlabs_style", 0.3))); row.addWidget(QLabel("stile")); row.addWidget(self.el_style)
row.addStretch()
form2.addRow("", row)
self._el_current = str(s.get("elevenlabs_voice_id") or "")
self.el_voice.addItem("(carica le voci con la chiave)" if not self._el_current else f"voce salvata: {self._el_current[:10]}…", self._el_current)
self.stt_model = _combo(STT_MODELS, s.get("stt_model")); form2.addRow("Riconoscimento vocale", self.stt_model)
from .avatar3d import list_models
models = list_models()
@@ -299,7 +313,30 @@ class SettingsDialog(QDialog):
self._async.emit("wa", (f"Errore: {err}", None))
threading.Thread(target=work, daemon=True).start()
def _el_voices(self) -> None:
key = self.el_key.text().strip() or self.settings.get_secret("elevenlabs_api_key")
if not key:
self.el_voice.clear(); self.el_voice.addItem("inserisci prima la chiave API", ""); return
if self.el_key.text().strip():
self.settings.update({"elevenlabs_api_key": key})
def work():
try:
from .tts import ElevenLabsVoice
self._async.emit("el", ElevenLabsVoice.list_voices(key))
except Exception as err:
self._async.emit("el_error", str(err)[:120])
threading.Thread(target=work, daemon=True).start()
def _on_async(self, kind: str, payload) -> None:
if kind == "el":
self.el_voice.clear()
for vid, label in payload:
self.el_voice.addItem(label, vid)
idx = self.el_voice.findData(self._el_current)
self.el_voice.setCurrentIndex(idx if idx >= 0 else 0)
return
if kind == "el_error":
self.el_voice.clear(); self.el_voice.addItem(f"errore: {payload}", ""); return
if kind == "wa":
msg, png = payload
self.wa_hint.setText(str(msg))
@@ -352,6 +389,8 @@ class SettingsDialog(QDialog):
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
"elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(),
"elevenlabs_stability": float(self.el_stab.value()), "elevenlabs_style": float(self.el_style.value()),
"chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()),
"chatterbox_ref": (self.cb_ref.text().strip() if self.cb_voice.currentData() == "custom" else self.cb_voice.currentData()),
"vad_threshold": float(self.vad.value()),
@@ -371,6 +410,8 @@ class SettingsDialog(QDialog):
values["anthropic_api_key"] = self.api_key.text().strip()
if self.local_key.text().strip():
values["local_api_key"] = self.local_key.text().strip()
if self.el_key.text().strip():
values["elevenlabs_api_key"] = self.el_key.text().strip()
self.settings.update(values)
self.accept()
if self.on_saved:
+49
View File
@@ -221,7 +221,56 @@ class ChatterboxVoice:
cls._proc = None
class ElevenLabsVoice:
"""ElevenLabs (cloud): voci molto espressive, prima parola in circa mezzo secondo."""
API = "https://api.elevenlabs.io/v1"
def __init__(self, api_key: str, voice_id: str, model: str = "eleven_flash_v2_5", stability: float = 0.45, similarity: float = 0.8, style: float = 0.3) -> None:
self.api_key, self.voice_id, self.model = api_key, voice_id, model
self.stability, self.similarity, self.style = stability, similarity, style
self.voice = f"elevenlabs:{voice_id}:{model}:{stability}:{style}"
@classmethod
def list_voices(cls, api_key: str) -> list[tuple[str, str]]:
import requests
r = requests.get(f"{cls.API}/voices", headers={"xi-api-key": api_key}, timeout=15)
r.raise_for_status()
out = []
for v in r.json().get("voices", []):
labels = v.get("labels") or {}
desc = ", ".join(x for x in (labels.get("gender"), labels.get("accent"), labels.get("language"), v.get("category")) if x)
out.append((v["voice_id"], f"{v['name']}" + (f" ({desc})" if desc else "")))
return out
def load(self) -> None:
if not self.api_key:
raise RuntimeError("ElevenLabs: manca la chiave API (Motore e Voce).")
if not self.voice_id:
raise RuntimeError("ElevenLabs: scegli una voce in Motore e Voce (pulsante «Carica voci»).")
def synthesize(self, text: str) -> np.ndarray:
import requests
r = requests.post(f"{self.API}/text-to-speech/{self.voice_id}", params={"output_format": "pcm_24000"},
headers={"xi-api-key": self.api_key, "Content-Type": "application/json"},
json={"text": text, "model_id": self.model, "language_code": "it",
"voice_settings": {"stability": self.stability, "similarity_boost": self.similarity, "style": self.style, "use_speaker_boost": True}},
timeout=60)
if r.status_code != 200:
try:
detail = r.json().get("detail", {})
msg = detail.get("message") if isinstance(detail, dict) else str(detail)
except Exception:
msg = r.text[:200]
raise RuntimeError(f"ElevenLabs {r.status_code}: {msg}")
return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0
def make_voice(settings, on_status=None):
if settings.get("tts_engine") == "elevenlabs":
return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""),
str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"),
float(settings.get("elevenlabs_stability", 0.45)), 0.8, float(settings.get("elevenlabs_style", 0.3)))
if settings.get("tts_engine") == "chatterbox":
return ChatterboxVoice(float(settings.get("chatterbox_exaggeration", 0.6)), float(settings.get("chatterbox_cfg", 0.3)),
str(settings.get("chatterbox_ref", "") or ""), on_status=on_status)