Motore voce ElevenLabs
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
1 parent
1400b5b104
commit
950e70b2c7
5 files changed
+98
-4
No files matched your search
@@ -7,7 +7,7 @@ L'interfaccia è riusata con licenza CC BY-NC 4.0 (uso personale, non commercial
|
|||||||
## Cosa fa
|
## Cosa fa
|
||||||
|
|
||||||
- **Conversazione a mani libere**: il microfono è sempre in ascolto; quando smetti di parlare, la frase viene trascrizione in locale (Whisper via MLX) e inviata al motore. In alternativa: premi-per-parlare, parola di attivazione "Hey Jarvis" (openwakeword, installabile dal pannello), o il campo di testo.
|
- **Conversazione a mani libere**: il microfono è sempre in ascolto; quando smetti di parlare, la frase viene trascrizione in locale (Whisper via MLX) e inviata al motore. In alternativa: premi-per-parlare, parola di attivazione "Hey Jarvis" (openwakeword, installabile dal pannello), o il campo di testo.
|
||||||
- **Voce**: tre motori. *Kokoro* in locale con le voci italiane Sara e Nicola, veloce e naturale (predefinito). *Chatterbox* (Resemble AI), espressivo, con controllo dell'enfasi e clonazione da un wav di riferimento: gira in un ambiente Python separato (`.venv-chatterbox`, creato con `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts "setuptools<81"`) come servizio locale avviato dall'app, scarica circa 2 GB al primo uso e su Apple Silicon genera a circa un terzo del tempo reale, quindi con pause di alcuni secondi tra le frasi. *Voce di sistema* di macOS. Le frasi vengono sintetizzate in anticipo mentre l'app pronuncia la precedente.
|
- **Voce**: tre motori. *Kokoro* in locale con le voci italiane Sara e Nicola, veloce e naturale (predefinito). *Chatterbox* (Resemble AI), espressivo, con controllo dell'enfasi e clonazione da un wav di riferimento: gira in un ambiente Python separato (`.venv-chatterbox`, creato con `uv venv .venv-chatterbox --python 3.12 && uv pip install --python .venv-chatterbox/bin/python chatterbox-tts "setuptools<81"`) come servizio locale avviato dall'app, scarica circa 2 GB al primo uso e su Apple Silicon genera a circa un terzo del tempo reale, quindi con pause di alcuni secondi tra le frasi. *ElevenLabs* (cloud, chiave API da elevenlabs.io): la più espressiva, prima parola in circa mezzo secondo, voci scelte dal tuo account con il pulsante «Carica voci», modello Flash (rapido) o Multilingual v2 (qualità), regolazioni di stabilità e stile; piano gratuito di 10 mila caratteri al mese. *Voce di sistema* di macOS. Le frasi vengono sintetizzate in anticipo mentre l'app pronuncia la precedente.
|
||||||
- **Volto**: la bocca segue lo spettro dell'audio (50 forme al secondo) fuso con il testo pronunciato; lo sguardo e le sopracciglia seguono lo stato (ascolta, pensa, parla, dorme).
|
- **Volto**: la bocca segue lo spettro dell'audio (50 forme al secondo) fuso con il testo pronunciato; lo sguardo e le sopracciglia seguono lo stato (ascolta, pensa, parla, dorme).
|
||||||
- **Tre motori**, selezionabili dal pulsante "Motore & Voce":
|
- **Tre motori**, selezionabili dal pulsante "Motore & Voce":
|
||||||
- *Claude (Anthropic)*: Claude Opus 5 via API, con ricerca web server-side. Serve una chiave, salvata nel portachiavi di macOS.
|
- *Claude (Anthropic)*: Claude Opus 5 via API, con ricerca web server-side. Serve una chiave, salvata nel portachiavi di macOS.
|
||||||
|
|||||||
+1
-1
@@ -234,7 +234,7 @@ class Assistant:
|
|||||||
self.settings.save()
|
self.settings.save()
|
||||||
else:
|
else:
|
||||||
eng = self.settings.get("tts_engine")
|
eng = self.settings.get("tts_engine")
|
||||||
label = "Sistema" if eng == "system" else ("Sara" if eng == "chatterbox" else
|
label = "Sistema" if eng == "system" else ("Sara" if eng in ("chatterbox", "elevenlabs") else
|
||||||
{"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara"))
|
{"if_sara": "Sara", "im_nicola": "Nicola"}.get(self.settings.get("kokoro_voice"), "Sara"))
|
||||||
try:
|
try:
|
||||||
if cm.get_voice() != label:
|
if cm.get_voice() != label:
|
||||||
|
|||||||
+5
-1
@@ -26,6 +26,10 @@ DEFAULTS: dict[str, Any] = {
|
|||||||
"kokoro_voice": "if_sara",
|
"kokoro_voice": "if_sara",
|
||||||
"chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica
|
"chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica
|
||||||
"chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo
|
"chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo
|
||||||
|
"elevenlabs_voice_id": "",
|
||||||
|
"elevenlabs_model": "eleven_flash_v2_5", # rapido; eleven_multilingual_v2 = qualità massima
|
||||||
|
"elevenlabs_stability": 0.45,
|
||||||
|
"elevenlabs_style": 0.3,
|
||||||
"chatterbox_ref": "preset:femminile", # preset:femminile | preset:maschile | percorso di un wav da imitare
|
"chatterbox_ref": "preset:femminile", # preset:femminile | preset:maschile | percorso di un wav da imitare
|
||||||
"system_voice": "", # "" = automatica
|
"system_voice": "", # "" = automatica
|
||||||
"stt_model": "mlx-community/whisper-small-mlx",
|
"stt_model": "mlx-community/whisper-small-mlx",
|
||||||
@@ -41,7 +45,7 @@ DEFAULTS: dict[str, Any] = {
|
|||||||
"monitor_escludi": "", # parole/frasi (separate da virgola) che escludono un messaggio dagli avvisi
|
"monitor_escludi": "", # parole/frasi (separate da virgola) che escludono un messaggio dagli avvisi
|
||||||
}
|
}
|
||||||
|
|
||||||
SECRET_KEYS = ("anthropic_api_key", "local_api_key", "telegram_api_hash")
|
SECRET_KEYS = ("anthropic_api_key", "local_api_key", "telegram_api_hash", "elevenlabs_api_key")
|
||||||
|
|
||||||
|
|
||||||
class Settings:
|
class Settings:
|
||||||
|
|||||||
@@ -99,7 +99,7 @@ class SettingsDialog(QDialog):
|
|||||||
self.stack.setCurrentIndex(self.provider.currentIndex())
|
self.stack.setCurrentIndex(self.provider.currentIndex())
|
||||||
|
|
||||||
form2 = QFormLayout()
|
form2 = QFormLayout()
|
||||||
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
|
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
|
||||||
form2.addRow("Motore voce", self.tts_engine)
|
form2.addRow("Motore voce", self.tts_engine)
|
||||||
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
|
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
|
||||||
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
|
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
|
||||||
@@ -115,6 +115,20 @@ class SettingsDialog(QDialog):
|
|||||||
row.addWidget(QLabel("voce")); row.addWidget(self.cb_voice)
|
row.addWidget(QLabel("voce")); row.addWidget(self.cb_voice)
|
||||||
self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1)
|
self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1)
|
||||||
form2.addRow("Voce Chatterbox", row)
|
form2.addRow("Voce Chatterbox", row)
|
||||||
|
row = QHBoxLayout()
|
||||||
|
self.el_key = QLineEdit(); self.el_key.setEchoMode(QLineEdit.EchoMode.Password)
|
||||||
|
self.el_key.setPlaceholderText("•••••• (salvata)" if s.get_secret("elevenlabs_api_key") else "chiave API da elevenlabs.io"); row.addWidget(self.el_key, 1)
|
||||||
|
self.el_voice = QComboBox(); self.el_voice.setMinimumWidth(220); row.addWidget(self.el_voice, 1)
|
||||||
|
b = QPushButton("Carica voci"); b.clicked.connect(self._el_voices); row.addWidget(b)
|
||||||
|
form2.addRow("ElevenLabs", row)
|
||||||
|
row = QHBoxLayout()
|
||||||
|
self.el_model = _combo([("eleven_flash_v2_5", "Flash v2.5 (rapido)"), ("eleven_multilingual_v2", "Multilingual v2 (qualità)"), ("eleven_v3", "v3 (più espressivo, sperimentale)")], s.get("elevenlabs_model")); row.addWidget(QLabel("modello")); row.addWidget(self.el_model)
|
||||||
|
self.el_stab = QDoubleSpinBox(); self.el_stab.setRange(0.0, 1.0); self.el_stab.setSingleStep(0.05); self.el_stab.setValue(float(s.get("elevenlabs_stability", 0.45))); row.addWidget(QLabel("stabilità")); row.addWidget(self.el_stab)
|
||||||
|
self.el_style = QDoubleSpinBox(); self.el_style.setRange(0.0, 1.0); self.el_style.setSingleStep(0.05); self.el_style.setValue(float(s.get("elevenlabs_style", 0.3))); row.addWidget(QLabel("stile")); row.addWidget(self.el_style)
|
||||||
|
row.addStretch()
|
||||||
|
form2.addRow("", row)
|
||||||
|
self._el_current = str(s.get("elevenlabs_voice_id") or "")
|
||||||
|
self.el_voice.addItem("(carica le voci con la chiave)" if not self._el_current else f"voce salvata: {self._el_current[:10]}…", self._el_current)
|
||||||
self.stt_model = _combo(STT_MODELS, s.get("stt_model")); form2.addRow("Riconoscimento vocale", self.stt_model)
|
self.stt_model = _combo(STT_MODELS, s.get("stt_model")); form2.addRow("Riconoscimento vocale", self.stt_model)
|
||||||
from .avatar3d import list_models
|
from .avatar3d import list_models
|
||||||
models = list_models()
|
models = list_models()
|
||||||
@@ -299,7 +313,30 @@ class SettingsDialog(QDialog):
|
|||||||
self._async.emit("wa", (f"Errore: {err}", None))
|
self._async.emit("wa", (f"Errore: {err}", None))
|
||||||
threading.Thread(target=work, daemon=True).start()
|
threading.Thread(target=work, daemon=True).start()
|
||||||
|
|
||||||
|
def _el_voices(self) -> None:
|
||||||
|
key = self.el_key.text().strip() or self.settings.get_secret("elevenlabs_api_key")
|
||||||
|
if not key:
|
||||||
|
self.el_voice.clear(); self.el_voice.addItem("inserisci prima la chiave API", ""); return
|
||||||
|
if self.el_key.text().strip():
|
||||||
|
self.settings.update({"elevenlabs_api_key": key})
|
||||||
|
def work():
|
||||||
|
try:
|
||||||
|
from .tts import ElevenLabsVoice
|
||||||
|
self._async.emit("el", ElevenLabsVoice.list_voices(key))
|
||||||
|
except Exception as err:
|
||||||
|
self._async.emit("el_error", str(err)[:120])
|
||||||
|
threading.Thread(target=work, daemon=True).start()
|
||||||
|
|
||||||
def _on_async(self, kind: str, payload) -> None:
|
def _on_async(self, kind: str, payload) -> None:
|
||||||
|
if kind == "el":
|
||||||
|
self.el_voice.clear()
|
||||||
|
for vid, label in payload:
|
||||||
|
self.el_voice.addItem(label, vid)
|
||||||
|
idx = self.el_voice.findData(self._el_current)
|
||||||
|
self.el_voice.setCurrentIndex(idx if idx >= 0 else 0)
|
||||||
|
return
|
||||||
|
if kind == "el_error":
|
||||||
|
self.el_voice.clear(); self.el_voice.addItem(f"errore: {payload}", ""); return
|
||||||
if kind == "wa":
|
if kind == "wa":
|
||||||
msg, png = payload
|
msg, png = payload
|
||||||
self.wa_hint.setText(str(msg))
|
self.wa_hint.setText(str(msg))
|
||||||
@@ -352,6 +389,8 @@ class SettingsDialog(QDialog):
|
|||||||
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
||||||
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
|
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
|
||||||
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
|
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
|
||||||
|
"elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(),
|
||||||
|
"elevenlabs_stability": float(self.el_stab.value()), "elevenlabs_style": float(self.el_style.value()),
|
||||||
"chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()),
|
"chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()),
|
||||||
"chatterbox_ref": (self.cb_ref.text().strip() if self.cb_voice.currentData() == "custom" else self.cb_voice.currentData()),
|
"chatterbox_ref": (self.cb_ref.text().strip() if self.cb_voice.currentData() == "custom" else self.cb_voice.currentData()),
|
||||||
"vad_threshold": float(self.vad.value()),
|
"vad_threshold": float(self.vad.value()),
|
||||||
@@ -371,6 +410,8 @@ class SettingsDialog(QDialog):
|
|||||||
values["anthropic_api_key"] = self.api_key.text().strip()
|
values["anthropic_api_key"] = self.api_key.text().strip()
|
||||||
if self.local_key.text().strip():
|
if self.local_key.text().strip():
|
||||||
values["local_api_key"] = self.local_key.text().strip()
|
values["local_api_key"] = self.local_key.text().strip()
|
||||||
|
if self.el_key.text().strip():
|
||||||
|
values["elevenlabs_api_key"] = self.el_key.text().strip()
|
||||||
self.settings.update(values)
|
self.settings.update(values)
|
||||||
self.accept()
|
self.accept()
|
||||||
if self.on_saved:
|
if self.on_saved:
|
||||||
|
|||||||
@@ -221,7 +221,56 @@ class ChatterboxVoice:
|
|||||||
cls._proc = None
|
cls._proc = None
|
||||||
|
|
||||||
|
|
||||||
|
class ElevenLabsVoice:
|
||||||
|
"""ElevenLabs (cloud): voci molto espressive, prima parola in circa mezzo secondo."""
|
||||||
|
|
||||||
|
API = "https://api.elevenlabs.io/v1"
|
||||||
|
|
||||||
|
def __init__(self, api_key: str, voice_id: str, model: str = "eleven_flash_v2_5", stability: float = 0.45, similarity: float = 0.8, style: float = 0.3) -> None:
|
||||||
|
self.api_key, self.voice_id, self.model = api_key, voice_id, model
|
||||||
|
self.stability, self.similarity, self.style = stability, similarity, style
|
||||||
|
self.voice = f"elevenlabs:{voice_id}:{model}:{stability}:{style}"
|
||||||
|
|
||||||
|
@classmethod
|
||||||
|
def list_voices(cls, api_key: str) -> list[tuple[str, str]]:
|
||||||
|
import requests
|
||||||
|
r = requests.get(f"{cls.API}/voices", headers={"xi-api-key": api_key}, timeout=15)
|
||||||
|
r.raise_for_status()
|
||||||
|
out = []
|
||||||
|
for v in r.json().get("voices", []):
|
||||||
|
labels = v.get("labels") or {}
|
||||||
|
desc = ", ".join(x for x in (labels.get("gender"), labels.get("accent"), labels.get("language"), v.get("category")) if x)
|
||||||
|
out.append((v["voice_id"], f"{v['name']}" + (f" ({desc})" if desc else "")))
|
||||||
|
return out
|
||||||
|
|
||||||
|
def load(self) -> None:
|
||||||
|
if not self.api_key:
|
||||||
|
raise RuntimeError("ElevenLabs: manca la chiave API (Motore e Voce).")
|
||||||
|
if not self.voice_id:
|
||||||
|
raise RuntimeError("ElevenLabs: scegli una voce in Motore e Voce (pulsante «Carica voci»).")
|
||||||
|
|
||||||
|
def synthesize(self, text: str) -> np.ndarray:
|
||||||
|
import requests
|
||||||
|
r = requests.post(f"{self.API}/text-to-speech/{self.voice_id}", params={"output_format": "pcm_24000"},
|
||||||
|
headers={"xi-api-key": self.api_key, "Content-Type": "application/json"},
|
||||||
|
json={"text": text, "model_id": self.model, "language_code": "it",
|
||||||
|
"voice_settings": {"stability": self.stability, "similarity_boost": self.similarity, "style": self.style, "use_speaker_boost": True}},
|
||||||
|
timeout=60)
|
||||||
|
if r.status_code != 200:
|
||||||
|
try:
|
||||||
|
detail = r.json().get("detail", {})
|
||||||
|
msg = detail.get("message") if isinstance(detail, dict) else str(detail)
|
||||||
|
except Exception:
|
||||||
|
msg = r.text[:200]
|
||||||
|
raise RuntimeError(f"ElevenLabs {r.status_code}: {msg}")
|
||||||
|
return np.frombuffer(r.content, dtype=np.int16).astype(np.float32) / 32768.0
|
||||||
|
|
||||||
|
|
||||||
def make_voice(settings, on_status=None):
|
def make_voice(settings, on_status=None):
|
||||||
|
if settings.get("tts_engine") == "elevenlabs":
|
||||||
|
return ElevenLabsVoice(settings.get_secret("elevenlabs_api_key"), str(settings.get("elevenlabs_voice_id") or ""),
|
||||||
|
str(settings.get("elevenlabs_model") or "eleven_flash_v2_5"),
|
||||||
|
float(settings.get("elevenlabs_stability", 0.45)), 0.8, float(settings.get("elevenlabs_style", 0.3)))
|
||||||
if settings.get("tts_engine") == "chatterbox":
|
if settings.get("tts_engine") == "chatterbox":
|
||||||
return ChatterboxVoice(float(settings.get("chatterbox_exaggeration", 0.6)), float(settings.get("chatterbox_cfg", 0.3)),
|
return ChatterboxVoice(float(settings.get("chatterbox_exaggeration", 0.6)), float(settings.get("chatterbox_cfg", 0.3)),
|
||||||
str(settings.get("chatterbox_ref", "") or ""), on_status=on_status)
|
str(settings.get("chatterbox_ref", "") or ""), on_status=on_status)
|
||||||
|
|||||||
Reference in new issue
Block a user