Chatterbox: preset di voce femminile e maschile (riferimenti generati con Kokoro)
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
1 parent
c33b72edfd
commit
1400b5b104
4 files changed
+34
-4
No files matched your search
+1
-1
@@ -26,7 +26,7 @@ DEFAULTS: dict[str, Any] = {
|
|||||||
"kokoro_voice": "if_sara",
|
"kokoro_voice": "if_sara",
|
||||||
"chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica
|
"chatterbox_exaggeration": 0.6, # 0.3 sobria … 0.9 molto enfatica
|
||||||
"chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo
|
"chatterbox_cfg": 0.3, # più basso = più veloce e meno aderente al testo
|
||||||
"chatterbox_ref": "", # wav di riferimento per clonare una voce (facoltativo)
|
"chatterbox_ref": "preset:femminile", # preset:femminile | preset:maschile | percorso di un wav da imitare
|
||||||
"system_voice": "", # "" = automatica
|
"system_voice": "", # "" = automatica
|
||||||
"stt_model": "mlx-community/whisper-small-mlx",
|
"stt_model": "mlx-community/whisper-small-mlx",
|
||||||
"vad_threshold": 0.08,
|
"vad_threshold": 0.08,
|
||||||
|
|||||||
@@ -107,8 +107,14 @@ class SettingsDialog(QDialog):
|
|||||||
row = QHBoxLayout()
|
row = QHBoxLayout()
|
||||||
self.cb_exag = QDoubleSpinBox(); self.cb_exag.setRange(0.2, 1.0); self.cb_exag.setSingleStep(0.1); self.cb_exag.setValue(float(s.get("chatterbox_exaggeration", 0.6))); row.addWidget(QLabel("enfasi")); row.addWidget(self.cb_exag)
|
self.cb_exag = QDoubleSpinBox(); self.cb_exag.setRange(0.2, 1.0); self.cb_exag.setSingleStep(0.1); self.cb_exag.setValue(float(s.get("chatterbox_exaggeration", 0.6))); row.addWidget(QLabel("enfasi")); row.addWidget(self.cb_exag)
|
||||||
self.cb_cfg = QDoubleSpinBox(); self.cb_cfg.setRange(0.0, 1.0); self.cb_cfg.setSingleStep(0.1); self.cb_cfg.setValue(float(s.get("chatterbox_cfg", 0.3))); row.addWidget(QLabel("aderenza (0 = più veloce)")); row.addWidget(self.cb_cfg)
|
self.cb_cfg = QDoubleSpinBox(); self.cb_cfg.setRange(0.0, 1.0); self.cb_cfg.setSingleStep(0.1); self.cb_cfg.setValue(float(s.get("chatterbox_cfg", 0.3))); row.addWidget(QLabel("aderenza (0 = più veloce)")); row.addWidget(self.cb_cfg)
|
||||||
self.cb_ref = QLineEdit(s.get("chatterbox_ref") or ""); self.cb_ref.setPlaceholderText("wav di riferimento per clonare una voce (facoltativo)"); row.addWidget(self.cb_ref, 1)
|
|
||||||
form2.addRow("Chatterbox", row)
|
form2.addRow("Chatterbox", row)
|
||||||
|
row = QHBoxLayout()
|
||||||
|
cur_ref = s.get("chatterbox_ref") or ""
|
||||||
|
self.cb_voice = _combo([("", "Predefinita del modello"), ("preset:femminile", "Femminile (timbro di Sara)"), ("preset:maschile", "Maschile (timbro di Nicola)"), ("custom", "Personalizzata: file wav")],
|
||||||
|
cur_ref if cur_ref in ("", "preset:femminile", "preset:maschile") else "custom")
|
||||||
|
row.addWidget(QLabel("voce")); row.addWidget(self.cb_voice)
|
||||||
|
self.cb_ref = QLineEdit(cur_ref if cur_ref.startswith("/") or cur_ref.startswith("~") else ""); self.cb_ref.setPlaceholderText("percorso di un wav di 10 secondi con la voce da imitare"); row.addWidget(self.cb_ref, 1)
|
||||||
|
form2.addRow("Voce Chatterbox", row)
|
||||||
self.stt_model = _combo(STT_MODELS, s.get("stt_model")); form2.addRow("Riconoscimento vocale", self.stt_model)
|
self.stt_model = _combo(STT_MODELS, s.get("stt_model")); form2.addRow("Riconoscimento vocale", self.stt_model)
|
||||||
from .avatar3d import list_models
|
from .avatar3d import list_models
|
||||||
models = list_models()
|
models = list_models()
|
||||||
@@ -346,7 +352,8 @@ class SettingsDialog(QDialog):
|
|||||||
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
||||||
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
|
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
|
||||||
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
|
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
|
||||||
"chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()), "chatterbox_ref": self.cb_ref.text().strip(),
|
"chatterbox_exaggeration": float(self.cb_exag.value()), "chatterbox_cfg": float(self.cb_cfg.value()),
|
||||||
|
"chatterbox_ref": (self.cb_ref.text().strip() if self.cb_voice.currentData() == "custom" else self.cb_voice.currentData()),
|
||||||
"vad_threshold": float(self.vad.value()),
|
"vad_threshold": float(self.vad.value()),
|
||||||
"avatar_model": self.avatar_model.currentData() or "",
|
"avatar_model": self.avatar_model.currentData() or "",
|
||||||
"telegram_api_id": self.tg_id.text().strip(),
|
"telegram_api_id": self.tg_id.text().strip(),
|
||||||
|
|||||||
+24
-1
@@ -179,10 +179,33 @@ class ChatterboxVoice:
|
|||||||
finally:
|
finally:
|
||||||
self._on_status(None)
|
self._on_status(None)
|
||||||
|
|
||||||
|
REFERENCE_TEXT = ("Ciao, sono la voce del tuo assistente. Oggi ti racconto come è andata la giornata, con calma e con un po' di allegria. "
|
||||||
|
"Se hai bisogno di qualcosa, basta chiedere: sono qui per aiutarti, e mi fa piacere farlo.")
|
||||||
|
PRESETS = {"preset:femminile": "if_sara", "preset:maschile": "im_nicola"}
|
||||||
|
|
||||||
|
def _reference_path(self) -> str | None:
|
||||||
|
"""File wav di riferimento: percorso dell'utente, oppure preset generato con Kokoro (una volta)."""
|
||||||
|
ref = (self.ref_audio or "").strip()
|
||||||
|
if not ref:
|
||||||
|
return None
|
||||||
|
if ref in self.PRESETS:
|
||||||
|
from pathlib import Path as _P
|
||||||
|
import soundfile as sf
|
||||||
|
out = _P(__file__).resolve().parent.parent / "data" / "voices" / f"{ref.split(':')[1]}.wav"
|
||||||
|
if not out.exists():
|
||||||
|
out.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
self._on_status("Preparo il campione di voce per Chatterbox…")
|
||||||
|
k = KokoroVoice(self.PRESETS[ref], speed=0.95)
|
||||||
|
k.load()
|
||||||
|
sf.write(str(out), k.synthesize(self.REFERENCE_TEXT), OUT_RATE)
|
||||||
|
self._on_status(None)
|
||||||
|
return str(out)
|
||||||
|
return ref
|
||||||
|
|
||||||
def synthesize(self, text: str) -> np.ndarray:
|
def synthesize(self, text: str) -> np.ndarray:
|
||||||
import requests
|
import requests
|
||||||
r = requests.post(f"http://127.0.0.1:{self.PORT}/tts", json={"text": text, "language": "it", "exaggeration": self.exaggeration,
|
r = requests.post(f"http://127.0.0.1:{self.PORT}/tts", json={"text": text, "language": "it", "exaggeration": self.exaggeration,
|
||||||
"cfg": self.cfg, "ref": self.ref_audio or None}, timeout=600)
|
"cfg": self.cfg, "ref": self._reference_path()}, timeout=600)
|
||||||
if r.status_code != 200:
|
if r.status_code != 200:
|
||||||
raise RuntimeError(r.json().get("error", r.text))
|
raise RuntimeError(r.json().get("error", r.text))
|
||||||
sr = int(r.headers.get("X-Sample-Rate", "24000"))
|
sr = int(r.headers.get("X-Sample-Rate", "24000"))
|
||||||
|
|||||||
Binary file not shown.
Reference in new issue
Block a user