Voce Qwen3-TTS dentro LuZa (mlx-audio): clonazione da campione in data/voices/qwen_ref.wav, quasi in tempo reale
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
9cdd3663c4
commit
75781ea14a
3 files changed
+46
-2
No files matched your search
+1
-1
@@ -281,7 +281,7 @@ class RemoteServer:
|
||||
"mlx_model": [(m, m.split("/")[-1]) for m in cached_models()],
|
||||
"mlx_thinking": [("auto", "Automatico"), ("off", "Spento"), ("on", "Acceso")],
|
||||
"claudecode_model": [("sonnet", "Sonnet"), ("opus", "Opus"), ("haiku", "Haiku")],
|
||||
"tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")],
|
||||
"tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("qwen", "Qwen3-TTS (clonata, in LuZa)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")],
|
||||
"kokoro_voice": list(KOKORO_VOICES.items()),
|
||||
"immagini_famiglia": [("z-image-turbo", "Z-Image Turbo"), ("schnell", "FLUX schnell"), ("dev", "FLUX dev"), ("qwen", "Qwen-Image")],
|
||||
}
|
||||
|
||||
@@ -139,7 +139,7 @@ class SettingsDialog(QDialog):
|
||||
self.stack.setCurrentIndex(self.provider.currentIndex())
|
||||
|
||||
form2 = QFormLayout()
|
||||
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
|
||||
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("qwen", "Qwen3-TTS dentro LuZa: voce clonata, quasi in tempo reale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
|
||||
form2.addRow("Motore voce", self.tts_engine)
|
||||
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
|
||||
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
|
||||
|
||||
@@ -347,7 +347,51 @@ class VoiceboxVoice:
|
||||
return data
|
||||
|
||||
|
||||
class QwenVoice:
|
||||
"""Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano
|
||||
sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread."""
|
||||
|
||||
MODEL = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16"
|
||||
|
||||
def __init__(self, ref_audio: str, ref_text: str, on_status=None) -> None:
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
self.ref_audio, self.ref_text = ref_audio, ref_text
|
||||
self.voice = f"qwen:{ref_audio}"
|
||||
self._on_status = on_status or (lambda m: None)
|
||||
self._pool = ThreadPoolExecutor(max_workers=1, thread_name_prefix="qwen-tts")
|
||||
self._model = None
|
||||
|
||||
def _load(self) -> None:
|
||||
if self._model is None:
|
||||
from mlx_audio.tts.utils import load_model
|
||||
self._model = load_model(self.MODEL)
|
||||
|
||||
def load(self) -> None:
|
||||
from pathlib import Path as _P
|
||||
if not _P(self.ref_audio).is_file() or not self.ref_text.strip():
|
||||
raise RuntimeError("Qwen3-TTS: manca il campione della voce (data/voices/qwen_ref.wav e .txt).")
|
||||
self._on_status("Carico Qwen3-TTS…")
|
||||
try:
|
||||
self._pool.submit(self._load).result()
|
||||
finally:
|
||||
self._on_status(None)
|
||||
|
||||
def _gen(self, text: str) -> np.ndarray:
|
||||
self._load()
|
||||
parts = [np.array(r.audio, dtype=np.float32) for r in self._model.generate(
|
||||
text, ref_audio=self.ref_audio, ref_text=self.ref_text, lang_code="italian")]
|
||||
return np.concatenate(parts) if parts else np.zeros(0, dtype=np.float32)
|
||||
|
||||
def synthesize(self, text: str) -> np.ndarray:
|
||||
return self._pool.submit(self._gen, clean_for_speech(text)).result()
|
||||
|
||||
|
||||
def make_voice(settings, on_status=None):
|
||||
if settings.get("tts_engine") == "qwen":
|
||||
from pathlib import Path as _P
|
||||
ref = _P(str(settings.get("qwen_ref") or "")) if settings.get("qwen_ref") else _P(__file__).resolve().parent.parent / "data" / "voices" / "qwen_ref.wav"
|
||||
txt = ref.with_suffix(".txt")
|
||||
return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status)
|
||||
if settings.get("tts_engine") == "voicebox":
|
||||
return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"),
|
||||
str(settings.get("voicebox_instruct") or ""), on_status=on_status)
|
||||
|
||||
Reference in new issue
Block a user