Voce Qwen3-TTS dentro LuZa (mlx-audio): clonazione da campione in data/voices/qwen_ref.wav, quasi in tempo reale

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
lucianoandClaude Opus 5.5 committed 2026-10-06 14:35:05 +02:00
1 parent 9cdd3663c4
commit 75781ea14a
3 files changed
+46 -2

No files matched your search

+1 -1
View File
@@ -281,7 +281,7 @@ class RemoteServer:
"mlx_model": [(m, m.split("/")[-1]) for m in cached_models()],
"mlx_thinking": [("auto", "Automatico"), ("off", "Spento"), ("on", "Acceso")],
"claudecode_model": [("sonnet", "Sonnet"), ("opus", "Opus"), ("haiku", "Haiku")],
"tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")],
"tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("qwen", "Qwen3-TTS (clonata, in LuZa)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")],
"kokoro_voice": list(KOKORO_VOICES.items()),
"immagini_famiglia": [("z-image-turbo", "Z-Image Turbo"), ("schnell", "FLUX schnell"), ("dev", "FLUX dev"), ("qwen", "Qwen-Image")],
}
+1 -1
View File
@@ -139,7 +139,7 @@ class SettingsDialog(QDialog):
self.stack.setCurrentIndex(self.provider.currentIndex())
form2 = QFormLayout()
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("qwen", "Qwen3-TTS dentro LuZa: voce clonata, quasi in tempo reale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
form2.addRow("Motore voce", self.tts_engine)
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
+44
View File
@@ -347,7 +347,51 @@ class VoiceboxVoice:
return data
class QwenVoice:
"""Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano
sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread."""
MODEL = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16"
def __init__(self, ref_audio: str, ref_text: str, on_status=None) -> None:
from concurrent.futures import ThreadPoolExecutor
self.ref_audio, self.ref_text = ref_audio, ref_text
self.voice = f"qwen:{ref_audio}"
self._on_status = on_status or (lambda m: None)
self._pool = ThreadPoolExecutor(max_workers=1, thread_name_prefix="qwen-tts")
self._model = None
def _load(self) -> None:
if self._model is None:
from mlx_audio.tts.utils import load_model
self._model = load_model(self.MODEL)
def load(self) -> None:
from pathlib import Path as _P
if not _P(self.ref_audio).is_file() or not self.ref_text.strip():
raise RuntimeError("Qwen3-TTS: manca il campione della voce (data/voices/qwen_ref.wav e .txt).")
self._on_status("Carico Qwen3-TTS…")
try:
self._pool.submit(self._load).result()
finally:
self._on_status(None)
def _gen(self, text: str) -> np.ndarray:
self._load()
parts = [np.array(r.audio, dtype=np.float32) for r in self._model.generate(
text, ref_audio=self.ref_audio, ref_text=self.ref_text, lang_code="italian")]
return np.concatenate(parts) if parts else np.zeros(0, dtype=np.float32)
def synthesize(self, text: str) -> np.ndarray:
return self._pool.submit(self._gen, clean_for_speech(text)).result()
def make_voice(settings, on_status=None):
if settings.get("tts_engine") == "qwen":
from pathlib import Path as _P
ref = _P(str(settings.get("qwen_ref") or "")) if settings.get("qwen_ref") else _P(__file__).resolve().parent.parent / "data" / "voices" / "qwen_ref.wav"
txt = ref.with_suffix(".txt")
return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status)
if settings.get("tts_engine") == "voicebox":
return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"),
str(settings.get("voicebox_instruct") or ""), on_status=on_status)