Voce Qwen3-TTS in streaming (parte in ~0,5 s) con il modello Base a 8 bit
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
e2dbd31388
commit
54895586c8
3 files changed
+52
-6
No files matched your search
@@ -429,6 +429,23 @@ class Assistant:
|
|||||||
if item is END:
|
if item is END:
|
||||||
self._audio_q.put((turn, END, None))
|
self._audio_q.put((turn, END, None))
|
||||||
continue
|
continue
|
||||||
|
if hasattr(self.voice, "stream"): # voce in streaming: i pezzi partono mentre la frase è ancora in generazione
|
||||||
|
try:
|
||||||
|
first = True
|
||||||
|
for chunk in self.voice.stream(item, cancel=lambda t=turn: t != self._turn):
|
||||||
|
if turn != self._turn:
|
||||||
|
break
|
||||||
|
remote = getattr(self, "remote", None)
|
||||||
|
if remote is not None and remote.has_clients():
|
||||||
|
try:
|
||||||
|
remote.on_audio(item if first else "", chunk)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
self._audio_q.put((turn, item if first else "", chunk))
|
||||||
|
first = False
|
||||||
|
except Exception as err:
|
||||||
|
self.ui.write_log(f"ERR: Sintesi vocale — {err}")
|
||||||
|
continue
|
||||||
try:
|
try:
|
||||||
audio = self.voice.synthesize(item)
|
audio = self.voice.synthesize(item)
|
||||||
except Exception as err:
|
except Exception as err:
|
||||||
|
|||||||
+2
-1
@@ -57,7 +57,8 @@ DEFAULTS: dict[str, Any] = {
|
|||||||
"abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria
|
"abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria
|
||||||
"umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente
|
"umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente
|
||||||
"io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate
|
"io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate
|
||||||
"qwen_voce": "luza_voce", # voce clonata (file in data/voices) per il motore Qwen3-TTS
|
"qwen_voce": "luza_voce",
|
||||||
|
"qwen_modello": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-8bit", # oppure ...-Base-bf16 (più lento, qualità piena) # voce clonata (file in data/voices) per il motore Qwen3-TTS
|
||||||
"ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth
|
"ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth
|
||||||
"ipixel_address": "", # identificativo BLE (rilevato da solo)
|
"ipixel_address": "", # identificativo BLE (rilevato da solo)
|
||||||
"ipixel_luminosita": 40,
|
"ipixel_luminosita": 40,
|
||||||
|
|||||||
+33
-5
@@ -374,12 +374,13 @@ class QwenVoice:
|
|||||||
"""Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano
|
"""Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano
|
||||||
sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread."""
|
sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread."""
|
||||||
|
|
||||||
MODEL = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-bf16"
|
MODEL = "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-8bit" # 8 bit: ~1,7× più veloce del bf16, qualità quasi uguale
|
||||||
|
|
||||||
def __init__(self, ref_audio: str, ref_text: str, on_status=None) -> None:
|
def __init__(self, ref_audio: str, ref_text: str, on_status=None, model: str = "") -> None:
|
||||||
from concurrent.futures import ThreadPoolExecutor
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
self.ref_audio, self.ref_text = ref_audio, ref_text
|
self.ref_audio, self.ref_text = ref_audio, ref_text
|
||||||
self.voice = f"qwen:{ref_audio}"
|
self.model_name = model or self.MODEL
|
||||||
|
self.voice = f"qwen:{self.model_name}:{ref_audio}"
|
||||||
self._on_status = on_status or (lambda m: None)
|
self._on_status = on_status or (lambda m: None)
|
||||||
self._pool = ThreadPoolExecutor(max_workers=1, thread_name_prefix="qwen-tts")
|
self._pool = ThreadPoolExecutor(max_workers=1, thread_name_prefix="qwen-tts")
|
||||||
self._model = None
|
self._model = None
|
||||||
@@ -387,7 +388,7 @@ class QwenVoice:
|
|||||||
def _load(self) -> None:
|
def _load(self) -> None:
|
||||||
if self._model is None:
|
if self._model is None:
|
||||||
from mlx_audio.tts.utils import load_model
|
from mlx_audio.tts.utils import load_model
|
||||||
self._model = load_model(self.MODEL)
|
self._model = load_model(self.model_name)
|
||||||
|
|
||||||
def load(self) -> None:
|
def load(self) -> None:
|
||||||
from pathlib import Path as _P
|
from pathlib import Path as _P
|
||||||
@@ -408,6 +409,32 @@ class QwenVoice:
|
|||||||
def synthesize(self, text: str) -> np.ndarray:
|
def synthesize(self, text: str) -> np.ndarray:
|
||||||
return self._pool.submit(self._gen, clean_for_speech(text)).result()
|
return self._pool.submit(self._gen, clean_for_speech(text)).result()
|
||||||
|
|
||||||
|
def stream(self, text: str, cancel=None):
|
||||||
|
"""Generatore di pezzi d'audio (~0,6 s) mentre la frase viene ancora generata: la voce parte dopo circa un secondo."""
|
||||||
|
import queue
|
||||||
|
q: queue.Queue = queue.Queue()
|
||||||
|
|
||||||
|
def work():
|
||||||
|
try:
|
||||||
|
self._load()
|
||||||
|
for r in self._model.generate(clean_for_speech(text), ref_audio=self.ref_audio, ref_text=self.ref_text,
|
||||||
|
lang_code="italian", stream=True, streaming_interval=0.6):
|
||||||
|
if cancel is not None and cancel():
|
||||||
|
break
|
||||||
|
q.put(np.array(r.audio, dtype=np.float32))
|
||||||
|
except Exception as err:
|
||||||
|
q.put(err)
|
||||||
|
q.put(None)
|
||||||
|
|
||||||
|
self._pool.submit(work)
|
||||||
|
while True:
|
||||||
|
item = q.get()
|
||||||
|
if item is None:
|
||||||
|
return
|
||||||
|
if isinstance(item, Exception):
|
||||||
|
raise item
|
||||||
|
yield item
|
||||||
|
|
||||||
|
|
||||||
def make_voice(settings, on_status=None):
|
def make_voice(settings, on_status=None):
|
||||||
if settings.get("tts_engine") == "qwen":
|
if settings.get("tts_engine") == "qwen":
|
||||||
@@ -416,7 +443,8 @@ def make_voice(settings, on_status=None):
|
|||||||
if not ref.exists() and voices:
|
if not ref.exists() and voices:
|
||||||
ref = VOICES_DIR / f"{voices[0]}.wav"
|
ref = VOICES_DIR / f"{voices[0]}.wav"
|
||||||
txt = ref.with_suffix(".txt")
|
txt = ref.with_suffix(".txt")
|
||||||
return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status)
|
return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status,
|
||||||
|
model=str(settings.get("qwen_modello") or ""))
|
||||||
if settings.get("tts_engine") == "voicebox":
|
if settings.get("tts_engine") == "voicebox":
|
||||||
return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"),
|
return VoiceboxVoice(str(settings.get("voicebox_profile_id") or ""), str(settings.get("voicebox_engine") or "qwen"),
|
||||||
str(settings.get("voicebox_instruct") or ""), on_status=on_status)
|
str(settings.get("voicebox_instruct") or ""), on_status=on_status)
|
||||||
|
|||||||
Reference in new issue
Block a user