From e2dbd31388f094c86b236e05a22637fcf1d1e569 Mon Sep 17 00:00:00 2001 From: luciano Date: Tue, 6 Oct 2026 14:49:39 +0200 Subject: [PATCH] Voci clonate Qwen3-TTS selezionabili (Motore e Voce e telefono), importazione di un campione con trascrizione automatica Co-Authored-By: Claude Opus 5.5 --- avatar/remote.py | 3 ++- avatar/settings.py | 1 + avatar/settings_dialog.py | 42 ++++++++++++++++++++++++++++++++++++++- avatar/tts.py | 29 +++++++++++++++++++++++++-- remote/app.js | 4 ++-- 5 files changed, 73 insertions(+), 6 deletions(-) diff --git a/avatar/remote.py b/avatar/remote.py index 34fe0d1..a226380 100644 --- a/avatar/remote.py +++ b/avatar/remote.py @@ -270,7 +270,7 @@ class RemoteServer: # ── impostazioni dal telefono: motore, modello, ragionamento, voce ───── SETTING_KEYS = ("provider", "effort", "mlx_model", "mlx_thinking", "claudecode_model", "local_model", - "tts_engine", "kokoro_voice", "voicebox_profile_id", "immagini_famiglia", "immagini_modello") + "tts_engine", "kokoro_voice", "voicebox_profile_id", "qwen_voce", "immagini_famiglia", "immagini_modello") def _options(self) -> dict: from avatar.tts import KOKORO_VOICES @@ -283,6 +283,7 @@ class RemoteServer: "claudecode_model": [("sonnet", "Sonnet"), ("opus", "Opus"), ("haiku", "Haiku")], "tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("qwen", "Qwen3-TTS (clonata, in LuZa)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")], "kokoro_voice": list(KOKORO_VOICES.items()), + "qwen_voce": [(v, v.replace("_", " ")) for v in __import__("avatar.tts", fromlist=["qwen_voices"]).qwen_voices()], "immagini_famiglia": [("z-image-turbo", "Z-Image Turbo"), ("schnell", "FLUX schnell"), ("dev", "FLUX dev"), ("qwen", "Qwen-Image")], } try: diff --git a/avatar/settings.py b/avatar/settings.py index 807c8de..7324fb4 100644 --- a/avatar/settings.py +++ b/avatar/settings.py @@ -57,6 +57,7 @@ DEFAULTS: dict[str, Any] = { "abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria "umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente "io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate + "qwen_voce": "luza_voce", # voce clonata (file in data/voices) per il motore Qwen3-TTS "ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth "ipixel_address": "", # identificativo BLE (rilevato da solo) "ipixel_luminosita": 40, diff --git a/avatar/settings_dialog.py b/avatar/settings_dialog.py index 86deeb3..c5dfaa1 100644 --- a/avatar/settings_dialog.py +++ b/avatar/settings_dialog.py @@ -141,6 +141,15 @@ class SettingsDialog(QDialog): form2 = QFormLayout() self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("qwen", "Qwen3-TTS dentro LuZa: voce clonata, quasi in tempo reale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine")) form2.addRow("Motore voce", self.tts_engine) + from .tts import qwen_voices + row = QHBoxLayout() + self.qwen_voce = QComboBox(); self.qwen_voce.addItems(qwen_voices()) + self.qwen_voce.setCurrentText(str(s.get("qwen_voce") or "")) + row.addWidget(self.qwen_voce, 1) + b = QPushButton("Ascolta"); b.clicked.connect(self._qwen_play); row.addWidget(b) + b = QPushButton("Importa campione…"); b.clicked.connect(self._qwen_import); row.addWidget(b) + form2.addRow("Voce Qwen3-TTS (clonata)", row) + self.qwen_hint = _note("Campione di 10-20 secondi di una sola voce, senza musica: viene trascritto da solo."); form2.addRow("", self.qwen_hint) self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice) self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice")) form2.addRow("Voce di sistema", self.system_voice) @@ -494,6 +503,31 @@ class SettingsDialog(QDialog): self._async.emit("el_error", str(err)[:120]) threading.Thread(target=work, daemon=True).start() + def _qwen_play(self) -> None: + import subprocess + from .tts import VOICES_DIR + f = VOICES_DIR / f"{self.qwen_voce.currentText()}.wav" + if f.exists(): + subprocess.Popen(["afplay", str(f)]) + + def _qwen_import(self) -> None: + from PyQt6.QtWidgets import QFileDialog, QInputDialog + path, _ = QFileDialog.getOpenFileName(self, "Campione della voce", str(Path.home() / "Downloads"), "Audio (*.mp3 *.m4a *.wav *.aac *.flac *.ogg)") + if not path: + return + name, ok = QInputDialog.getText(self, "Nome della voce", "Nome:", text=Path(path).stem) + if not ok: + return + self.qwen_hint.setText("Importo e trascrivo il campione…") + + def go(): + try: + from .tts import import_voice + self._async.emit("qwen", import_voice(path, name)) + except Exception as err: + self._async.emit("qwen", f"!{err}") + threading.Thread(target=go, daemon=True).start() + def _mcp_example(self) -> None: self.mcp_text.setPlainText(json.dumps({"mcpServers": { "filesystem": {"command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", str(Path.home() / "Documents")]}, @@ -535,6 +569,12 @@ class SettingsDialog(QDialog): def _on_async(self, kind: str, payload) -> None: if kind == "ha": self.ha_hint.setText(str(payload)); return + if kind == "qwen": + if str(payload).startswith("!"): + self.qwen_hint.setText("Importazione fallita: " + str(payload)[1:]); return + from .tts import qwen_voices, VOICES_DIR + self.qwen_voce.clear(); self.qwen_voce.addItems(qwen_voices()); self.qwen_voce.setCurrentText(str(payload)) + self.qwen_hint.setText("Voce «%s» importata. Trascrizione: %s" % (payload, (VOICES_DIR / f"{payload}.txt").read_text(encoding="utf-8")[:160])); return if kind == "mcp": self.mcp_hint.setText(str(payload)); return if kind == "vb": @@ -606,7 +646,7 @@ class SettingsDialog(QDialog): "search_api_key": self.search_key.text().strip(), "claudecode_model": self.cc_model.currentData(), "claudecode_access": self.cc_access.currentData(), "claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(), - "tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(), + "tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(), "qwen_voce": self.qwen_voce.currentText(), "system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(), "voicebox_profile_id": self.vb_profile.currentData() or "", "voicebox_engine": self.vb_engine.currentData(), "voicebox_instruct": self.vb_instruct.text().strip(), "elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(), diff --git a/avatar/tts.py b/avatar/tts.py index d834f8b..17b5c97 100644 --- a/avatar/tts.py +++ b/avatar/tts.py @@ -347,6 +347,29 @@ class VoiceboxVoice: return data +VOICES_DIR = __import__("pathlib").Path(__file__).resolve().parent.parent / "data" / "voices" + + +def qwen_voices() -> list[str]: + """Voci clonabili: file .wav in data/voices con la trascrizione .txt accanto.""" + return sorted(p.stem for p in VOICES_DIR.glob("*.wav") if p.with_suffix(".txt").exists()) + + +def import_voice(src: str, name: str = "") -> str: + """Converte un campione (mp3, m4a, wav…) in data/voices/.wav a 24 kHz mono, max 20 s, e lo trascrive.""" + import re + import subprocess + from pathlib import Path as _P + name = re.sub(r"[^a-z0-9_]+", "_", (name or _P(src).stem).lower()).strip("_") or "voce" + VOICES_DIR.mkdir(parents=True, exist_ok=True) + out = VOICES_DIR / f"{name}.wav" + subprocess.run(["ffmpeg", "-loglevel", "error", "-y", "-i", src, "-t", "20", "-ac", "1", "-ar", "24000", str(out)], check=True) + import mlx_whisper + text = mlx_whisper.transcribe(str(out), path_or_hf_repo="mlx-community/whisper-large-v3-turbo", language="it")["text"].strip() + out.with_suffix(".txt").write_text(text, encoding="utf-8") + return name + + class QwenVoice: """Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread.""" @@ -388,8 +411,10 @@ class QwenVoice: def make_voice(settings, on_status=None): if settings.get("tts_engine") == "qwen": - from pathlib import Path as _P - ref = _P(str(settings.get("qwen_ref") or "")) if settings.get("qwen_ref") else _P(__file__).resolve().parent.parent / "data" / "voices" / "qwen_ref.wav" + voices = qwen_voices() + ref = VOICES_DIR / f"{settings.get('qwen_voce') or ''}.wav" + if not ref.exists() and voices: + ref = VOICES_DIR / f"{voices[0]}.wav" txt = ref.with_suffix(".txt") return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status) if settings.get("tts_engine") == "voicebox": diff --git a/remote/app.js b/remote/app.js index fae0d04..49ba243 100644 --- a/remote/app.js +++ b/remote/app.js @@ -164,9 +164,9 @@ connect(); // ── impostazioni: motore, modello, ragionamento, voce ───────────────────── const LABELS = {provider: 'Motore', effort: 'Profondità (Claude)', mlx_model: 'Modello interno', mlx_thinking: 'Ragionamento (interno)', claudecode_model: 'Modello Claude Code', - tts_engine: 'Voce', kokoro_voice: 'Voce Kokoro', voicebox_profile_id: 'Profilo Voicebox', immagini_famiglia: 'Immagini: famiglia', immagini_modello: 'Immagini: modello'}; + tts_engine: 'Voce', kokoro_voice: 'Voce Kokoro', qwen_voce: 'Voce clonata', voicebox_profile_id: 'Profilo Voicebox', immagini_famiglia: 'Immagini: famiglia', immagini_modello: 'Immagini: modello'}; const SHOW_IF = {mlx_model: v => v.provider === 'mlx', mlx_thinking: v => v.provider === 'mlx', claudecode_model: v => v.provider === 'claudecode', effort: v => v.provider !== 'mlx' && v.provider !== 'local', - kokoro_voice: v => v.tts_engine === 'kokoro', voicebox_profile_id: v => v.tts_engine === 'voicebox'}; + kokoro_voice: v => v.tts_engine === 'kokoro', qwen_voce: v => v.tts_engine === 'qwen', voicebox_profile_id: v => v.tts_engine === 'voicebox'}; let sdata = null; async function openSettings() { $('settings').style.display = 'block'; $('shint').textContent = 'carico…';