Voci clonate Qwen3-TTS selezionabili (Motore e Voce e telefono), importazione di un campione con trascrizione automatica
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
75781ea14a
commit
e2dbd31388
5 files changed
+73
-6
No files matched your search
+2
-1
@@ -270,7 +270,7 @@ class RemoteServer:
|
|||||||
|
|
||||||
# ── impostazioni dal telefono: motore, modello, ragionamento, voce ─────
|
# ── impostazioni dal telefono: motore, modello, ragionamento, voce ─────
|
||||||
SETTING_KEYS = ("provider", "effort", "mlx_model", "mlx_thinking", "claudecode_model", "local_model",
|
SETTING_KEYS = ("provider", "effort", "mlx_model", "mlx_thinking", "claudecode_model", "local_model",
|
||||||
"tts_engine", "kokoro_voice", "voicebox_profile_id", "immagini_famiglia", "immagini_modello")
|
"tts_engine", "kokoro_voice", "voicebox_profile_id", "qwen_voce", "immagini_famiglia", "immagini_modello")
|
||||||
|
|
||||||
def _options(self) -> dict:
|
def _options(self) -> dict:
|
||||||
from avatar.tts import KOKORO_VOICES
|
from avatar.tts import KOKORO_VOICES
|
||||||
@@ -283,6 +283,7 @@ class RemoteServer:
|
|||||||
"claudecode_model": [("sonnet", "Sonnet"), ("opus", "Opus"), ("haiku", "Haiku")],
|
"claudecode_model": [("sonnet", "Sonnet"), ("opus", "Opus"), ("haiku", "Haiku")],
|
||||||
"tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("qwen", "Qwen3-TTS (clonata, in LuZa)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")],
|
"tts_engine": [("kokoro", "Kokoro (locale, rapida)"), ("qwen", "Qwen3-TTS (clonata, in LuZa)"), ("voicebox", "Voicebox (clonata)"), ("elevenlabs", "ElevenLabs"), ("chatterbox", "Chatterbox"), ("system", "Voce di sistema")],
|
||||||
"kokoro_voice": list(KOKORO_VOICES.items()),
|
"kokoro_voice": list(KOKORO_VOICES.items()),
|
||||||
|
"qwen_voce": [(v, v.replace("_", " ")) for v in __import__("avatar.tts", fromlist=["qwen_voices"]).qwen_voices()],
|
||||||
"immagini_famiglia": [("z-image-turbo", "Z-Image Turbo"), ("schnell", "FLUX schnell"), ("dev", "FLUX dev"), ("qwen", "Qwen-Image")],
|
"immagini_famiglia": [("z-image-turbo", "Z-Image Turbo"), ("schnell", "FLUX schnell"), ("dev", "FLUX dev"), ("qwen", "Qwen-Image")],
|
||||||
}
|
}
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -57,6 +57,7 @@ DEFAULTS: dict[str, Any] = {
|
|||||||
"abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria
|
"abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria
|
||||||
"umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente
|
"umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente
|
||||||
"io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate
|
"io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate
|
||||||
|
"qwen_voce": "luza_voce", # voce clonata (file in data/voices) per il motore Qwen3-TTS
|
||||||
"ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth
|
"ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth
|
||||||
"ipixel_address": "", # identificativo BLE (rilevato da solo)
|
"ipixel_address": "", # identificativo BLE (rilevato da solo)
|
||||||
"ipixel_luminosita": 40,
|
"ipixel_luminosita": 40,
|
||||||
|
|||||||
@@ -141,6 +141,15 @@ class SettingsDialog(QDialog):
|
|||||||
form2 = QFormLayout()
|
form2 = QFormLayout()
|
||||||
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("qwen", "Qwen3-TTS dentro LuZa: voce clonata, quasi in tempo reale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
|
self.tts_engine = _combo([("kokoro", "Kokoro, voce neurale in locale"), ("qwen", "Qwen3-TTS dentro LuZa: voce clonata, quasi in tempo reale"), ("voicebox", "Voicebox: Qwen3-TTS in locale, espressiva, voce clonata"), ("elevenlabs", "ElevenLabs, espressiva nel cloud (chiave API)"), ("chatterbox", "Chatterbox, espressiva in locale (lenta)"), ("system", "Voce di sistema (macOS)")], s.get("tts_engine"))
|
||||||
form2.addRow("Motore voce", self.tts_engine)
|
form2.addRow("Motore voce", self.tts_engine)
|
||||||
|
from .tts import qwen_voices
|
||||||
|
row = QHBoxLayout()
|
||||||
|
self.qwen_voce = QComboBox(); self.qwen_voce.addItems(qwen_voices())
|
||||||
|
self.qwen_voce.setCurrentText(str(s.get("qwen_voce") or ""))
|
||||||
|
row.addWidget(self.qwen_voce, 1)
|
||||||
|
b = QPushButton("Ascolta"); b.clicked.connect(self._qwen_play); row.addWidget(b)
|
||||||
|
b = QPushButton("Importa campione…"); b.clicked.connect(self._qwen_import); row.addWidget(b)
|
||||||
|
form2.addRow("Voce Qwen3-TTS (clonata)", row)
|
||||||
|
self.qwen_hint = _note("Campione di 10-20 secondi di una sola voce, senza musica: viene trascritto da solo."); form2.addRow("", self.qwen_hint)
|
||||||
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
|
self.kokoro_voice = _combo(list(KOKORO_VOICES.items()), s.get("kokoro_voice")); form2.addRow("Voce Kokoro", self.kokoro_voice)
|
||||||
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
|
self.system_voice = _combo([("", "Automatica (Alice)")] + [(v, v) for v in SystemVoice.list_voices()], s.get("system_voice"))
|
||||||
form2.addRow("Voce di sistema", self.system_voice)
|
form2.addRow("Voce di sistema", self.system_voice)
|
||||||
@@ -494,6 +503,31 @@ class SettingsDialog(QDialog):
|
|||||||
self._async.emit("el_error", str(err)[:120])
|
self._async.emit("el_error", str(err)[:120])
|
||||||
threading.Thread(target=work, daemon=True).start()
|
threading.Thread(target=work, daemon=True).start()
|
||||||
|
|
||||||
|
def _qwen_play(self) -> None:
|
||||||
|
import subprocess
|
||||||
|
from .tts import VOICES_DIR
|
||||||
|
f = VOICES_DIR / f"{self.qwen_voce.currentText()}.wav"
|
||||||
|
if f.exists():
|
||||||
|
subprocess.Popen(["afplay", str(f)])
|
||||||
|
|
||||||
|
def _qwen_import(self) -> None:
|
||||||
|
from PyQt6.QtWidgets import QFileDialog, QInputDialog
|
||||||
|
path, _ = QFileDialog.getOpenFileName(self, "Campione della voce", str(Path.home() / "Downloads"), "Audio (*.mp3 *.m4a *.wav *.aac *.flac *.ogg)")
|
||||||
|
if not path:
|
||||||
|
return
|
||||||
|
name, ok = QInputDialog.getText(self, "Nome della voce", "Nome:", text=Path(path).stem)
|
||||||
|
if not ok:
|
||||||
|
return
|
||||||
|
self.qwen_hint.setText("Importo e trascrivo il campione…")
|
||||||
|
|
||||||
|
def go():
|
||||||
|
try:
|
||||||
|
from .tts import import_voice
|
||||||
|
self._async.emit("qwen", import_voice(path, name))
|
||||||
|
except Exception as err:
|
||||||
|
self._async.emit("qwen", f"!{err}")
|
||||||
|
threading.Thread(target=go, daemon=True).start()
|
||||||
|
|
||||||
def _mcp_example(self) -> None:
|
def _mcp_example(self) -> None:
|
||||||
self.mcp_text.setPlainText(json.dumps({"mcpServers": {
|
self.mcp_text.setPlainText(json.dumps({"mcpServers": {
|
||||||
"filesystem": {"command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", str(Path.home() / "Documents")]},
|
"filesystem": {"command": "npx", "args": ["-y", "@modelcontextprotocol/server-filesystem", str(Path.home() / "Documents")]},
|
||||||
@@ -535,6 +569,12 @@ class SettingsDialog(QDialog):
|
|||||||
def _on_async(self, kind: str, payload) -> None:
|
def _on_async(self, kind: str, payload) -> None:
|
||||||
if kind == "ha":
|
if kind == "ha":
|
||||||
self.ha_hint.setText(str(payload)); return
|
self.ha_hint.setText(str(payload)); return
|
||||||
|
if kind == "qwen":
|
||||||
|
if str(payload).startswith("!"):
|
||||||
|
self.qwen_hint.setText("Importazione fallita: " + str(payload)[1:]); return
|
||||||
|
from .tts import qwen_voices, VOICES_DIR
|
||||||
|
self.qwen_voce.clear(); self.qwen_voce.addItems(qwen_voices()); self.qwen_voce.setCurrentText(str(payload))
|
||||||
|
self.qwen_hint.setText("Voce «%s» importata. Trascrizione: %s" % (payload, (VOICES_DIR / f"{payload}.txt").read_text(encoding="utf-8")[:160])); return
|
||||||
if kind == "mcp":
|
if kind == "mcp":
|
||||||
self.mcp_hint.setText(str(payload)); return
|
self.mcp_hint.setText(str(payload)); return
|
||||||
if kind == "vb":
|
if kind == "vb":
|
||||||
@@ -606,7 +646,7 @@ class SettingsDialog(QDialog):
|
|||||||
"search_api_key": self.search_key.text().strip(),
|
"search_api_key": self.search_key.text().strip(),
|
||||||
"claudecode_model": self.cc_model.currentData(), "claudecode_access": self.cc_access.currentData(),
|
"claudecode_model": self.cc_model.currentData(), "claudecode_access": self.cc_access.currentData(),
|
||||||
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
||||||
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(),
|
"tts_engine": self.tts_engine.currentData(), "kokoro_voice": self.kokoro_voice.currentData(), "qwen_voce": self.qwen_voce.currentText(),
|
||||||
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
|
"system_voice": self.system_voice.currentData(), "stt_model": self.stt_model.currentData(),
|
||||||
"voicebox_profile_id": self.vb_profile.currentData() or "", "voicebox_engine": self.vb_engine.currentData(), "voicebox_instruct": self.vb_instruct.text().strip(),
|
"voicebox_profile_id": self.vb_profile.currentData() or "", "voicebox_engine": self.vb_engine.currentData(), "voicebox_instruct": self.vb_instruct.text().strip(),
|
||||||
"elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(),
|
"elevenlabs_voice_id": self.el_voice.currentData() or "", "elevenlabs_model": self.el_model.currentData(),
|
||||||
|
|||||||
+27
-2
@@ -347,6 +347,29 @@ class VoiceboxVoice:
|
|||||||
return data
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
VOICES_DIR = __import__("pathlib").Path(__file__).resolve().parent.parent / "data" / "voices"
|
||||||
|
|
||||||
|
|
||||||
|
def qwen_voices() -> list[str]:
|
||||||
|
"""Voci clonabili: file .wav in data/voices con la trascrizione .txt accanto."""
|
||||||
|
return sorted(p.stem for p in VOICES_DIR.glob("*.wav") if p.with_suffix(".txt").exists())
|
||||||
|
|
||||||
|
|
||||||
|
def import_voice(src: str, name: str = "") -> str:
|
||||||
|
"""Converte un campione (mp3, m4a, wav…) in data/voices/<nome>.wav a 24 kHz mono, max 20 s, e lo trascrive."""
|
||||||
|
import re
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path as _P
|
||||||
|
name = re.sub(r"[^a-z0-9_]+", "_", (name or _P(src).stem).lower()).strip("_") or "voce"
|
||||||
|
VOICES_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
out = VOICES_DIR / f"{name}.wav"
|
||||||
|
subprocess.run(["ffmpeg", "-loglevel", "error", "-y", "-i", src, "-t", "20", "-ac", "1", "-ar", "24000", str(out)], check=True)
|
||||||
|
import mlx_whisper
|
||||||
|
text = mlx_whisper.transcribe(str(out), path_or_hf_repo="mlx-community/whisper-large-v3-turbo", language="it")["text"].strip()
|
||||||
|
out.with_suffix(".txt").write_text(text, encoding="utf-8")
|
||||||
|
return name
|
||||||
|
|
||||||
|
|
||||||
class QwenVoice:
|
class QwenVoice:
|
||||||
"""Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano
|
"""Qwen3-TTS dentro LuZa (mlx-audio), voce clonata da un campione + trascrizione. Modello e generazione girano
|
||||||
sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread."""
|
sempre sullo stesso thread: MLX non condivide gli stream GPU tra thread."""
|
||||||
@@ -388,8 +411,10 @@ class QwenVoice:
|
|||||||
|
|
||||||
def make_voice(settings, on_status=None):
|
def make_voice(settings, on_status=None):
|
||||||
if settings.get("tts_engine") == "qwen":
|
if settings.get("tts_engine") == "qwen":
|
||||||
from pathlib import Path as _P
|
voices = qwen_voices()
|
||||||
ref = _P(str(settings.get("qwen_ref") or "")) if settings.get("qwen_ref") else _P(__file__).resolve().parent.parent / "data" / "voices" / "qwen_ref.wav"
|
ref = VOICES_DIR / f"{settings.get('qwen_voce') or ''}.wav"
|
||||||
|
if not ref.exists() and voices:
|
||||||
|
ref = VOICES_DIR / f"{voices[0]}.wav"
|
||||||
txt = ref.with_suffix(".txt")
|
txt = ref.with_suffix(".txt")
|
||||||
return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status)
|
return QwenVoice(str(ref), txt.read_text(encoding="utf-8") if txt.exists() else "", on_status=on_status)
|
||||||
if settings.get("tts_engine") == "voicebox":
|
if settings.get("tts_engine") == "voicebox":
|
||||||
|
|||||||
+2
-2
@@ -164,9 +164,9 @@ connect();
|
|||||||
|
|
||||||
// ── impostazioni: motore, modello, ragionamento, voce ─────────────────────
|
// ── impostazioni: motore, modello, ragionamento, voce ─────────────────────
|
||||||
const LABELS = {provider: 'Motore', effort: 'Profondità (Claude)', mlx_model: 'Modello interno', mlx_thinking: 'Ragionamento (interno)', claudecode_model: 'Modello Claude Code',
|
const LABELS = {provider: 'Motore', effort: 'Profondità (Claude)', mlx_model: 'Modello interno', mlx_thinking: 'Ragionamento (interno)', claudecode_model: 'Modello Claude Code',
|
||||||
tts_engine: 'Voce', kokoro_voice: 'Voce Kokoro', voicebox_profile_id: 'Profilo Voicebox', immagini_famiglia: 'Immagini: famiglia', immagini_modello: 'Immagini: modello'};
|
tts_engine: 'Voce', kokoro_voice: 'Voce Kokoro', qwen_voce: 'Voce clonata', voicebox_profile_id: 'Profilo Voicebox', immagini_famiglia: 'Immagini: famiglia', immagini_modello: 'Immagini: modello'};
|
||||||
const SHOW_IF = {mlx_model: v => v.provider === 'mlx', mlx_thinking: v => v.provider === 'mlx', claudecode_model: v => v.provider === 'claudecode', effort: v => v.provider !== 'mlx' && v.provider !== 'local',
|
const SHOW_IF = {mlx_model: v => v.provider === 'mlx', mlx_thinking: v => v.provider === 'mlx', claudecode_model: v => v.provider === 'claudecode', effort: v => v.provider !== 'mlx' && v.provider !== 'local',
|
||||||
kokoro_voice: v => v.tts_engine === 'kokoro', voicebox_profile_id: v => v.tts_engine === 'voicebox'};
|
kokoro_voice: v => v.tts_engine === 'kokoro', qwen_voce: v => v.tts_engine === 'qwen', voicebox_profile_id: v => v.tts_engine === 'voicebox'};
|
||||||
let sdata = null;
|
let sdata = null;
|
||||||
async function openSettings() {
|
async function openSettings() {
|
||||||
$('settings').style.display = 'block'; $('shint').textContent = 'carico…';
|
$('settings').style.display = 'block'; $('shint').textContent = 'carico…';
|
||||||
|
|||||||
Reference in new issue
Block a user