Riscrittura in Python dell'assistente Avatar con interfaccia HUD (derivata da Mark LIV, CC BY-NC 4.0, vedi NOTICE.md). Tre motori (Claude API, server locale OpenAI-compatibile, Claude Code), voce Kokoro/macOS, Whisper MLX, avatar 3D con sincronizzazione labiale, memoria per categorie, allegati con OCR, monitor con avvisi, plugin per Calendario, Mail, Promemoria, Note, Musica, app, Mac, timer, meteo, contatti, Messaggi, file, Comandi Rapidi, browser, Telegram, WhatsApp (archivio e tempo reale). Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
276 lines
11 KiB
Python
276 lines
11 KiB
Python
"""
|
||
Text → mouth shape, fused with the audio the avatar is actually speaking.
|
||
|
||
Why both sources
|
||
----------------
|
||
Formant analysis of the audio (see `_pcm_visemes` in main.py) gives excellent
|
||
*timing* and a decent read on vowels, but it is blind to exactly the consonants
|
||
lip-reading depends on. /m/, /b/ and /p/ are made with the lips pressed shut,
|
||
and nothing in the spectrum reliably says "the lips are closed" — a nasal /m/
|
||
and a nasal /n/ look nearly identical to a filter bank while looking completely
|
||
different on a face.
|
||
|
||
The transcript knows those consonants for certain. So the text supplies *which
|
||
shape*, the audio supplies *when* and *how strongly*, and the two are blended.
|
||
If the transcript is late or missing the mouth silently falls back to the
|
||
audio-only shape, which is what the previous version did on its own.
|
||
|
||
Language independence
|
||
---------------------
|
||
There is no per-language table here. Every character is reduced to one of the
|
||
26 bare Latin letters — by Unicode decomposition for accents, by transliteration
|
||
for Cyrillic and Greek — and articulation is looked up on that. So Turkish,
|
||
English, German, French, Spanish, Polish, Vietnamese, Russian, Ukrainian and
|
||
Greek all work from the same twenty-odd rules, and adding a language costs
|
||
nothing because there is nothing to add.
|
||
|
||
Scripts whose spelling does not reveal pronunciation (CJK, Arabic, Devanagari,
|
||
Hebrew, Thai) are detected by coverage and skipped, and the mouth runs on the
|
||
audio-only shape — which is itself language-independent, being physics. The
|
||
result is never wrong, only less detailed.
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import unicodedata
|
||
from collections import deque
|
||
|
||
# (openness 0..1, width -1..+1, closure 0..1)
|
||
# closure forces the lips together regardless of loudness — it is the whole
|
||
# reason the transcript is worth consulting.
|
||
VISEMES: dict[str, tuple[float, float, float]] = {
|
||
"REST": (0.00, 0.00, 0.00),
|
||
"AA": (0.92, -0.05, 0.00), # a
|
||
"E": (0.52, 0.42, 0.00), # e
|
||
"I": (0.20, 0.62, 0.00), # i, ı
|
||
"O": (0.55, -0.52, 0.00), # o, ö
|
||
"U": (0.26, -0.74, 0.00), # u, ü, w
|
||
"MBP": (0.00, 0.00, 1.00), # m, b, p — lips pressed shut
|
||
"FV": (0.10, 0.22, 0.55), # f, v — lower lip to the teeth
|
||
"S": (0.16, 0.42, 0.00), # s, ş, z, c, ç, j
|
||
"L": (0.36, 0.18, 0.00), # l
|
||
"TD": (0.28, 0.12, 0.00), # t, d, n
|
||
"K": (0.30, -0.04, 0.00), # k, g, ğ, h
|
||
"R": (0.28, -0.16, 0.00), # r
|
||
}
|
||
|
||
# Relative duration of each class. Vowels carry the syllable; plosives are a tap.
|
||
_DUR = {"REST": 1.0, "AA": 1.15, "E": 1.05, "I": 1.0, "O": 1.1, "U": 1.05,
|
||
"MBP": 0.5, "FV": 0.8, "S": 0.9, "L": 0.65, "TD": 0.5, "K": 0.55,
|
||
"R": 0.55}
|
||
|
||
# Articulation is a property of the *sound*, not of a language, so the table is
|
||
# keyed on the 26 bare Latin letters and every script reaches it by reduction:
|
||
# * diacritics are stripped by Unicode decomposition (é→e, ü→u, ế→e, ł→l …),
|
||
# which covers every Latin-script language at once rather than one at a time;
|
||
# * letters that do not decompose get a short explicit entry below;
|
||
# * Cyrillic and Greek transliterate into the same 26 letters.
|
||
# Anything else — CJK, Arabic, Devanagari, Hebrew, Thai — is not derivable from
|
||
# its written form without a pronunciation dictionary, so those simply fall back
|
||
# to the audio-only mouth. That is a clean degradation, not a missing language.
|
||
_LETTER = {
|
||
"a": "AA",
|
||
"e": "E",
|
||
"i": "I", "y": "I",
|
||
"o": "O",
|
||
"u": "U", "w": "U",
|
||
"b": "MBP", "p": "MBP", "m": "MBP",
|
||
"f": "FV", "v": "FV",
|
||
"s": "S", "z": "S", "c": "S", "j": "S", "x": "S",
|
||
"l": "L",
|
||
"t": "TD", "d": "TD", "n": "TD",
|
||
"k": "K", "g": "K", "h": "K", "q": "K",
|
||
"r": "R",
|
||
}
|
||
|
||
# Letters with no Unicode decomposition into a Latin base.
|
||
_UNDECOMPOSED = {
|
||
"ı": "i", "ø": "o", "đ": "d", "ħ": "h", "ŀ": "l", "ŧ": "t",
|
||
"ß": "s", "æ": "a", "œ": "o", "þ": "t", "ð": "d", "ŋ": "n",
|
||
"ł": "l",
|
||
}
|
||
|
||
_CYRILLIC = {
|
||
"а": "a", "б": "b", "в": "v", "г": "g", "д": "d", "е": "e", "ё": "e",
|
||
"ж": "j", "з": "z", "и": "i", "й": "i", "к": "k", "л": "l", "м": "m",
|
||
"н": "n", "о": "o", "п": "p", "р": "r", "с": "s", "т": "t", "у": "u",
|
||
"ф": "f", "х": "h", "ц": "s", "ч": "s", "ш": "s", "щ": "s", "ъ": "",
|
||
"ы": "i", "ь": "", "э": "e", "ю": "u", "я": "a",
|
||
"і": "i", "ї": "i", "є": "e", "ґ": "g", "ў": "u",
|
||
}
|
||
|
||
_GREEK = {
|
||
"α": "a", "β": "v", "γ": "g", "δ": "d", "ε": "e", "ζ": "z", "η": "i",
|
||
"θ": "t", "ι": "i", "κ": "k", "λ": "l", "μ": "m", "ν": "n", "ξ": "s",
|
||
"ο": "o", "π": "p", "ρ": "r", "σ": "s", "ς": "s", "τ": "t", "υ": "i",
|
||
"φ": "f", "χ": "h", "ψ": "s", "ω": "o",
|
||
}
|
||
|
||
# Below this share of mappable letters the text is in a script we cannot read
|
||
# phonetically, and forcing shapes onto it would be worse than not trying.
|
||
_MIN_COVERAGE = 0.55
|
||
|
||
# English spellings that do not survive letter-by-letter reading.
|
||
_DIGRAPH = {
|
||
"sh": "S", "ch": "S", "ts": "S",
|
||
"th": "TD", "ck": "K", "ng": "K", "gh": "K",
|
||
"ph": "FV",
|
||
"oo": "U", "ou": "O", "ow": "O", "wh": "U",
|
||
"ee": "I", "ea": "I", "ie": "I",
|
||
"qu": "K",
|
||
}
|
||
|
||
_PAUSE = set(".,;:!?…\n")
|
||
|
||
|
||
def to_latin(ch: str) -> str:
|
||
"""Reduce any character to a bare Latin letter, or "" if it has none.
|
||
|
||
This is what makes the mouth language-agnostic: one reduction step replaces
|
||
a per-language spelling table.
|
||
"""
|
||
c = ch.lower()
|
||
if "a" <= c <= "z":
|
||
return c
|
||
if c in _UNDECOMPOSED:
|
||
return _UNDECOMPOSED[c]
|
||
if c in _CYRILLIC:
|
||
return _CYRILLIC[c]
|
||
if c in _GREEK:
|
||
return _GREEK[c]
|
||
# Strip combining marks: é→e, ü→u, ş→s, ğ→g, ế→e, ñ→n, å→a …
|
||
base = "".join(k for k in unicodedata.normalize("NFD", c)
|
||
if not unicodedata.combining(k))
|
||
if len(base) == 1 and "a" <= base <= "z":
|
||
return base
|
||
if base and base != c: # e.g. fi → fi, take the first
|
||
return to_latin(base[0])
|
||
return ""
|
||
|
||
|
||
def coverage(text: str) -> float:
|
||
"""Fraction of the letters in `text` we can reduce to a Latin sound."""
|
||
letters = [c for c in (text or "") if c.isalpha()]
|
||
if not letters:
|
||
return 0.0
|
||
return sum(1 for c in letters if to_latin(c)) / len(letters)
|
||
|
||
|
||
def text_to_visemes(text: str) -> list[tuple[str, float]]:
|
||
"""Split a line of speech into (viseme, duration-weight) pairs.
|
||
|
||
Returns [] for scripts whose written form does not reveal pronunciation, so
|
||
the caller falls back to the audio-only mouth instead of miming nonsense.
|
||
"""
|
||
s = (text or "").lower()
|
||
if coverage(s) < _MIN_COVERAGE:
|
||
return []
|
||
|
||
out: list[tuple[str, float]] = []
|
||
i, n = 0, len(s)
|
||
while i < n:
|
||
ch = s[i]
|
||
if ch in _PAUSE:
|
||
out.append(("REST", 1.4))
|
||
i += 1
|
||
continue
|
||
if ch.isspace():
|
||
# A word gap is a beat, not a closed mouth — closing between every
|
||
# word makes the avatar look like it is chewing.
|
||
if out and out[-1][0] != "REST":
|
||
out.append((out[-1][0], 0.35))
|
||
i += 1
|
||
continue
|
||
|
||
# Digraphs are an orthographic quirk of Latin spelling; check them on
|
||
# the reduced letters so "SCH"/"Sch" and accented forms match too.
|
||
two = to_latin(ch) + (to_latin(s[i + 1]) if i + 1 < n else "")
|
||
if len(two) == 2 and two in _DIGRAPH:
|
||
v = _DIGRAPH[two]
|
||
i += 2
|
||
else:
|
||
base = to_latin(ch)
|
||
i += 1
|
||
if not base:
|
||
continue
|
||
v = _LETTER.get(base)
|
||
if v is None:
|
||
continue
|
||
# A doubled letter is one sound in every orthography we handle here.
|
||
if out and out[-1][0] == v:
|
||
continue
|
||
out.append((v, _DUR[v]))
|
||
return out
|
||
|
||
|
||
class VisemeStream:
|
||
"""Fuses the transcript's shape sequence onto the audio's timing.
|
||
|
||
Thread note: `feed_text` runs on the receive coroutine and `frames` on the
|
||
playback coroutine. Both live in the same asyncio loop, and `deque` append
|
||
and popleft are atomic, so no lock is needed.
|
||
"""
|
||
|
||
# Seconds a phoneme occupies at a normal speaking rate. The clock adapts
|
||
# between these when the queue runs long (the model is talking fast) or
|
||
# short (it is trailing off).
|
||
_MIN_STEP = 0.045
|
||
_MAX_STEP = 0.105
|
||
|
||
def __init__(self) -> None:
|
||
self._q: deque[tuple[str, float]] = deque()
|
||
self._cur = ("REST", 1.0)
|
||
self._carry = 0.0
|
||
|
||
def reset(self) -> None:
|
||
self._q.clear()
|
||
self._cur = ("REST", 1.0)
|
||
self._carry = 0.0
|
||
|
||
def feed_text(self, text: str) -> None:
|
||
for item in text_to_visemes(text):
|
||
self._q.append(item)
|
||
# Never let a stalled turn pile up an unbounded backlog.
|
||
while len(self._q) > 600:
|
||
self._q.popleft()
|
||
|
||
@property
|
||
def pending(self) -> int:
|
||
return len(self._q)
|
||
|
||
def _step_seconds(self) -> float:
|
||
# A long backlog means speech is outrunning the clock; shorten the step
|
||
# so the mouth catches up instead of drifting further behind the voice.
|
||
backlog = min(1.0, len(self._q) / 45.0)
|
||
return self._MAX_STEP - (self._MAX_STEP - self._MIN_STEP) * backlog
|
||
|
||
def frames(self, audio, hop: float):
|
||
"""Blend audio frames [(level, openness, width)] with the text queue."""
|
||
out = []
|
||
for level, a_open, a_wide in audio:
|
||
if level <= 0.0:
|
||
# Silence: let the queue wait rather than burning through it
|
||
# during a pause, or the mouth ends up ahead of the voice.
|
||
out.append((0.0, 0.0, 0.0))
|
||
continue
|
||
|
||
self._carry += hop / max(1e-3, self._step_seconds() * self._cur[1])
|
||
while self._carry >= 1.0 and self._q:
|
||
self._cur = self._q.popleft()
|
||
self._carry -= 1.0
|
||
if self._carry >= 1.0:
|
||
self._carry = 1.0 # queue empty — hold the last shape
|
||
|
||
t_open, t_wide, closure = VISEMES.get(self._cur[0], VISEMES["REST"])
|
||
if self._q or self._cur[0] != "REST":
|
||
# Text leads the shape; the audio keeps it honest so a bad
|
||
# transcript alignment still tracks the real voice.
|
||
o = 0.72 * t_open + 0.28 * a_open
|
||
w = 0.78 * t_wide + 0.22 * a_wide
|
||
else:
|
||
o, w = a_open, a_wide
|
||
closure = 0.0
|
||
o *= 1.0 - closure
|
||
out.append((level, max(0.0, min(1.0, o)), max(-1.0, min(1.0, w))))
|
||
return out
|