AvatarPy: assistente personale con volto 3D, voce, memoria e plugin
Riscrittura in Python dell'assistente Avatar con interfaccia HUD (derivata da Mark LIV, CC BY-NC 4.0, vedi NOTICE.md). Tre motori (Claude API, server locale OpenAI-compatibile, Claude Code), voce Kokoro/macOS, Whisper MLX, avatar 3D con sincronizzazione labiale, memoria per categorie, allegati con OCR, monitor con avvisi, plugin per Calendario, Mail, Promemoria, Note, Musica, app, Mac, timer, meteo, contatti, Messaggi, file, Comandi Rapidi, browser, Telegram, WhatsApp (archivio e tempo reale). Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
This commit is contained in:
commit
ff79832c30
76 files changed
+82468
No files matched your search
Whitespace-only changes.
@@ -0,0 +1,421 @@
|
||||
"""
|
||||
core/audio_devices.py — pick which microphone and which speakers JARVIS uses.
|
||||
|
||||
WHY
|
||||
Both audio streams in main.py were opened without a `device=` argument, so
|
||||
they always took whatever the operating system called "default". On a laptop
|
||||
with a built-in mic, a webcam mic and a headset that is a coin toss — and on
|
||||
Windows the default *moves on its own* the moment you plug a headset in.
|
||||
"JARVIS can't hear me" almost always means "JARVIS is listening to the
|
||||
monitor's microphone".
|
||||
|
||||
WHY NAMES, NOT INDICES
|
||||
sounddevice identifies devices by integer index, and those indices shift
|
||||
whenever a device appears or disappears. Storing index 3 means that after
|
||||
unplugging a USB interface the saved setting silently points at something
|
||||
else. We store the device *name* and resolve it to an index at open time.
|
||||
|
||||
WHY THIS IS CACHED
|
||||
`sd.query_devices()` talks to the host audio API and can take a few hundred
|
||||
milliseconds on a Windows machine with many endpoints. Mark LV learned this
|
||||
lesson the expensive way — a 2.1-second `openwakeword` import on the Qt
|
||||
thread made the settings drawer look like it was broken. So the list is
|
||||
fetched once on a background thread at startup and served from cache.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
|
||||
# The label shown for "let the OS decide", and the value stored in config for
|
||||
# it. Empty string, so an untouched install and a deliberately-default install
|
||||
# are the same thing — nothing changes for anyone who never opens the picker.
|
||||
DEFAULT_LABEL = "System default"
|
||||
DEFAULT_VALUE = ""
|
||||
|
||||
_cache: dict[str, list[str]] | None = None
|
||||
_cache_lock = threading.Lock()
|
||||
|
||||
# Which host API each direction settled on, so resolve() opens the same endpoint
|
||||
# the picker listed. Filled in by _query().
|
||||
_chosen_api: dict = {"input": None, "output": None}
|
||||
|
||||
|
||||
# ── Why the raw list is unusable, and what is filtered out ───────────────────
|
||||
#
|
||||
# `sd.query_devices()` returns one entry per (device × host API), not one per
|
||||
# device. Measured on a normal Windows machine: 41 entries for what the Windows
|
||||
# sound settings show as 4 microphones and 4 speakers. The same Realtek
|
||||
# microphone appears four times — once each under MME, DirectSound, WASAPI and
|
||||
# WDM-KS — and none of the four is labelled to say which is which.
|
||||
#
|
||||
# Handing that to a person is not a choice, it is a quiz. So the list is reduced
|
||||
# the way the operating system's own settings panel does it:
|
||||
#
|
||||
# 1. ONE host API per direction — chosen by measurement, not by reasoning.
|
||||
# See the preference note below.
|
||||
# 2. No pseudo-devices. "Microsoft Sound Mapper", "Primary Sound Driver",
|
||||
# ALSA's "default"/"sysdefault"/"dmix" are aliases for "whatever the OS
|
||||
# picks" — which is precisely the "System default" entry already at the top
|
||||
# of the list. Offering them again as if they were hardware is noise.
|
||||
# 3. No zero-channel or unnamed entries (WASAPI reports one of each).
|
||||
# 4. Deduplicated by name.
|
||||
#
|
||||
# Nothing is hidden that a person could actually want: the same hardware is
|
||||
# still there, listed once, under the name their operating system uses for it.
|
||||
|
||||
# ── Preference order, which is only a starting point ─────────────────────────
|
||||
#
|
||||
# Two earlier versions of this file were wrong in the same way: they decided
|
||||
# which host API to use by reasoning about it instead of measuring it.
|
||||
#
|
||||
# v1 ranked APIs by how clean their device names were and picked WASAPI.
|
||||
# WASAPI in shared mode does not resample — the hardware runs at 48 kHz,
|
||||
# this app streams 16 kHz in and 24 kHz out, and every open failed with
|
||||
# "Invalid sample rate". The picker looked right and did nothing.
|
||||
#
|
||||
# v2 added a rate check and picked DirectSound, which passes that check on
|
||||
# both sides. PortAudio's DirectSound *output* is a silent sink: the
|
||||
# stream opens, every write returns success in ~0 ms, and nothing is ever
|
||||
# heard. Same failure, one layer deeper.
|
||||
#
|
||||
# So this list is a preference, not a promise. Which API actually gets used is
|
||||
# decided below by _usable() (open it for real) and _transport_works() (does
|
||||
# audio actually move), per direction. On this platform that lands on
|
||||
# DirectSound for the microphone and MME for the speakers — a split that no
|
||||
# amount of reasoning would have produced.
|
||||
_PREFERRED_APIS = {
|
||||
"Windows": ("directsound", "mme", "wasapi"),
|
||||
# macOS has only Core Audio, so there is nothing to disambiguate.
|
||||
"Darwin": ("core audio",),
|
||||
# PulseAudio/PipeWire present one clean endpoint per device; raw ALSA
|
||||
# presents dozens of routing permutations of the same card.
|
||||
"Linux": ("pulse", "pipewire", "jack", "alsa"),
|
||||
}
|
||||
|
||||
# ── "It opens" is not "it works" ─────────────────────────────────────────────
|
||||
#
|
||||
# Opening a stream successfully proves nothing. Measured, writing 2.0 s of audio:
|
||||
#
|
||||
# device=None (MME) 2.02 s consumed in real time
|
||||
# DirectSound, any output device 0.00 s swallowed instantly
|
||||
#
|
||||
# No flag or capability field reports this. The only thing that separates a real
|
||||
# sink from a fake one is whether it consumes audio at the rate audio is
|
||||
# consumed at — so that is what gets measured, once per host API per direction,
|
||||
# on the background thread at startup, using silence.
|
||||
#
|
||||
# Each direction is probed **the way main.py actually uses it**. That is not a
|
||||
# detail: DirectSound input passes a callback stream and fails a blocking read,
|
||||
# so an earlier version of this probe rejected a microphone that works perfectly
|
||||
# in the app. Probe the mode you ship, not the mode that is easier to write.
|
||||
_PROBE_SECONDS = {"output": 0.6, "input": 0.35}
|
||||
|
||||
# Cache: {(api_name_or_None, kind): bool}
|
||||
_probe_results: dict = {}
|
||||
|
||||
|
||||
def _transport_works(idx: int, kind: str, api_key) -> bool:
|
||||
"""Does this host API actually move audio, or only pretend to?
|
||||
|
||||
Probed once per API per direction and cached. Output writes silence, so the
|
||||
probe is inaudible; input reads and discards."""
|
||||
if api_key in _probe_results:
|
||||
return _probe_results[api_key]
|
||||
|
||||
ok = False
|
||||
try:
|
||||
import sounddevice as sd
|
||||
rate = _RATES.get(kind, 16000)
|
||||
secs = _PROBE_SECONDS.get(kind, 0.5)
|
||||
|
||||
# Each direction is probed the way main.py actually uses it. That is not
|
||||
# a detail: DirectSound input passes a callback stream and fails a
|
||||
# blocking read, so probing the wrong mode rejected a microphone that
|
||||
# works perfectly in the app.
|
||||
if kind == "output":
|
||||
# main.py writes with stream.write() — a real sink is rate-limited
|
||||
# by the hardware clock, a fake one swallows the buffer instantly.
|
||||
st = sd.RawOutputStream(samplerate=rate, channels=1, dtype="int16",
|
||||
blocksize=1024, device=idx)
|
||||
st.start()
|
||||
t0 = time.monotonic()
|
||||
st.write(bytes(int(rate * secs) * 2)) # silence — inaudible
|
||||
elapsed = time.monotonic() - t0
|
||||
st.stop(); st.close()
|
||||
ok = elapsed > secs * 0.5
|
||||
if not ok:
|
||||
print(f"[Audio] output: host API reports success but moves no "
|
||||
f"audio ({elapsed*1000:.0f} ms for {secs*1000:.0f} ms) "
|
||||
f"— skipping it")
|
||||
else:
|
||||
# main.py reads through a callback — count what arrives.
|
||||
frames = [0]
|
||||
|
||||
def _cb(indata, n, *_a):
|
||||
frames[0] += n
|
||||
|
||||
st = sd.InputStream(samplerate=rate, channels=1, dtype="int16",
|
||||
blocksize=1024, device=idx, callback=_cb)
|
||||
st.start()
|
||||
time.sleep(secs)
|
||||
st.stop(); st.close()
|
||||
ok = frames[0] > rate * secs * 0.3
|
||||
if not ok:
|
||||
print(f"[Audio] input: host API delivered {frames[0]} frames in "
|
||||
f"{secs*1000:.0f} ms — skipping it")
|
||||
except Exception as e:
|
||||
print(f"[Audio] {kind} transport probe failed: {e}")
|
||||
ok = False
|
||||
|
||||
_probe_results[api_key] = ok
|
||||
return ok
|
||||
|
||||
|
||||
def _display_name(name: str, devices) -> str:
|
||||
"""MME truncates device names to 31 characters, so the API that actually
|
||||
carries the audio may not be the one that can spell. If another host API
|
||||
knows a longer name that starts with this one, show that instead — the user
|
||||
reads 'Realtek HD Audio 2nd output (Realtek(R) Audio)' while the stream runs
|
||||
on the endpoint called 'Realtek HD Audio 2nd output (Re'."""
|
||||
if len(name) < 30:
|
||||
return name
|
||||
best = name
|
||||
for dev in devices:
|
||||
other = (dev.get("name") or "").strip()
|
||||
if len(other) > len(best) and other.startswith(name):
|
||||
best = other
|
||||
return best
|
||||
|
||||
# The rates the app opens its streams at. Defaults match main.py; main.py calls
|
||||
# configure() at startup with its own constants so the two can never drift apart
|
||||
# and silently reintroduce the bug above.
|
||||
_RATES = {"input": 16000, "output": 24000}
|
||||
|
||||
|
||||
def configure(input_rate: int, output_rate: int) -> None:
|
||||
"""Tell this module the sample rates the audio streams will use, so the
|
||||
picker can rule out devices that cannot be opened at them.
|
||||
|
||||
Drops any cached list: which devices are usable depends on the rate, so a
|
||||
list built under the old rates would be stale."""
|
||||
global _cache
|
||||
_RATES["input"] = int(input_rate)
|
||||
_RATES["output"] = int(output_rate)
|
||||
with _cache_lock:
|
||||
_cache = None
|
||||
|
||||
|
||||
def _usable(idx: int, kind: str) -> bool:
|
||||
"""Can this device actually be opened at the rate we need?
|
||||
|
||||
Deliberately opens a real stream rather than asking
|
||||
`check_output_settings`, because that function lies: it passed for an MME
|
||||
endpoint that then failed to open with "The specified format is not
|
||||
supported or cannot be translated" [MME error 32]. Opening and immediately
|
||||
closing costs milliseconds and is the only answer that holds."""
|
||||
st = None
|
||||
try:
|
||||
import sounddevice as sd
|
||||
rate = _RATES.get(kind, 16000)
|
||||
if kind == "input":
|
||||
st = sd.InputStream(samplerate=rate, channels=1, dtype="int16",
|
||||
blocksize=1024, device=idx,
|
||||
callback=lambda *_a: None)
|
||||
else:
|
||||
st = sd.RawOutputStream(samplerate=rate, channels=1, dtype="int16",
|
||||
blocksize=1024, device=idx)
|
||||
st.start()
|
||||
return True
|
||||
except Exception:
|
||||
return False
|
||||
finally:
|
||||
if st is not None:
|
||||
try:
|
||||
st.stop(); st.close()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
# Aliases for "the default device" and internal routing endpoints. Matched
|
||||
# case-insensitively as substrings against the device name.
|
||||
_PSEUDO_DEVICES = (
|
||||
"sound mapper", # Windows MME
|
||||
"primary sound", # Windows DirectSound ("Primary Sound Capture Driver")
|
||||
"sysdefault", # ALSA
|
||||
"default", # ALSA / PulseAudio alias
|
||||
"dmix", "dsnoop", # ALSA software mixing plugins
|
||||
"surround", # ALSA channel-layout permutations of one card
|
||||
"samplerate", "speexrate", "upmix", "vdownmix", "null",
|
||||
)
|
||||
|
||||
|
||||
def _is_pseudo(name: str) -> bool:
|
||||
low = name.lower()
|
||||
return any(tok in low for tok in _PSEUDO_DEVICES)
|
||||
|
||||
|
||||
def _query() -> dict[str, list[str]]:
|
||||
"""Return {'input': [names...], 'output': [names...]}. Never raises.
|
||||
|
||||
Only real, selectable devices — see the note above."""
|
||||
out: dict[str, list[str]] = {"input": [], "output": []}
|
||||
try:
|
||||
import platform
|
||||
import sounddevice as sd
|
||||
|
||||
devices = list(sd.query_devices())
|
||||
try:
|
||||
apis = [a.get("name", "") for a in sd.query_hostapis()]
|
||||
except Exception:
|
||||
apis = []
|
||||
|
||||
preferred = _PREFERRED_APIS.get(platform.system(), ())
|
||||
|
||||
def _collect(api_filter, kind) -> list[tuple[int, str]]:
|
||||
"""(index, name) for named, non-pseudo devices on one side that can
|
||||
be opened at the rate that side runs at."""
|
||||
chan = "max_input_channels" if kind == "input" else "max_output_channels"
|
||||
found, seen = [], set()
|
||||
for idx, dev in enumerate(devices):
|
||||
name = (dev.get("name") or "").strip()
|
||||
if not name or _is_pseudo(name) or name in seen:
|
||||
continue
|
||||
if dev.get(chan, 0) <= 0:
|
||||
continue
|
||||
if api_filter is not None:
|
||||
api = apis[dev["hostapi"]].lower() if dev.get("hostapi", -1) < len(apis) else ""
|
||||
if api_filter not in api:
|
||||
continue
|
||||
if not _usable(idx, kind):
|
||||
continue
|
||||
seen.add(name)
|
||||
found.append((idx, name))
|
||||
return found
|
||||
|
||||
# Each direction picks its own host API. They are genuinely different
|
||||
# problems — on Windows the microphone works on DirectSound while the
|
||||
# speakers only work on MME — and a single global choice cannot be right
|
||||
# for both.
|
||||
for kind in ("input", "output"):
|
||||
for api_filter in list(preferred) + [None]:
|
||||
found = _collect(api_filter, kind)
|
||||
if not found:
|
||||
continue
|
||||
# One probe per API per direction, cached, on this thread.
|
||||
if not _transport_works(found[0][0], kind, (api_filter, kind)):
|
||||
continue
|
||||
_chosen_api[kind] = api_filter
|
||||
out[kind] = [_display_name(n, devices) for _i, n in found]
|
||||
break
|
||||
if out[kind]:
|
||||
print(f"[Audio] {kind}: using "
|
||||
f"{_chosen_api[kind] or 'any host API'} "
|
||||
f"({len(out[kind])} devices)")
|
||||
return out
|
||||
|
||||
except Exception as e:
|
||||
print(f"[Audio] Device enumeration failed: {e}")
|
||||
return out
|
||||
|
||||
|
||||
def prefetch() -> None:
|
||||
"""Warm the cache on a background thread. Called once at startup so the
|
||||
settings drawer never pays for enumeration on the Qt thread."""
|
||||
def _work():
|
||||
global _cache
|
||||
result = _query()
|
||||
with _cache_lock:
|
||||
_cache = result
|
||||
print(f"[Audio] {len(result['input'])} input / "
|
||||
f"{len(result['output'])} output devices found")
|
||||
threading.Thread(target=_work, daemon=True, name="audio-devices").start()
|
||||
|
||||
|
||||
def list_devices(kind: str, refresh: bool = False) -> list[str]:
|
||||
"""Device names for 'input' or 'output'. Falls back to a synchronous query
|
||||
if the prefetch has not landed yet — correctness over the cache."""
|
||||
global _cache
|
||||
with _cache_lock:
|
||||
cached = None if refresh else _cache
|
||||
if cached is None:
|
||||
cached = _query()
|
||||
with _cache_lock:
|
||||
_cache = cached
|
||||
return list(cached.get(kind, []))
|
||||
|
||||
|
||||
def resolve(name: str, kind: str):
|
||||
"""Turn a saved device name into something sounddevice accepts.
|
||||
|
||||
Returns None for "system default" — which is also what we return when the
|
||||
saved device is gone, because a missing headset must degrade to the built-in
|
||||
speakers, not to a crash on startup.
|
||||
|
||||
Candidates are walked in the same host-API order the picker used, so a name
|
||||
the user chose from the WASAPI list resolves to the WASAPI endpoint. Without
|
||||
that ordering a full name would fall through to MME's truncated copy of the
|
||||
same device — which happens to work, but means the setting quietly refers to
|
||||
a different endpoint than the one on screen."""
|
||||
wanted = (name or "").strip()
|
||||
if not wanted or wanted == DEFAULT_LABEL:
|
||||
return None
|
||||
|
||||
try:
|
||||
import platform
|
||||
import sounddevice as sd
|
||||
|
||||
devices = list(sd.query_devices())
|
||||
try:
|
||||
apis = [a.get("name", "") for a in sd.query_hostapis()]
|
||||
except Exception:
|
||||
apis = []
|
||||
|
||||
want_in = (kind == "input")
|
||||
chan_key = "max_input_channels" if want_in else "max_output_channels"
|
||||
|
||||
def _candidates(api_filter):
|
||||
for idx, dev in enumerate(devices):
|
||||
if dev.get(chan_key, 0) <= 0:
|
||||
continue
|
||||
if api_filter is not None:
|
||||
api = apis[dev["hostapi"]].lower() if dev.get("hostapi", -1) < len(apis) else ""
|
||||
if api_filter not in api:
|
||||
continue
|
||||
yield idx, (dev.get("name") or "").strip()
|
||||
|
||||
# The API the picker settled on for this direction comes first — the
|
||||
# endpoint that was listed must be the endpoint that gets opened, or the
|
||||
# setting means something different from what it says. list_devices()
|
||||
# populates it; calling it here is a no-op once the cache is warm.
|
||||
list_devices(kind)
|
||||
chosen = _chosen_api.get(kind)
|
||||
orders = ([chosen] if chosen is not None else []) \
|
||||
+ [a for a in _PREFERRED_APIS.get(platform.system(), ()) if a != chosen] \
|
||||
+ [None]
|
||||
|
||||
# A candidate only counts if it can be opened at the rate this side runs
|
||||
# at. The prefix match matters because the API that carries the audio is
|
||||
# not always the one that can spell: MME truncates names to 31 characters
|
||||
# while DirectSound and WASAPI do not, so the name shown in the picker
|
||||
# can be longer than the name of the endpoint it actually opens.
|
||||
for api_filter in orders:
|
||||
partial = None
|
||||
for idx, dev_name in _candidates(api_filter):
|
||||
if dev_name == wanted:
|
||||
if _usable(idx, kind):
|
||||
return idx
|
||||
continue
|
||||
if partial is None and (dev_name.startswith(wanted[:24])
|
||||
or wanted.startswith(dev_name[:24])):
|
||||
if _usable(idx, kind):
|
||||
partial = idx
|
||||
if partial is not None:
|
||||
return partial
|
||||
|
||||
print(f"[Audio] Saved {kind} device '{wanted}' cannot be opened at "
|
||||
f"{_RATES.get(kind)} Hz on any host API — using system default")
|
||||
return None
|
||||
except Exception as e:
|
||||
print(f"[Audio] resolve({kind}) failed: {e} — using system default")
|
||||
return None
|
||||
+715
@@ -0,0 +1,715 @@
|
||||
"""
|
||||
Holographic AI head for the HUD centre — the thing that used to be a ring stack
|
||||
with the assistant's name in the middle.
|
||||
|
||||
Design notes
|
||||
------------
|
||||
* **The face is real human geometry.** `core.avatar_mesh` builds the head around
|
||||
MediaPipe's canonical face model, so eyelids, nostrils, lips and cheekbones
|
||||
are measured anatomy rather than fitted curves. This renderer's whole job is
|
||||
to light it, pose it and animate it.
|
||||
* **Software rendered, on purpose.** Everything is QPainter, so there is no
|
||||
OpenGL context, no shader compile, no GPU driver to disagree with us and no
|
||||
new pip dependency. It looks the same on a gaming rig, a 2013 laptop, a VM
|
||||
and a remote desktop session.
|
||||
* **Lip-sync comes from the audio pipeline, not from the avatar.** `main.py`
|
||||
already computes a real RMS level off the PCM (`_pcm_level`) for both the mic
|
||||
and JARVIS's own output. The avatar just consumes that number, so there is no
|
||||
second audio path to fall out of sync. The mouth only tracks the level while
|
||||
JARVIS is *speaking* — during listening the same level drives the aura, so the
|
||||
head never lip-syncs to the user's voice.
|
||||
|
||||
The renderer is theme-agnostic: `paint()` takes its colours as arguments, which
|
||||
is what lets the HueWheel accent picker retint the avatar for free.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
import random
|
||||
|
||||
import numpy as np
|
||||
from PyQt6.QtCore import QLineF, QPointF, QRectF, Qt
|
||||
from PyQt6.QtGui import QBrush, QColor, QPainter, QPen, QPolygonF, QRadialGradient
|
||||
|
||||
from core.avatar_mesh import JAW_MAX, JAW_PIVOT, get_head_mesh
|
||||
|
||||
# Perspective camera distance in head-half-heights. Large enough that the nose
|
||||
# does not balloon, small enough to keep a sense of depth.
|
||||
_CAM_D = 4.6
|
||||
|
||||
# Wireframe opacity buckets, so the whole lattice draws in a handful of batched
|
||||
# drawLines() calls instead of one call per line.
|
||||
_BUCKETS = 4
|
||||
_MIN_ALPHA = 0.05
|
||||
|
||||
# Resolution of the surface-shading colour ramp. Banding across a filled facet
|
||||
# is far more visible than banding in line alpha, so this is fine-grained — and
|
||||
# being a lookup it costs nothing per face.
|
||||
_LUT_N = 192
|
||||
|
||||
# How far the brows travel at full lift, in head-half-heights. Derived, not
|
||||
# tuned: the brow-to-eye gap is 0.198 and a real raise covers about a third of
|
||||
# it, then the drawn landmarks only carry half the rig weight.
|
||||
_BROW_LIFT = 0.14
|
||||
|
||||
# Mouth timing, as time constants in seconds rather than per-frame fractions.
|
||||
# A fixed per-frame lerp silently changes speed with the frame rate: the HUD
|
||||
# runs at 60 Hz here, throttles its paint to 30, and drops to 20 when idle, so
|
||||
# the same constant meant three different mouths. These do not.
|
||||
#
|
||||
# Shutting is the fastest of the three, and it is measured rather than chosen.
|
||||
# A short closure occupies a single 20 ms schedule frame, so the mouth has one
|
||||
# step to reach it: at 20 ms the jaw got a third of the way and the closure
|
||||
# vanished, at 12 ms it arrives, and going below that changes nothing because
|
||||
# the analysis window is then the limit, not the smoothing. Halving it doubled
|
||||
# the closures the mouth visibly makes across a test paragraph, 5 of 21 to 10,
|
||||
# with no loss of opening on the vowels. Only the return to rest, once talking
|
||||
# has actually stopped, is leisurely.
|
||||
_TAU_OPEN = 0.022 # jaw dropping toward a vowel
|
||||
_TAU_SHUT = 0.012 # lips closing on a consonant, mid-word
|
||||
_TAU_REST = 0.055 # settling back to rest after speech ends
|
||||
_TAU_SHAPE = 0.018 # viseme openness following the schedule
|
||||
|
||||
# Only the microphone path needs a level floor: it has one coarse RMS and no way
|
||||
# to tell speech from room tone. JARVIS's own voice arrives as a per-20 ms
|
||||
# schedule whose silences are already silent, so it needs no floor and must not
|
||||
# have one — a floor there swallows the gaps between words.
|
||||
_MIC_FLOOR = 0.14
|
||||
|
||||
# How far below this voice's own loud level counts as a closure: -20 dB, which
|
||||
# is what a stop consonant actually drops to. Expressed as a ratio so it holds
|
||||
# at any speaker volume.
|
||||
_CLOSE_FRAC = 0.10
|
||||
|
||||
|
||||
def _rate(dt: float, tau: float) -> float:
|
||||
"""Per-frame lerp factor for an exponential approach with time constant
|
||||
`tau`. Frame-rate independent: the motion takes the same wall-clock time at
|
||||
20, 30 or 60 fps, and a long frame catches up instead of stalling."""
|
||||
return 1.0 - math.exp(-dt / tau)
|
||||
|
||||
|
||||
def _c(col: QColor, a: float) -> QColor:
|
||||
"""Copy of `col` at alpha `a` (0-255, clamped)."""
|
||||
q = QColor(col)
|
||||
q.setAlpha(int(max(0.0, min(255.0, a))))
|
||||
return q
|
||||
|
||||
|
||||
def _blend(bg: QColor, col: QColor, a: float) -> QColor:
|
||||
"""`col` at alpha `a` pre-mixed onto `bg`, returned fully **opaque**.
|
||||
|
||||
Qt's raster engine has a fast path for opaque antialiased lines and a much
|
||||
slower blended path for everything else — measured at 1.0 ms versus 2.9 ms
|
||||
for the same 780 lines. The HUD paints a flat background behind the avatar,
|
||||
so mixing the alpha in by hand is visually equivalent and three times cheaper.
|
||||
"""
|
||||
f = max(0.0, min(1.0, a / 255.0))
|
||||
return QColor(int(bg.red() + (col.red() - bg.red()) * f),
|
||||
int(bg.green() + (col.green() - bg.green()) * f),
|
||||
int(bg.blue() + (col.blue() - bg.blue()) * f))
|
||||
|
||||
|
||||
class HoloAvatar:
|
||||
"""Animated holographic head. One instance per HUD canvas.
|
||||
|
||||
Lifecycle:
|
||||
av = HoloAvatar()
|
||||
av.step(dt, amp, speaking=..., muted=...) # once per tick
|
||||
av.paint(painter, cx, cy, r, primary, accent, bg) # once per frame
|
||||
"""
|
||||
|
||||
# Look: True paints a lit, solid head with a wireframe over it; False is a
|
||||
# see-through glass wireframe. Flip here, or per instance.
|
||||
shaded = True
|
||||
|
||||
def __init__(self) -> None:
|
||||
mesh = get_head_mesh()
|
||||
self._v0 = mesh["verts"]
|
||||
self._n0 = mesh["normals"]
|
||||
self._jaw = mesh["jaw"]
|
||||
self._brow_w = mesh["brow"]
|
||||
self._lips_w = mesh["lips"]
|
||||
self._lip_c = mesh["lip_centre"]
|
||||
self._fade = mesh["fade"]
|
||||
self._f = mesh["faces"]
|
||||
self._fgroup = mesh["face_group"]
|
||||
self._fa, self._fb, self._fc = (self._f[:, i] for i in range(3))
|
||||
self._e0 = mesh["edges"][:, 0]
|
||||
self._e1 = mesh["edges"][:, 1]
|
||||
self._lm = mesh["landmarks"]
|
||||
# The inner-lip ring runs lower-lip left→right, then upper-lip back.
|
||||
# Splitting it lets the upper arc anchor a strip of teeth, which is what
|
||||
# keeps an open mouth from reading as a hole punched in the face.
|
||||
lips_in = mesh["landmarks"]["lips_in"]
|
||||
self._lip_up = np.concatenate([lips_in[10:], lips_in[:1]])
|
||||
|
||||
# Crown (+1.0) down to the bottom of the neck, in head-half-heights.
|
||||
# Callers size the head to the room they have with this.
|
||||
self.SPAN = mesh["span"][0] - mesh["span"][1]
|
||||
|
||||
self._lut_cache: list = []
|
||||
self._lut_key = None
|
||||
|
||||
n = self._v0.shape[0]
|
||||
self._v = np.empty((n, 3), dtype=np.float32)
|
||||
|
||||
self._t = 0.0
|
||||
self._sway = 0.0 # integrated sway phase — see step()
|
||||
self._yaw = 0.0
|
||||
self._pitch = 0.0
|
||||
self._mouth = 0.0 # 0..1 smoothed jaw opening
|
||||
self._glow = 0.0 # 0..1 smoothed overall energy
|
||||
self._scan = -1.6 # vertical position of the energy sweep
|
||||
self._blink = 0.0 # 0 = open, 1 = shut
|
||||
self._blink_at = 3.0
|
||||
|
||||
# ── expression ──────────────────────────────────────────────────────
|
||||
# Speech is not just a moving jaw. Brows ride the loudness envelope,
|
||||
# eyes widen with the brows, and the gaze flicks between fixation
|
||||
# points — those three are what make it read as talking rather than as
|
||||
# a puppet chewing.
|
||||
self._amp_slow = 0.0
|
||||
self._expr = 0.0
|
||||
self._expr_tgt = 0.0
|
||||
self._expr_at = 0.0
|
||||
self._brow = 0.0 # smoothed brow lift, -0.4 .. 1.2
|
||||
self._emph = 0.0 # syllable emphasis, drives the head nod
|
||||
self._gaze = [0.0, 0.0]
|
||||
self._gaze_tgt = [0.0, 0.0]
|
||||
self._gaze_at = 0.0
|
||||
|
||||
# ── state expression ────────────────────────────────────────────────
|
||||
# The face is the fastest status indicator in the app: you read a gaze
|
||||
# before you read a word. Saccades orbit a bias that the assistant's
|
||||
# state moves — eyes off to the side while it thinks, back on you while
|
||||
# it listens, lids low while it sleeps.
|
||||
self._gaze_bias = [0.0, 0.0]
|
||||
self._bias_tgt = [0.0, 0.0]
|
||||
self._bias_at = 0.0
|
||||
self._lids = 1.0 # 1 = wide, 0 = shut; low while asleep
|
||||
self._brow_bias = 0.0 # concentration pulls the brows down
|
||||
self._glance = None # (dx, dy, until_t) — a deliberate look
|
||||
|
||||
# ── viseme ──────────────────────────────────────────────────────────
|
||||
# Loudness alone only answers "how far open", which is why an RMS-driven
|
||||
# mouth flaps rather than speaks. These two carry the *shape*: how open
|
||||
# the jaw is for this sound, and whether the lips are spread (/i/) or
|
||||
# rounded (/u/). They come from a formant read of the audio actually
|
||||
# being played — see `_pcm_visemes` in main.py.
|
||||
self._v_open = 1.0
|
||||
self._v_wide = 0.0
|
||||
self._wide = 0.0 # smoothed lip spread, -1 round .. +1 spread
|
||||
self._v_peak = 0.18 # running estimate of this voice's loud level
|
||||
|
||||
# ── animation ───────────────────────────────────────────────────────────
|
||||
|
||||
def _mouth_step(self, dt: float, amp: float, live: bool,
|
||||
v_open: float | None, v_level: float | None) -> None:
|
||||
"""One increment of the jaw. Called once per viseme frame while JARVIS
|
||||
speaks, once per rendered frame otherwise."""
|
||||
if v_open is None:
|
||||
shape = 1.0
|
||||
else:
|
||||
self._v_open += (v_open - self._v_open) * _rate(dt, _TAU_SHAPE)
|
||||
shape = self._v_open
|
||||
|
||||
if v_level is None:
|
||||
gated = max(0.0, (amp - _MIC_FLOOR) / (1.0 - _MIC_FLOOR))
|
||||
drive = (gated ** 0.6) * (shape ** 0.75)
|
||||
else:
|
||||
# Speech RMS spends most of its time well below full scale, so the
|
||||
# raw value alone would only ever half-open the jaw. Normalise it
|
||||
# against a running estimate of this voice's own loud level rather
|
||||
# than a constant: it then reads the same whether the user has the
|
||||
# volume low or the model happens to be speaking softly.
|
||||
self._v_peak = max(v_level, self._v_peak - dt * 0.55)
|
||||
ref = max(0.18, self._v_peak)
|
||||
# The floor is a fraction of this voice's own loud level, not a
|
||||
# fixed number, so it means the same thing at any volume and in any
|
||||
# language. A stop consonant drops 20 dB or more below the vowels
|
||||
# around it, which is this ratio — so a real closure lands at
|
||||
# exactly zero rather than at some small positive value the curve
|
||||
# would otherwise lift back up. That lift is what kept the mouth
|
||||
# from ever quite shutting between words.
|
||||
q = (v_level - _CLOSE_FRAC * ref) / (ref * (1.0 - _CLOSE_FRAC))
|
||||
drive = max(0.0, min(1.0, q)) ** 0.85 * (shape ** 0.75)
|
||||
|
||||
target = min(1.0, drive) if live else 0.0
|
||||
if target > self._mouth:
|
||||
tau = _TAU_OPEN
|
||||
elif live:
|
||||
tau = _TAU_SHUT # mid-word: a consonant, and it must shut now
|
||||
else:
|
||||
tau = _TAU_REST # speech is over; settle, don't snap
|
||||
self._mouth += (target - self._mouth) * _rate(dt, tau)
|
||||
if self._mouth < 0.002:
|
||||
self._mouth = 0.0
|
||||
|
||||
def step(self, dt: float, amp: float, speaking: bool = False,
|
||||
muted: bool = False, state: str = "",
|
||||
v_open: float | None = None, v_wide: float = 0.0,
|
||||
v_level: float | None = None,
|
||||
v_seq: list | None = None, v_hop: float = 0.02) -> None:
|
||||
"""Advance the animation.
|
||||
|
||||
`amp` is the 0..1 display audio level. `v_open` / `v_wide` / `v_level`
|
||||
are the viseme schedule's shape and true level for this instant; passing
|
||||
None falls back to loudness-only articulation, which is what the
|
||||
microphone path uses. `v_seq` is every schedule frame the last rendered
|
||||
frame spanned, so no closure is lost when the paint rate drops.
|
||||
"""
|
||||
dt = max(0.001, min(0.10, float(dt)))
|
||||
self._t += dt
|
||||
t = self._t
|
||||
amp = max(0.0, min(1.0, float(amp)))
|
||||
live = speaking and not muted
|
||||
|
||||
# Idle sway. The phase is *integrated* rather than taken as
|
||||
# sin(t * rate * speed): multiplying absolute time by a speed that
|
||||
# changes when JARVIS starts or stops talking jumps the phase by
|
||||
# t * rate * delta, which after a minute of uptime is several radians
|
||||
# and visibly teleports the head the instant a sentence ends.
|
||||
speed = (1.0 if not muted else 0.55) * (1.25 if live else 1.0)
|
||||
self._sway += dt * speed
|
||||
s = self._sway
|
||||
self._yaw = 0.26 * math.sin(s * 0.31) + 0.09 * math.sin(s * 0.73 + 1.3)
|
||||
self._pitch = (0.060 * math.sin(s * 0.23 + 0.7)
|
||||
+ 0.024 * math.sin(s * 0.61))
|
||||
|
||||
# Mouth. Which level is driving it matters more than any rate here.
|
||||
#
|
||||
# `v_level` is this 20 ms frame's own RMS, taken from the very audio
|
||||
# about to be heard, so its silences are real silences. `amp` is the
|
||||
# waveform display's level, and that one is a *peak hold*: it keeps the
|
||||
# loudest value it has seen and decays gently, on purpose, so the bars
|
||||
# do not stutter between audio chunks. Driving a mouth from a peak hold
|
||||
# is why the gaps between words never closed — the hold spans exactly
|
||||
# the consonant it was supposed to reveal. So the schedule drives the
|
||||
# jaw whenever there is one, and `amp` is left to the microphone path,
|
||||
# which has nothing better.
|
||||
# Advance the mouth once per *schedule* frame rather than once per
|
||||
# rendered frame. A bilabial closure lasts around 40 ms — two frames of
|
||||
# a 50 Hz schedule — and the HUD throttles its paint to 30 fps and to 20
|
||||
# when idle. Point-sampling at 20 fps steps 50 ms at a time, so a whole
|
||||
# closure can fall between two samples and simply never be seen; that is
|
||||
# information loss no smoothing constant can recover. Sub-stepping costs
|
||||
# a few float operations per frame and makes the mouth identical at 20,
|
||||
# 30 and 60 fps.
|
||||
# An empty list is meaningful and is not the same as None: it says a
|
||||
# schedule is playing but this tick landed inside a frame already
|
||||
# spoken. The mouth's clock is the schedule's, so the right thing then
|
||||
# is to do nothing. Re-stepping the same frame — which is what a
|
||||
# truthiness test here would do — advances the jaw twice for one 20 ms
|
||||
# of audio, and at 60 fps that alone made the mouth behave differently
|
||||
# than at 20.
|
||||
if v_seq is not None:
|
||||
for lv, op, _wd in v_seq:
|
||||
self._mouth_step(v_hop, amp, live, op, lv)
|
||||
else:
|
||||
self._mouth_step(dt, amp, live, v_open, v_level)
|
||||
|
||||
# Syllable emphasis. Applied unconditionally: `_emph` decays to zero on
|
||||
# its own once the mouth closes, whereas gating it on `live` deleted the
|
||||
# whole offset in a single frame and snapped the head at sentence end.
|
||||
self._emph += (self._mouth - self._emph) * _rate(
|
||||
dt, 0.055 if self._mouth > self._emph else 0.32)
|
||||
self._pitch -= self._emph * 0.028
|
||||
self._yaw += 0.018 * math.sin(t * 1.7) * self._emph
|
||||
|
||||
# Loudness envelope, deliberately lazier than the mouth: brows track the
|
||||
# shape of a phrase, not individual syllables.
|
||||
env = amp if live else 0.0
|
||||
self._amp_slow += (env - self._amp_slow) * _rate(
|
||||
dt, 0.16 if env > self._amp_slow else 0.36)
|
||||
|
||||
if live:
|
||||
if t >= self._expr_at:
|
||||
self._expr_tgt = random.uniform(-0.35, 1.0)
|
||||
self._expr_at = t + 1.1 + 2.0 * random.random()
|
||||
else:
|
||||
self._expr_tgt = 0.0
|
||||
self._expr_at = t + 0.8
|
||||
self._expr += (self._expr_tgt - self._expr) * 0.075
|
||||
|
||||
brow_t = 0.55 * self._amp_slow + 0.60 * self._expr + self._brow_bias
|
||||
self._brow += (max(-0.4, min(1.2, brow_t)) - self._brow) * 0.20
|
||||
|
||||
# ── what the state does to the face ─────────────────────────────────
|
||||
st = (state or "").upper()
|
||||
thinking = st in ("THINKING", "PROCESSING")
|
||||
asleep = st in ("SLEEPING", "STANDBY", "OFFLINE")
|
||||
|
||||
if thinking:
|
||||
# People look away to think, and hold it. The direction re-rolls
|
||||
# slowly so it reads as thought rather than as scanning.
|
||||
if t >= self._bias_at:
|
||||
self._bias_tgt = [random.choice((-1.0, 1.0)) * random.uniform(0.45, 0.8),
|
||||
random.uniform(0.25, 0.55)]
|
||||
self._bias_at = t + 1.4 + 1.6 * random.random()
|
||||
brow_bias, lid_tgt = -0.28, 0.94
|
||||
elif asleep:
|
||||
self._bias_tgt = [0.0, -0.25]
|
||||
brow_bias, lid_tgt = -0.05, 0.22
|
||||
else:
|
||||
# LISTENING / idle / speaking: eyes come back to the user.
|
||||
self._bias_tgt = [0.0, 0.0]
|
||||
self._bias_at = 0.0
|
||||
brow_bias = 0.10 if st == "LISTENING" else 0.0
|
||||
lid_tgt = 1.0
|
||||
|
||||
for i in (0, 1):
|
||||
self._gaze_bias[i] += (self._bias_tgt[i] - self._gaze_bias[i]) * 0.06
|
||||
self._lids += (lid_tgt - self._lids) * 0.08
|
||||
self._brow_bias += (brow_bias - self._brow_bias) * 0.06
|
||||
|
||||
# Gaze: saccades are near-instant jumps between fixations, and they get
|
||||
# more frequent when there is something to say. While thinking they slow
|
||||
# right down — a darting eye reads as nervous, not thoughtful.
|
||||
if t >= self._gaze_at:
|
||||
reach = 0.9 if live else (0.35 if thinking else 0.55)
|
||||
self._gaze_tgt = [random.uniform(-1.0, 1.0) * reach,
|
||||
random.uniform(-1.0, 1.0) * reach * 0.55]
|
||||
if live:
|
||||
self._gaze_at = t + 0.55 + 1.7 * random.random()
|
||||
elif thinking:
|
||||
self._gaze_at = t + 1.8 + 2.4 * random.random()
|
||||
else:
|
||||
self._gaze_at = t + 1.3 + 2.8 * random.random()
|
||||
|
||||
# A deliberate glance (something appeared on screen) overrides the
|
||||
# wandering for a moment, then hands control back.
|
||||
if self._glance is not None:
|
||||
gx, gy, until = self._glance
|
||||
if t < until:
|
||||
self._gaze_tgt = [gx, gy]
|
||||
else:
|
||||
self._glance = None
|
||||
|
||||
for i, b in enumerate(self._gaze_bias):
|
||||
tgt = max(-1.0, min(1.0, self._gaze_tgt[i] + b))
|
||||
self._gaze[i] += (tgt - self._gaze[i]) * 0.30
|
||||
|
||||
# Lips lead the jaw slightly in real speech, so they track a touch
|
||||
# faster; they also relax to neutral the moment the voice stops.
|
||||
wide_t = v_wide if (live and v_open is not None) else 0.0
|
||||
self._wide += (max(-1.0, min(1.0, wide_t)) - self._wide) * _rate(dt, 0.030)
|
||||
|
||||
self._glow += ((0.0 if muted else amp) - self._glow) * (
|
||||
0.35 if (0.0 if muted else amp) > self._glow else 0.10)
|
||||
|
||||
self._scan += dt * (0.55 + 1.5 * self._glow)
|
||||
if self._scan > 1.35:
|
||||
self._scan = -1.75
|
||||
|
||||
if self._blink > 0.0:
|
||||
self._blink = max(0.0, self._blink - dt * 8.5)
|
||||
elif t >= self._blink_at:
|
||||
# Concentration suppresses blinking; a sleeping face has no need of
|
||||
# it at all, since the lids are already down.
|
||||
if asleep:
|
||||
self._blink_at = t + 6.0
|
||||
else:
|
||||
self._blink = 1.0
|
||||
gap = 5.5 if thinking else 3.4
|
||||
self._blink_at = t + gap + 3.1 * random.random()
|
||||
|
||||
def glance(self, dx: float, dy: float, hold: float = 1.1) -> None:
|
||||
"""Look deliberately somewhere for `hold` seconds, then wander again.
|
||||
|
||||
Used when something appears on screen: a face that looks at what just
|
||||
showed up tells the user it landed, without a word being spoken.
|
||||
"""
|
||||
self._glance = (max(-1.0, min(1.0, float(dx))),
|
||||
max(-1.0, min(1.0, float(dy))),
|
||||
self._t + max(0.1, float(hold)))
|
||||
|
||||
# ── posing ──────────────────────────────────────────────────────────────
|
||||
|
||||
def _pose(self):
|
||||
"""Jaw drop, brow lift and head rotation, applied to the real geometry."""
|
||||
v = self._v
|
||||
np.copyto(v, self._v0)
|
||||
|
||||
if self._brow > 0.004 or self._brow < -0.004:
|
||||
# The brow-to-eye gap is 0.198 head-half-heights and a real raise
|
||||
# moves a third of it. The old 0.045 — halved again by the landmark
|
||||
# weights, which average 0.5 — worked out to six pixels on a 250 px
|
||||
# head, which is to say invisible.
|
||||
v[:, 1] += self._brow_w * (self._brow * _BROW_LIFT)
|
||||
|
||||
if abs(self._wide) > 0.01 and self._mouth > 0.0:
|
||||
# Spread pulls the corners out and flattens the lips back; rounding
|
||||
# draws them in and pushes them forward into a purse.
|
||||
k = self._lips_w * (self._wide * self._mouth)
|
||||
v[:, 0] += k * (v[:, 0] - self._lip_c[0]) * 0.55
|
||||
v[:, 1] += k * (v[:, 1] - self._lip_c[1]) * 0.30
|
||||
v[:, 2] -= k * 0.055
|
||||
|
||||
if self._mouth > 0.004:
|
||||
px, py, pz = JAW_PIVOT
|
||||
ang = self._jaw * (self._mouth * JAW_MAX)
|
||||
ca, sa = np.cos(ang), np.sin(ang)
|
||||
dy = v[:, 1] - py
|
||||
dz = v[:, 2] - pz
|
||||
v[:, 1] = py + dy * ca - dz * sa
|
||||
v[:, 2] = pz + dy * sa + dz * ca
|
||||
|
||||
cy, sy = math.cos(self._yaw), math.sin(self._yaw)
|
||||
cp, sp = math.cos(self._pitch), math.sin(self._pitch)
|
||||
m = np.array([
|
||||
[cy, 0.0, sy],
|
||||
[sp * sy, cp, -sp * cy],
|
||||
[-cp * sy, sp, cp * cy],
|
||||
], dtype=np.float32)
|
||||
|
||||
return v @ m.T, self._n0 @ m.T
|
||||
|
||||
# ── rendering ───────────────────────────────────────────────────────────
|
||||
|
||||
def _lut(self, bg: QColor, primary: QColor):
|
||||
"""Cached ramp of opaque surface brushes from `bg` to `primary`."""
|
||||
key = (bg.rgb(), primary.rgb())
|
||||
if self._lut_key != key:
|
||||
self._lut_cache = [QBrush(_blend(bg, primary, 255.0 * (i + 0.5) / _LUT_N))
|
||||
for i in range(_LUT_N)]
|
||||
self._lut_key = key
|
||||
return self._lut_cache
|
||||
|
||||
def paint(self, p: QPainter, cx: float, cy: float, r: float,
|
||||
primary: QColor, accent: QColor, bg: QColor | None = None) -> None:
|
||||
"""Draw the avatar with its head centre at (cx, cy).
|
||||
|
||||
`r` is the head's half-height in pixels — the caller owns the layout, so
|
||||
the HUD can fit the head to whatever room the status line leaves it.
|
||||
"""
|
||||
if bg is None:
|
||||
bg = QColor(0, 0, 0)
|
||||
amp = self._glow
|
||||
verts, norms = self._pose()
|
||||
|
||||
# ── aura ────────────────────────────────────────────────────────────
|
||||
ar = r * 1.95
|
||||
grad = QRadialGradient(cx, cy, ar)
|
||||
grad.setColorAt(0.00, _c(primary, 34 + 66 * amp))
|
||||
grad.setColorAt(0.38, _c(primary, 20 + 40 * amp))
|
||||
grad.setColorAt(1.00, _c(primary, 0))
|
||||
p.setPen(Qt.PenStyle.NoPen)
|
||||
p.setBrush(QBrush(grad))
|
||||
p.drawEllipse(QRectF(cx - ar, cy - ar, ar * 2, ar * 2))
|
||||
|
||||
# ── project ─────────────────────────────────────────────────────────
|
||||
w = _CAM_D - verts[:, 2]
|
||||
np.maximum(w, 0.35, out=w)
|
||||
k = (_CAM_D / w) * r
|
||||
xs = cx + verts[:, 0] * k
|
||||
ys = cy - verts[:, 1] * k
|
||||
|
||||
if self.shaded:
|
||||
self._paint_surface(p, xs, ys, norms, verts, primary, bg, amp)
|
||||
self._paint_wire(p, xs, ys, norms, verts, primary, bg, amp)
|
||||
self._paint_features(p, xs, ys, norms, r, primary, accent, bg, amp)
|
||||
|
||||
def _paint_surface(self, p: QPainter, xs, ys, norms, verts,
|
||||
primary: QColor, bg: QColor, amp: float) -> None:
|
||||
"""Fill the camera-facing triangles so the head reads as a lit volume."""
|
||||
a, b, c = self._fa, self._fb, self._fc
|
||||
|
||||
# Flat normals, taken from each triangle's own posed geometry — NOT the
|
||||
# averaged vertex normals. Averaging smears the nose, lips and brow
|
||||
# relief into their neighbours and renders the face as a blank egg;
|
||||
# per-facet normals are exactly what makes the anatomy visible.
|
||||
fn = np.cross(verts[b] - verts[a], verts[c] - verts[a])
|
||||
fn /= np.maximum(np.linalg.norm(fn, axis=1, keepdims=True), 1e-9)
|
||||
|
||||
# Point them outwards by agreeing with the vertex normals, which were
|
||||
# oriented at build time. Flipping on the sign of n_z instead would
|
||||
# negate x and y as well and scramble the lighting into moiré.
|
||||
ref = norms[a] + norms[b] + norms[c]
|
||||
fn *= np.sign((fn * ref).sum(1))[:, None]
|
||||
|
||||
nz = fn[:, 2]
|
||||
area = np.abs((xs[b] - xs[a]) * (ys[c] - ys[a])
|
||||
- (xs[c] - xs[a]) * (ys[b] - ys[a]))
|
||||
vis = np.flatnonzero((nz > 0.015) & (area > 3.0))
|
||||
if vis.size == 0:
|
||||
return
|
||||
fn = fn[vis]
|
||||
nz = nz[vis]
|
||||
|
||||
ax, ay = xs[a][vis], ys[a][vis]
|
||||
bx, by = xs[b][vis], ys[b][vis]
|
||||
cxx, cyy = xs[c][vis], ys[c][vis]
|
||||
|
||||
# A rim term for the glass edge plus a key light high on the left. The
|
||||
# light leans off-axis on purpose: weight it towards the camera and
|
||||
# every front-facing facet returns the same value, which is a flat mask.
|
||||
fres = np.clip(1.0 - nz, 0.0, 2.0) ** 1.7
|
||||
lam = np.clip(fn[:, 0] * -0.55 + fn[:, 1] * 0.50 + nz * 0.52, 0.0, 1.0)
|
||||
bright = 0.26 + 0.20 * fres + 0.66 * lam ** 1.05
|
||||
bright *= (self._fade[a][vis] + self._fade[b][vis] + self._fade[c][vis]) / 3.0
|
||||
bright *= 0.88 + 0.24 * amp
|
||||
|
||||
idx = np.clip((bright * _LUT_N).astype(np.int32), 0, _LUT_N - 1)
|
||||
|
||||
# Far facets first: the neck passes behind the jaw and the head is not
|
||||
# convex around the chin.
|
||||
# Sort far-to-near, but group first: neck facets all draw before head
|
||||
# facets, because the two meshes interpenetrate and a pure depth sort
|
||||
# interleaves them into a torn seam.
|
||||
fz = (verts[a, 2][vis] + verts[b, 2][vis] + verts[c, 2][vis]) * (1.0 / 3.0)
|
||||
order = np.argsort(self._fgroup[vis] * 1000.0 + fz, kind="stable")
|
||||
tris = np.stack([ax, ay, bx, by, cxx, cyy], axis=1)[order].tolist()
|
||||
shade = idx[order].tolist()
|
||||
lut = self._lut(bg, primary)
|
||||
|
||||
# Aliased fills: adjacent antialiased polygons leave hairline seams, and
|
||||
# the interior of a tiled surface has no silhouette worth smoothing —
|
||||
# the antialiased wireframe drawn afterwards covers the outline.
|
||||
p.setRenderHint(QPainter.RenderHint.Antialiasing, False)
|
||||
p.setPen(Qt.PenStyle.NoPen)
|
||||
for q, sh in zip(tris, shade):
|
||||
p.setBrush(lut[sh])
|
||||
p.drawPolygon(QPolygonF([QPointF(q[0], q[1]), QPointF(q[2], q[3]),
|
||||
QPointF(q[4], q[5])]))
|
||||
p.setRenderHint(QPainter.RenderHint.Antialiasing, True)
|
||||
|
||||
def _paint_wire(self, p: QPainter, xs, ys, norms, verts,
|
||||
primary: QColor, bg: QColor, amp: float) -> None:
|
||||
nz = norms[:, 2]
|
||||
fres = np.abs(1.0 - np.abs(nz)) ** 1.5
|
||||
if self.shaded:
|
||||
# The lit surface underneath is opaque, so back-facing edges would
|
||||
# float on top of the face — cull them and let the wire read as
|
||||
# structure lines over skin.
|
||||
front = nz > -0.05
|
||||
va = np.where(front, 0.10 + 0.42 * fres, 0.0)
|
||||
va += 0.30 * np.exp(-((verts[:, 1] - self._scan) / 0.13) ** 2) * front
|
||||
else:
|
||||
va = np.where(nz < 0.0, 0.13 + 0.26 * fres, 0.28 + 0.72 * fres)
|
||||
va += 0.42 * np.exp(-((verts[:, 1] - self._scan) / 0.13) ** 2)
|
||||
va *= self._fade * (0.80 + 0.45 * amp)
|
||||
|
||||
ea = 0.5 * (va[self._e0] + va[self._e1])
|
||||
keep = np.flatnonzero(ea > _MIN_ALPHA)
|
||||
if keep.size == 0:
|
||||
return
|
||||
|
||||
# Sort by opacity bucket once so each bucket is a contiguous *slice* of
|
||||
# one QLineF list; masking and rebuilding per bucket cost more than the
|
||||
# drawing itself.
|
||||
bucket = np.clip((ea[keep] * _BUCKETS).astype(np.int32), 0, _BUCKETS - 1)
|
||||
order = np.argsort(bucket, kind="stable")
|
||||
keep = keep[order]
|
||||
bounds = np.searchsorted(bucket[order], np.arange(_BUCKETS + 1))
|
||||
|
||||
e0, e1 = self._e0[keep], self._e1[keep]
|
||||
quad = np.stack([xs[e0], ys[e0], xs[e1], ys[e1]], axis=1).tolist()
|
||||
lines = [QLineF(q[0], q[1], q[2], q[3]) for q in quad]
|
||||
|
||||
skin = _blend(bg, primary, 132) if self.shaded else bg
|
||||
p.setBrush(Qt.BrushStyle.NoBrush)
|
||||
for b in range(_BUCKETS):
|
||||
lo, hi = int(bounds[b]), int(bounds[b + 1])
|
||||
if hi <= lo:
|
||||
continue
|
||||
seg = lines[lo:hi]
|
||||
a = 255.0 * min(1.0, (b + 0.5) / _BUCKETS)
|
||||
if self.shaded:
|
||||
# These sit on lit skin, so pre-mix against a representative
|
||||
# *skin* tone rather than the background — same fast opaque
|
||||
# path, and the lines still read as highlights over the face.
|
||||
p.setPen(QPen(_blend(skin, primary, a * 0.75), 1.0))
|
||||
else:
|
||||
p.setPen(QPen(_blend(bg, primary, a), 1.0))
|
||||
p.drawLines(seg)
|
||||
|
||||
# ── face ────────────────────────────────────────────────────────────────
|
||||
|
||||
def _ring(self, xs, ys, idx) -> QPolygonF:
|
||||
return QPolygonF([QPointF(float(x), float(y))
|
||||
for x, y in zip(xs[idx], ys[idx])])
|
||||
|
||||
def _paint_features(self, p: QPainter, xs, ys, norms, r: float,
|
||||
primary: QColor, accent: QColor, bg: QColor,
|
||||
amp: float) -> None:
|
||||
"""Eyes, brows and the mouth cavity, drawn from the real landmark rings.
|
||||
|
||||
The canonical model's eyes and lips are closed skin — the geometry gives
|
||||
the *shape* of the lids and mouth but no opening, so the openings are
|
||||
painted here, exactly on the landmarks that bound them.
|
||||
"""
|
||||
face = max(0.0, math.cos(self._yaw) * math.cos(self._pitch)) ** 2
|
||||
if face < 0.02:
|
||||
return
|
||||
|
||||
lm = self._lm
|
||||
vis = 1.0 - self._blink
|
||||
|
||||
# ── eyes ────────────────────────────────────────────────────────────
|
||||
for key in ("eye_l", "eye_r"):
|
||||
idx = lm[key]
|
||||
ex, ey = xs[idx], ys[idx]
|
||||
mid_y = float(ey.mean())
|
||||
if vis < 0.999:
|
||||
ey = mid_y + (ey - mid_y) * max(0.04, vis)
|
||||
poly = QPolygonF([QPointF(float(a), float(b)) for a, b in zip(ex, ey)])
|
||||
|
||||
p.setPen(Qt.PenStyle.NoPen)
|
||||
p.setBrush(QBrush(_blend(bg, primary, 22))) # socket shadow
|
||||
p.drawPolygon(poly)
|
||||
p.setBrush(Qt.BrushStyle.NoBrush)
|
||||
p.setPen(QPen(_c(primary, 210 * face), 1.3)) # lid line
|
||||
p.drawPolygon(poly)
|
||||
|
||||
if vis > 0.35:
|
||||
br = poly.boundingRect()
|
||||
gx = br.center().x() + self._gaze[0] * br.width() * 0.16
|
||||
gy = br.center().y() + self._gaze[1] * br.height() * 0.20
|
||||
cpt = QPointF(gx, gy)
|
||||
rad = min(br.height() * 0.62, br.width() * 0.20)
|
||||
p.setPen(Qt.PenStyle.NoPen)
|
||||
p.setBrush(QBrush(_c(accent, (70 + 60 * amp) * face * vis)))
|
||||
p.drawEllipse(cpt, rad, rad * vis) # iris
|
||||
p.setBrush(QBrush(_c(accent, 245 * face * vis)))
|
||||
p.drawEllipse(cpt, rad * 0.42, rad * 0.42 * vis) # pupil
|
||||
|
||||
# ── brows ───────────────────────────────────────────────────────────
|
||||
p.setBrush(Qt.BrushStyle.NoBrush)
|
||||
p.setPen(QPen(_c(primary, 150 * face), 1.7))
|
||||
for key in ("brow_l", "brow_r"):
|
||||
idx = lm[key]
|
||||
p.drawPolyline(self._ring(xs, ys, idx))
|
||||
|
||||
# ── mouth ───────────────────────────────────────────────────────────
|
||||
inner = self._ring(xs, ys, lm["lips_in"])
|
||||
open_h = inner.boundingRect().height()
|
||||
|
||||
p.setPen(Qt.PenStyle.NoPen)
|
||||
if self._mouth > 0.02:
|
||||
# The cavity is dark but never pure black — a black oval on a glowing
|
||||
# head reads as a hole, not a mouth. Tinting it with the theme keeps
|
||||
# it part of the hologram.
|
||||
p.setBrush(QBrush(_blend(bg, primary, 16 + 26 * self._mouth)))
|
||||
p.drawPolygon(inner)
|
||||
|
||||
# Upper teeth: a bright strip hanging from the upper lip. It is the
|
||||
# single cheapest thing that makes an open mouth look like speech.
|
||||
ux, uy = xs[self._lip_up], ys[self._lip_up]
|
||||
th = open_h * 0.30
|
||||
pts = [QPointF(float(x), float(y)) for x, y in zip(ux, uy)]
|
||||
pts += [QPointF(float(x), float(y) + th)
|
||||
for x, y in zip(ux[::-1], uy[::-1])]
|
||||
p.setBrush(QBrush(_blend(bg, primary, 150 + 60 * self._mouth)))
|
||||
p.drawPolygon(QPolygonF(pts))
|
||||
|
||||
# A warm pool at the back of the throat, strongest when wide open.
|
||||
p.setBrush(QBrush(_c(accent, 40 * self._mouth * face)))
|
||||
p.drawPolygon(inner)
|
||||
|
||||
p.setBrush(Qt.BrushStyle.NoBrush)
|
||||
p.setPen(QPen(_c(primary, (150 + 70 * self._mouth) * face), 1.3))
|
||||
p.drawPolygon(inner) # lip edge
|
||||
p.setPen(QPen(_c(primary, 110 * face), 1.1))
|
||||
p.drawPolygon(self._ring(xs, ys, lm["lips_out"]))
|
||||
@@ -0,0 +1,351 @@
|
||||
"""
|
||||
Human head mesh for the HUD avatar.
|
||||
|
||||
The face is **real measured human geometry** — MediaPipe's canonical face model
|
||||
(`core/face_model.obj`, Apache-2.0, 468 vertices / 898 triangles), which carries
|
||||
actual eyelids, nostrils, lips and cheekbones. Everything a formula cannot give
|
||||
you comes from there.
|
||||
|
||||
Earlier revisions of this file generated the whole head procedurally from an
|
||||
ellipsoid pushed around by gaussians. It could be tuned endlessly and still read
|
||||
as an egg with a face drawn on it, because there was no human anatomy in it —
|
||||
only smooth blobs. Real topology fixed in one step what parameter tweaking could
|
||||
not fix at all.
|
||||
|
||||
What is still generated here, around that face:
|
||||
* the cranium — the model is an open mask, so its 36-vertex border is swept
|
||||
back and up over a skull-shaped ellipsoid and closed at the occiput;
|
||||
* a tapering neck stub that fades out instead of needing shoulders;
|
||||
* vertex normals, jaw-rig weights, a thinned wireframe, and the landmark
|
||||
index rings (eyes, brows, lips) the renderer animates.
|
||||
|
||||
Coordinate system after normalisation (head-local, right-handed):
|
||||
+x → viewer's right +y → up +z → out of the face
|
||||
y = +1.0 crown, y = -1.0 chin, eyes land on y ≈ 0.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import collections
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
_OBJ = Path(__file__).resolve().parent / "face_model.obj"
|
||||
|
||||
# Cranium shape, in the model's own units (chin ≈ -9.4, forehead ≈ +8.3).
|
||||
# Tuned so that brow→crown is ~0.36 of the head's height, which is the real
|
||||
# proportion; a taller cranium than that immediately reads as a long face even
|
||||
# though the face itself is untouched measured geometry.
|
||||
_SKULL_C = (0.0, 2.0, -1.0) # centre of the cranial ellipsoid
|
||||
_SKULL_R = (8.4, 12.4, 8.2) # its radii
|
||||
_SKULL_POLE = (0.0, 0.42, -1.0) # direction of the occiput, where the sweep closes
|
||||
_SKULL_RINGS = 6
|
||||
_SKULL_BLEND = 1.7 # how fast the sweep leaves the face border
|
||||
_SKULL_BULGE = 1.04
|
||||
|
||||
_NECK_RINGS, _NECK_SEGS = 9, 14
|
||||
_NECK_Z = -1.6 # the neck tube's axis, in model units
|
||||
_WIRE_STRIDE = 3 # keep every n-th edge; the surface carries the form
|
||||
|
||||
# MediaPipe landmark rings. Verified against the geometry at build time — see
|
||||
# `_check_landmarks` — so a wrong index can never silently animate the cheek.
|
||||
LANDMARKS: dict[str, list[int]] = {
|
||||
"eye_l": [33, 7, 163, 144, 145, 153, 154, 155, 133, 173, 157, 158, 159,
|
||||
160, 161, 246],
|
||||
"eye_r": [263, 249, 390, 373, 374, 380, 381, 382, 362, 398, 384, 385, 386,
|
||||
387, 388, 466],
|
||||
"brow_l": [70, 63, 105, 66, 107],
|
||||
"brow_r": [300, 293, 334, 296, 336],
|
||||
"lips_out": [61, 146, 91, 181, 84, 17, 314, 405, 321, 375, 291, 409, 270,
|
||||
269, 267, 0, 37, 39, 40, 185],
|
||||
"lips_in": [78, 95, 88, 178, 87, 14, 317, 402, 318, 324, 308, 415, 310,
|
||||
311, 312, 13, 82, 81, 80, 191],
|
||||
}
|
||||
|
||||
# Jaw rig, in normalised units. The pivot sits between the ears, which is where
|
||||
# a real mandible hinges.
|
||||
JAW_PIVOT = (0.0, 0.06, -0.34)
|
||||
JAW_MAX = 0.115 # radians of drop at full amplitude (~6.6°)
|
||||
# Speech barely moves a real jaw, and a talking head is watched at HUD size
|
||||
# where a small, precise mouth reads better than a large one. The lip rig
|
||||
# (spread / round) now carries most of the articulation, so the jaw does not
|
||||
# have to swing to show that something is being said.
|
||||
|
||||
|
||||
def _load_obj(path: Path):
|
||||
verts, faces = [], []
|
||||
for line in path.read_text(encoding="utf-8").splitlines():
|
||||
if line.startswith("v "):
|
||||
verts.append([float(x) for x in line.split()[1:4]])
|
||||
elif line.startswith("f "):
|
||||
faces.append([int(t.split("/")[0]) - 1 for t in line.split()[1:4]])
|
||||
return np.array(verts, dtype=np.float64), np.array(faces, dtype=np.int64)
|
||||
|
||||
|
||||
def _boundary_loop(faces: np.ndarray) -> np.ndarray:
|
||||
"""Ordered ring of vertices along the open border of a triangle mesh."""
|
||||
seen = collections.Counter()
|
||||
for a, b, c in faces:
|
||||
for e in ((a, b), (b, c), (c, a)):
|
||||
seen[(min(e), max(e))] += 1
|
||||
border = [e for e, n in seen.items() if n == 1]
|
||||
|
||||
adj = collections.defaultdict(list)
|
||||
for a, b in border:
|
||||
adj[a].append(b)
|
||||
adj[b].append(a)
|
||||
|
||||
start = border[0][0]
|
||||
loop, prev, cur = [start], None, start
|
||||
while True:
|
||||
nxt = [v for v in adj[cur] if v != prev]
|
||||
if not nxt or nxt[0] == start:
|
||||
break
|
||||
prev, cur = cur, nxt[0]
|
||||
loop.append(cur)
|
||||
return np.array(loop)
|
||||
|
||||
|
||||
def _slerp(a: np.ndarray, b: np.ndarray, t):
|
||||
dot = np.clip((a * b).sum(-1, keepdims=True), -1.0, 1.0)
|
||||
om = np.arccos(dot)
|
||||
so = np.sin(om)
|
||||
safe = np.where(so < 1e-6, 1.0, so)
|
||||
out = np.where(so < 1e-6, a * (1 - t) + b * t,
|
||||
(np.sin((1 - t) * om) / safe) * a + (np.sin(t * om) / safe) * b)
|
||||
return out / np.maximum(np.linalg.norm(out, axis=-1, keepdims=True), 1e-9)
|
||||
|
||||
|
||||
def _add_cranium(verts: np.ndarray, faces: np.ndarray):
|
||||
"""Sweep the mask's open border back over a skull and close it at the occiput."""
|
||||
loop = _boundary_loop(faces)
|
||||
|
||||
# Orient the loop so the generated triangles wind the same way as the face's.
|
||||
centre2d = verts[loop, :2].mean(0)
|
||||
ang = np.arctan2(verts[loop, 1] - centre2d[1], verts[loop, 0] - centre2d[0])
|
||||
if np.diff(np.unwrap(ang)).sum() < 0:
|
||||
loop = loop[::-1]
|
||||
|
||||
n = len(loop)
|
||||
C = np.array(_SKULL_C)
|
||||
R = np.array(_SKULL_R)
|
||||
pole = np.array(_SKULL_POLE)
|
||||
pole = pole / np.linalg.norm(pole)
|
||||
chin_y = verts[:, 1].min()
|
||||
|
||||
rim = verts[loop] - C
|
||||
rim_r = np.linalg.norm(rim, axis=1, keepdims=True)
|
||||
rim_d = rim / rim_r
|
||||
|
||||
def ell_r(d):
|
||||
return 1.0 / np.sqrt(((d / R) ** 2).sum(-1, keepdims=True))
|
||||
|
||||
out_v = [verts]
|
||||
out_f = list(faces)
|
||||
prev_idx = loop
|
||||
ts = np.linspace(0.0, 1.0, _SKULL_RINGS + 1)[1:]
|
||||
|
||||
for t in ts:
|
||||
d = _slerp(pole, rim_d, 1.0 - t)
|
||||
w = (1.0 - t) ** _SKULL_BLEND # meets the rim exactly at t = 0
|
||||
# A skull is fuller than the border it springs from; peak it mid-sweep.
|
||||
r = ell_r(d) * (1.0 + (_SKULL_BULGE - 1.0) * np.sin(np.pi * t) ** 0.8)
|
||||
ring = C + d * (w * rim_r + (1.0 - w) * r)
|
||||
# Never dip below the chin: the sweep passing under the jaw would
|
||||
# otherwise hang a lip of geometry below the face. Vertices that hit
|
||||
# the clamp are also drawn in towards the neck axis, so the underside
|
||||
# closes as a small floor instead of a flat skirt sticking out.
|
||||
below = ring[:, 1] < chin_y
|
||||
if below.any():
|
||||
ring[below, 1] = chin_y
|
||||
ring[below, 0] *= 0.55
|
||||
ring[below, 2] = _NECK_Z + (ring[below, 2] - _NECK_Z) * 0.55
|
||||
if t == ts[-1]:
|
||||
ring = np.repeat((C + pole * ell_r(pole[None])[0])[None], n, axis=0)
|
||||
|
||||
base = sum(len(a) for a in out_v)
|
||||
out_v.append(ring)
|
||||
idx = np.arange(base, base + n)
|
||||
for i in range(n):
|
||||
a0, b0 = prev_idx[i], prev_idx[(i + 1) % n]
|
||||
a1, b1 = idx[i], idx[(i + 1) % n]
|
||||
out_f.append([a0, a1, b1])
|
||||
out_f.append([a0, b1, b0])
|
||||
prev_idx = idx
|
||||
|
||||
return np.vstack(out_v), np.array(out_f, dtype=np.int64)
|
||||
|
||||
|
||||
def _add_neck(verts: np.ndarray, faces: np.ndarray):
|
||||
"""A tapering tube dropped from inside the jaw; it fades out, so no shoulders."""
|
||||
ph = np.linspace(0.0, 2.0 * np.pi, _NECK_SEGS, endpoint=False)
|
||||
# Short, and flaring hard at the bottom: a straight vertical tube reads as
|
||||
# a pedestal, whereas a neck that widens into the top of the shoulders
|
||||
# reads as a bust — and the shorter it is, the larger the head can be drawn
|
||||
# in the same HUD band.
|
||||
ys = np.linspace(-5.5, -13.0, _NECK_RINGS)
|
||||
d = (ys + 5.5) / -7.5
|
||||
|
||||
rx = 4.6 * (1.0 + 0.52 * d ** 1.9)
|
||||
rz = 4.1 * (1.0 + 0.38 * d ** 1.9)
|
||||
nx = rx[:, None] * np.cos(ph)[None, :]
|
||||
nz = _NECK_Z + rz[:, None] * np.sin(ph)[None, :]
|
||||
ny = ys[:, None] * np.ones_like(ph)[None, :]
|
||||
|
||||
nv = np.stack([nx.ravel(), ny.ravel(), nz.ravel()], axis=1)
|
||||
base = len(verts)
|
||||
idx = base + np.arange(_NECK_RINGS * _NECK_SEGS).reshape(_NECK_RINGS, _NECK_SEGS)
|
||||
|
||||
nf = []
|
||||
for i in range(_NECK_RINGS - 1):
|
||||
for j in range(_NECK_SEGS):
|
||||
a, b = idx[i, j], idx[i, (j + 1) % _NECK_SEGS]
|
||||
c, e = idx[i + 1, (j + 1) % _NECK_SEGS], idx[i + 1, j]
|
||||
nf.append([a, b, c])
|
||||
nf.append([a, c, e])
|
||||
|
||||
# Enough rings that the fade steps stay small. Each quad splits into one
|
||||
# triangle with two top vertices and one with two bottom vertices, so a
|
||||
# steep per-vertex fade gradient makes the pair land on visibly different
|
||||
# brightnesses and the neck grows a sawtooth edge.
|
||||
fade = np.ones(base)
|
||||
nd = np.repeat(d, _NECK_SEGS)
|
||||
fade = np.concatenate([fade, 1.0 - 0.72 * np.clip(nd, 0.0, 1.0) ** 1.5])
|
||||
return np.vstack([verts, nv]), np.vstack([faces, np.array(nf)]), fade
|
||||
|
||||
|
||||
def _vertex_normals(verts: np.ndarray, faces: np.ndarray,
|
||||
outward: np.ndarray) -> np.ndarray:
|
||||
"""Area-weighted vertex normals, flipped to agree with `outward`.
|
||||
|
||||
`outward` must be a per-vertex direction that genuinely points out of the
|
||||
surface. A single "away from the mesh centroid" rule is NOT good enough:
|
||||
down at the base of the neck that vector points almost straight down while
|
||||
the real normal is horizontal, so the dot product hovers around zero and
|
||||
the sign flips at random — which tears the neck into an asymmetric slab of
|
||||
half-culled, half-lit triangles.
|
||||
"""
|
||||
a, b, c = verts[faces[:, 0]], verts[faces[:, 1]], verts[faces[:, 2]]
|
||||
fn = np.cross(b - a, c - a) # length carries the area — the weighting
|
||||
|
||||
n = np.zeros_like(verts)
|
||||
for k in range(3):
|
||||
np.add.at(n, faces[:, k], fn)
|
||||
n /= np.maximum(np.linalg.norm(n, axis=1, keepdims=True), 1e-9)
|
||||
|
||||
flip = (n * outward).sum(1) < 0
|
||||
n[flip] *= -1.0
|
||||
return n
|
||||
|
||||
|
||||
def _unique_edges(faces: np.ndarray) -> np.ndarray:
|
||||
e = np.vstack([faces[:, [0, 1]], faces[:, [1, 2]], faces[:, [2, 0]]])
|
||||
e = np.sort(e, axis=1)
|
||||
return np.unique(e, axis=0)
|
||||
|
||||
|
||||
def _check_landmarks(verts: np.ndarray) -> None:
|
||||
"""Fail loudly at build time if a landmark ring is not where it should be."""
|
||||
for left, right in (("eye_l", "eye_r"), ("brow_l", "brow_r")):
|
||||
cl = verts[LANDMARKS[left]].mean(0)
|
||||
cr = verts[LANDMARKS[right]].mean(0)
|
||||
assert cl[0] < 0 < cr[0], f"{left}/{right} are not on opposite sides"
|
||||
assert abs(cl[1] - cr[1]) < 0.5, f"{left}/{right} are at different heights"
|
||||
eye_y = verts[LANDMARKS["eye_l"]].mean(0)[1]
|
||||
brow_y = verts[LANDMARKS["brow_l"]].mean(0)[1]
|
||||
lips = verts[LANDMARKS["lips_out"]].mean(0)
|
||||
assert brow_y > eye_y, "brow is not above the eye"
|
||||
assert lips[1] < eye_y, "lips are not below the eyes"
|
||||
assert abs(lips[0]) < 0.5, "lips are not centred"
|
||||
|
||||
|
||||
def build_head() -> dict:
|
||||
"""Assemble the full head. Called once; `get_head_mesh()` caches the result."""
|
||||
verts, faces = _load_obj(_OBJ)
|
||||
_check_landmarks(verts)
|
||||
|
||||
n_face = len(verts)
|
||||
verts, faces = _add_cranium(verts, faces)
|
||||
n_head = len(verts)
|
||||
verts, faces, fade = _add_neck(verts, faces)
|
||||
|
||||
# ── normalise: crown → +1, chin → -1, eyes land on y ≈ 0 ────────────────
|
||||
head_y = verts[:n_head, 1]
|
||||
crown, chin = head_y.max(), head_y.min()
|
||||
scale = 2.0 / (crown - chin)
|
||||
centre = np.array([0.0, (crown + chin) * 0.5, 0.0])
|
||||
verts = (verts - centre) * scale
|
||||
|
||||
# Outward reference, per part: the head is star-shaped about its own centre,
|
||||
# while the neck is a tube whose outward direction is radial in x/z only.
|
||||
outward = verts - np.array([0.0, verts[:n_head, 1].mean(), 0.0])
|
||||
outward[n_head:] = verts[n_head:] - np.array([0.0, 0.0, _NECK_Z * scale])
|
||||
outward[n_head:, 1] = 0.0
|
||||
normals = _vertex_normals(verts, faces, outward)
|
||||
|
||||
# ── jaw rig ─────────────────────────────────────────────────────────────
|
||||
# Everything below the mouth swings on the mandible, tapering to nothing at
|
||||
# the ears and around the back so the nape and the neck stay put.
|
||||
mouth_y = verts[LANDMARKS["lips_out"], 1].mean()
|
||||
chin_y = verts[:n_head, 1].min()
|
||||
jaw = np.clip((mouth_y - verts[:, 1]) / (mouth_y - chin_y), 0.0, 1.0) ** 0.8
|
||||
jaw *= np.clip(0.30 + 0.85 * (verts[:, 2] / 0.55), 0.0, 1.0)
|
||||
jaw[n_head:] = 0.0 # the neck never moves
|
||||
jaw[LANDMARKS["lips_in"][:10]] = 1.0 # lower inner lip leads
|
||||
jaw[LANDMARKS["lips_out"][:10]] = 0.95
|
||||
|
||||
# ── brow rig ────────────────────────────────────────────────────────────
|
||||
# Raising the brows displaces the actual surface rather than sliding a drawn
|
||||
# line over it, so the brow ridge relights as it lifts.
|
||||
brow_y = verts[LANDMARKS["brow_l"] + LANDMARKS["brow_r"], 1].mean()
|
||||
brow = np.exp(-((verts[:, 1] - brow_y) / 0.115) ** 2)
|
||||
brow *= np.clip(verts[:, 2] / 0.35, 0.0, 1.0) # front of the face only
|
||||
brow *= np.exp(-(verts[:, 0] / 0.42) ** 2) # fades out past the temples
|
||||
brow[n_head:] = 0.0
|
||||
|
||||
# ── lip rig ─────────────────────────────────────────────────────────────
|
||||
# Vowels are not just "how far open" — /i/ spreads the lips wide, /u/ purses
|
||||
# them forward. This weight lets the renderer widen or round the mouth
|
||||
# region as a whole, so the surrounding skin follows instead of tearing away
|
||||
# from the landmark rings.
|
||||
lip_c = verts[LANDMARKS["lips_out"]].mean(axis=0)
|
||||
lips = np.exp(-((verts[:, 1] - lip_c[1]) / 0.155) ** 2)
|
||||
lips *= np.exp(-(verts[:, 0] / 0.30) ** 2)
|
||||
lips *= np.clip(verts[:, 2] / 0.40, 0.0, 1.0)
|
||||
lips[n_head:] = 0.0
|
||||
|
||||
edges = _unique_edges(faces)[::_WIRE_STRIDE]
|
||||
|
||||
# Neck and head interpenetrate, and a painter's-algorithm sort by triangle
|
||||
# depth interleaves them into a torn edge. Grouping fixes it: the neck is
|
||||
# always behind the head where they overlap, so draw every neck facet first.
|
||||
face_group = (faces >= n_head).all(axis=1).astype(np.int32) # 1 = neck
|
||||
|
||||
return {
|
||||
"face_group": np.ascontiguousarray(1 - face_group, dtype=np.float32),
|
||||
"brow": np.ascontiguousarray(brow, dtype=np.float32),
|
||||
"lips": np.ascontiguousarray(lips, dtype=np.float32),
|
||||
"lip_centre": np.ascontiguousarray(lip_c, dtype=np.float32),
|
||||
"verts": np.ascontiguousarray(verts, dtype=np.float32),
|
||||
"normals": np.ascontiguousarray(normals, dtype=np.float32),
|
||||
"faces": np.ascontiguousarray(faces, dtype=np.int32),
|
||||
"edges": np.ascontiguousarray(edges, dtype=np.int32),
|
||||
"jaw": np.ascontiguousarray(jaw, dtype=np.float32),
|
||||
"fade": np.ascontiguousarray(fade, dtype=np.float32),
|
||||
"landmarks": {k: np.array(v, dtype=np.int32) for k, v in LANDMARKS.items()},
|
||||
"n_face": n_face,
|
||||
"n_head": n_head,
|
||||
"span": (1.0, float(verts[:, 1].min())), # crown, bottom of the neck
|
||||
}
|
||||
|
||||
|
||||
_CACHE: dict | None = None
|
||||
|
||||
|
||||
def get_head_mesh() -> dict:
|
||||
"""Process-wide cached mesh — every HudCanvas shares the same arrays."""
|
||||
global _CACHE
|
||||
if _CACHE is None:
|
||||
_CACHE = build_head()
|
||||
return _CACHE
|
||||
+161
@@ -0,0 +1,161 @@
|
||||
"""
|
||||
core/confirm.py — a confirmation the model cannot forge.
|
||||
|
||||
THE PROBLEM WITH THE OLD GATE
|
||||
computer_settings guarded shutdown and restart like this:
|
||||
|
||||
confirmed = str(params.get("confirmed", "")).lower()
|
||||
if confirmed not in ("yes", "true", "1", "confirm"):
|
||||
return "Please confirm by calling again with confirmed=yes."
|
||||
|
||||
`confirmed` is a tool parameter, which means the *model* writes it. Nothing
|
||||
stops it from sending confirmed=yes on the first call, and nothing checks
|
||||
that a human was ever involved. It is a convention, not a gate — and its
|
||||
coverage was two actions, so deleting files and switching off the WiFi the
|
||||
assistant is talking over went through with no gate at all.
|
||||
|
||||
THE DESIGN HERE
|
||||
The confirmation token is issued by the *interface*, never by the model:
|
||||
|
||||
1. An action calls `request(...)` with a callable that does the real work.
|
||||
2. This module hands the UI a banner with CONFIRM / CANCEL and returns
|
||||
IMMEDIATELY with a sentence for the model to say out loud.
|
||||
3. If — and only if — the user presses CONFIRM, the UI calls `resolve()`,
|
||||
which runs the stored callable off the Qt thread.
|
||||
|
||||
Nothing blocks. The model keeps talking while the banner is up, so this
|
||||
costs no latency at all; in fact it is cheaper than the old gate, which
|
||||
burned two tool round trips (reject, then re-call) on every shutdown.
|
||||
|
||||
WHAT BELONGS HERE AND WHAT DOES NOT
|
||||
Only genuinely irreversible things. Anything that can be reversed should be
|
||||
done at once and pushed onto core/undo.py instead — undo is faster than a
|
||||
question, and an assistant that asks before every action is one nobody uses.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
import time
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Optional
|
||||
|
||||
# A pending confirmation is abandoned after this long. Chosen to outlast a
|
||||
# normal "hang on, let me look at the screen" pause without leaving a live
|
||||
# shutdown button sitting on the HUD for the rest of the day.
|
||||
TIMEOUT_SECONDS = 90.0
|
||||
|
||||
|
||||
@dataclass
|
||||
class _Pending:
|
||||
key: str
|
||||
title: str
|
||||
detail: str
|
||||
run: Callable[[], str]
|
||||
at: float
|
||||
|
||||
|
||||
_pending: Optional[_Pending] = None
|
||||
_lock = threading.Lock()
|
||||
|
||||
# Set once at startup by main.py. Signature: (title, detail) -> None for show,
|
||||
# and () -> None for hide. Both are marshalled onto the Qt thread by the UI.
|
||||
_show_cb: Optional[Callable[[str, str], None]] = None
|
||||
_hide_cb: Optional[Callable[[], None]] = None
|
||||
_log_cb: Optional[Callable[[str], None]] = None
|
||||
|
||||
|
||||
def bind(show, hide, log=None) -> None:
|
||||
"""Wire this module to the HUD. Called once from main.py at startup."""
|
||||
global _show_cb, _hide_cb, _log_cb
|
||||
_show_cb, _hide_cb, _log_cb = show, hide, log
|
||||
|
||||
|
||||
def _log(msg: str) -> None:
|
||||
if _log_cb:
|
||||
try:
|
||||
_log_cb(msg)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def request(key: str, title: str, detail: str, run: Callable[[], str]) -> str:
|
||||
"""Park an irreversible action behind the on-screen gate.
|
||||
|
||||
Returns the sentence the tool should hand back to the model — phrased as an
|
||||
instruction so the assistant asks the user out loud in their own language,
|
||||
rather than reading an English string verbatim."""
|
||||
global _pending
|
||||
|
||||
if _show_cb is None:
|
||||
# No interface bound (headless, or a very early call). Refuse rather
|
||||
# than silently performing something irreversible.
|
||||
return (f"I cannot confirm '{title}' right now because the interface is "
|
||||
f"not available, so I have not done it.")
|
||||
|
||||
with _lock:
|
||||
_pending = _Pending(key=key, title=title, detail=detail,
|
||||
run=run, at=time.monotonic())
|
||||
|
||||
try:
|
||||
_show_cb(title, detail)
|
||||
except Exception as e:
|
||||
with _lock:
|
||||
_pending = None
|
||||
return f"Could not ask for confirmation: {e}. Nothing was done."
|
||||
|
||||
_log(f"SYS: Awaiting confirmation — {title}")
|
||||
return (
|
||||
f"[CONFIRMATION_PENDING] I have put a confirmation on screen for: {title}. "
|
||||
f"Say ONE short sentence in the user's own language telling them you need "
|
||||
f"them to confirm it on the HUD before you do it. Do not claim it is done."
|
||||
)
|
||||
|
||||
|
||||
def resolve(accepted: bool) -> None:
|
||||
"""Called by the UI when the user presses CONFIRM or CANCEL.
|
||||
|
||||
Runs the stored callable on a worker thread — this is invoked from the Qt
|
||||
thread, and shutting the machine down from inside a button handler would
|
||||
freeze the interface on its way out."""
|
||||
global _pending
|
||||
|
||||
with _lock:
|
||||
p, _pending = _pending, None
|
||||
|
||||
if _hide_cb:
|
||||
try:
|
||||
_hide_cb()
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if p is None:
|
||||
return
|
||||
|
||||
if time.monotonic() - p.at > TIMEOUT_SECONDS:
|
||||
_log(f"SYS: Confirmation expired — {p.title}")
|
||||
return
|
||||
|
||||
if not accepted:
|
||||
_log(f"SYS: Cancelled — {p.title}")
|
||||
return
|
||||
|
||||
def _worker():
|
||||
try:
|
||||
result = p.run() or "Done."
|
||||
_log(f"SYS: Confirmed — {p.title}. {result}")
|
||||
except Exception as e:
|
||||
_log(f"ERR: {p.title} failed — {e}")
|
||||
|
||||
threading.Thread(target=_worker, daemon=True,
|
||||
name=f"confirm-{p.key}").start()
|
||||
|
||||
|
||||
def pending_title() -> str:
|
||||
"""'' when nothing is waiting. Lets an action avoid stacking two banners."""
|
||||
with _lock:
|
||||
if _pending is None:
|
||||
return ""
|
||||
if time.monotonic() - _pending.at > TIMEOUT_SECONDS:
|
||||
return ""
|
||||
return _pending.title
|
||||
+1378
File diff suppressed because it is too large.
Load diff
+173
@@ -0,0 +1,173 @@
|
||||
"""
|
||||
Push-to-talk — hold a key, speak, release.
|
||||
|
||||
Why this exists
|
||||
---------------
|
||||
Wake-word is hands-free but it is not always what you want: in a meeting, in a
|
||||
noisy room, or when you simply do not feel like saying a name out loud, a key
|
||||
you hold is faster and never mishears. It is also the natural way to use the
|
||||
assistant while another window has focus.
|
||||
|
||||
How it works, and what it costs
|
||||
-------------------------------
|
||||
Zero new dependencies, best available mechanism per platform:
|
||||
|
||||
* **Windows** — `GetAsyncKeyState` polled from one small thread. This is
|
||||
deliberately *not* `RegisterHotKey`, which only reports a press: push-to-talk
|
||||
needs the release too, and it needs to work while another application has
|
||||
focus. Polling two virtual-key codes 30 times a second is a rounding error of
|
||||
CPU and needs no message loop.
|
||||
* **macOS / Linux** — no portable way to read global key state without pulling
|
||||
in a new package or asking for accessibility permissions, so the chord is
|
||||
bound as an application shortcut instead: it works whenever the assistant's
|
||||
window has focus. `scope` reports which of the two you got, so the UI can say
|
||||
so honestly rather than pretending.
|
||||
|
||||
The class never raises. If the platform hook cannot be installed it simply
|
||||
reports `scope == "window"` and the Qt shortcut carries it.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import platform
|
||||
import threading
|
||||
import time
|
||||
from typing import Callable
|
||||
|
||||
_OS = platform.system()
|
||||
|
||||
# The default chord. Ctrl+Space is free in most desktop environments and is the
|
||||
# same finger shape on every keyboard layout, which matters for a worldwide app.
|
||||
DEFAULT_CHORD = ("ctrl", "space")
|
||||
|
||||
# Windows virtual-key codes for the names we accept.
|
||||
_VK = {
|
||||
"ctrl": 0x11, "shift": 0x10, "alt": 0x12,
|
||||
"space": 0x20, "f8": 0x77, "f9": 0x78, "f10": 0x79,
|
||||
"capslock": 0x14, "insert": 0x2D,
|
||||
}
|
||||
|
||||
# Qt key sequence text for the same chord, used by the windowed fallback.
|
||||
_QT_NAME = {"ctrl": "Ctrl", "shift": "Shift", "alt": "Alt", "space": "Space",
|
||||
"f8": "F8", "f9": "F9", "f10": "F10",
|
||||
"capslock": "CapsLock", "insert": "Ins"}
|
||||
|
||||
_POLL_HZ = 30.0
|
||||
# A key has to be down this long before we call it speech. It stops a stray
|
||||
# brush of the chord from opening the microphone.
|
||||
_DEBOUNCE_S = 0.06
|
||||
|
||||
|
||||
def chord_label(chord=DEFAULT_CHORD) -> str:
|
||||
"""Human-readable name of the chord, for the UI and the logs."""
|
||||
return "+".join(_QT_NAME.get(k, k.title()) for k in chord)
|
||||
|
||||
|
||||
def qt_sequence(chord=DEFAULT_CHORD) -> str:
|
||||
"""The same chord as a QKeySequence string."""
|
||||
return "+".join(_QT_NAME.get(k, k.title()) for k in chord)
|
||||
|
||||
|
||||
class PushToTalk:
|
||||
"""Calls `on_change(held: bool)` whenever the chord is pressed or released.
|
||||
|
||||
Start it once; it is safe to start and stop repeatedly, and safe to stop a
|
||||
detector that never started.
|
||||
"""
|
||||
|
||||
def __init__(self, on_change: Callable[[bool], None], chord=DEFAULT_CHORD):
|
||||
self._on_change = on_change
|
||||
self._chord = tuple(chord)
|
||||
self._thread: threading.Thread | None = None
|
||||
self._stop = threading.Event()
|
||||
self._held = False
|
||||
self._scope = "window"
|
||||
|
||||
# ── state ───────────────────────────────────────────────────────────────
|
||||
|
||||
@property
|
||||
def held(self) -> bool:
|
||||
return self._held
|
||||
|
||||
@property
|
||||
def scope(self) -> str:
|
||||
"""'global' once a system-wide hook is running, else 'window'."""
|
||||
return self._scope
|
||||
|
||||
@property
|
||||
def label(self) -> str:
|
||||
return chord_label(self._chord)
|
||||
|
||||
# ── lifecycle ───────────────────────────────────────────────────────────
|
||||
|
||||
def start(self) -> str:
|
||||
"""Begin watching. Returns the scope actually achieved."""
|
||||
self.stop()
|
||||
self._stop.clear()
|
||||
if _OS == "Windows" and self._can_poll():
|
||||
self._scope = "global"
|
||||
self._thread = threading.Thread(
|
||||
target=self._poll_loop, name="push-to-talk", daemon=True)
|
||||
self._thread.start()
|
||||
else:
|
||||
self._scope = "window"
|
||||
return self._scope
|
||||
|
||||
def stop(self) -> None:
|
||||
self._stop.set()
|
||||
t, self._thread = self._thread, None
|
||||
if t is not None and t.is_alive():
|
||||
t.join(timeout=1.0)
|
||||
self._set_held(False)
|
||||
|
||||
# ── the windowed fallback drives this directly ──────────────────────────
|
||||
|
||||
def set_held(self, held: bool) -> None:
|
||||
"""Feed a press/release from a Qt shortcut (non-Windows, or no hook)."""
|
||||
self._set_held(bool(held))
|
||||
|
||||
# ── internals ───────────────────────────────────────────────────────────
|
||||
|
||||
def _can_poll(self) -> bool:
|
||||
try:
|
||||
import ctypes
|
||||
ctypes.windll.user32.GetAsyncKeyState # noqa: B018 — presence check
|
||||
return all(k in _VK for k in self._chord)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def _set_held(self, held: bool) -> None:
|
||||
if held == self._held:
|
||||
return
|
||||
self._held = held
|
||||
try:
|
||||
self._on_change(held)
|
||||
except Exception:
|
||||
pass # a listener fault must never kill the watcher
|
||||
|
||||
def _poll_loop(self) -> None:
|
||||
import ctypes
|
||||
user32 = ctypes.windll.user32
|
||||
codes = [_VK[k] for k in self._chord]
|
||||
period = 1.0 / _POLL_HZ
|
||||
down_since = 0.0
|
||||
|
||||
while not self._stop.is_set():
|
||||
try:
|
||||
# The high bit of the return value is "currently down".
|
||||
down = all(user32.GetAsyncKeyState(c) & 0x8000 for c in codes)
|
||||
except Exception:
|
||||
break # driver or session teardown — fall back to windowed
|
||||
now = time.monotonic()
|
||||
if down:
|
||||
if down_since == 0.0:
|
||||
down_since = now
|
||||
elif now - down_since >= _DEBOUNCE_S:
|
||||
self._set_held(True)
|
||||
else:
|
||||
down_since = 0.0
|
||||
self._set_held(False)
|
||||
self._stop.wait(period)
|
||||
|
||||
self._set_held(False)
|
||||
self._scope = "window"
|
||||
+275
@@ -0,0 +1,275 @@
|
||||
"""
|
||||
Text → mouth shape, fused with the audio the avatar is actually speaking.
|
||||
|
||||
Why both sources
|
||||
----------------
|
||||
Formant analysis of the audio (see `_pcm_visemes` in main.py) gives excellent
|
||||
*timing* and a decent read on vowels, but it is blind to exactly the consonants
|
||||
lip-reading depends on. /m/, /b/ and /p/ are made with the lips pressed shut,
|
||||
and nothing in the spectrum reliably says "the lips are closed" — a nasal /m/
|
||||
and a nasal /n/ look nearly identical to a filter bank while looking completely
|
||||
different on a face.
|
||||
|
||||
The transcript knows those consonants for certain. So the text supplies *which
|
||||
shape*, the audio supplies *when* and *how strongly*, and the two are blended.
|
||||
If the transcript is late or missing the mouth silently falls back to the
|
||||
audio-only shape, which is what the previous version did on its own.
|
||||
|
||||
Language independence
|
||||
---------------------
|
||||
There is no per-language table here. Every character is reduced to one of the
|
||||
26 bare Latin letters — by Unicode decomposition for accents, by transliteration
|
||||
for Cyrillic and Greek — and articulation is looked up on that. So Turkish,
|
||||
English, German, French, Spanish, Polish, Vietnamese, Russian, Ukrainian and
|
||||
Greek all work from the same twenty-odd rules, and adding a language costs
|
||||
nothing because there is nothing to add.
|
||||
|
||||
Scripts whose spelling does not reveal pronunciation (CJK, Arabic, Devanagari,
|
||||
Hebrew, Thai) are detected by coverage and skipped, and the mouth runs on the
|
||||
audio-only shape — which is itself language-independent, being physics. The
|
||||
result is never wrong, only less detailed.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import unicodedata
|
||||
from collections import deque
|
||||
|
||||
# (openness 0..1, width -1..+1, closure 0..1)
|
||||
# closure forces the lips together regardless of loudness — it is the whole
|
||||
# reason the transcript is worth consulting.
|
||||
VISEMES: dict[str, tuple[float, float, float]] = {
|
||||
"REST": (0.00, 0.00, 0.00),
|
||||
"AA": (0.92, -0.05, 0.00), # a
|
||||
"E": (0.52, 0.42, 0.00), # e
|
||||
"I": (0.20, 0.62, 0.00), # i, ı
|
||||
"O": (0.55, -0.52, 0.00), # o, ö
|
||||
"U": (0.26, -0.74, 0.00), # u, ü, w
|
||||
"MBP": (0.00, 0.00, 1.00), # m, b, p — lips pressed shut
|
||||
"FV": (0.10, 0.22, 0.55), # f, v — lower lip to the teeth
|
||||
"S": (0.16, 0.42, 0.00), # s, ş, z, c, ç, j
|
||||
"L": (0.36, 0.18, 0.00), # l
|
||||
"TD": (0.28, 0.12, 0.00), # t, d, n
|
||||
"K": (0.30, -0.04, 0.00), # k, g, ğ, h
|
||||
"R": (0.28, -0.16, 0.00), # r
|
||||
}
|
||||
|
||||
# Relative duration of each class. Vowels carry the syllable; plosives are a tap.
|
||||
_DUR = {"REST": 1.0, "AA": 1.15, "E": 1.05, "I": 1.0, "O": 1.1, "U": 1.05,
|
||||
"MBP": 0.5, "FV": 0.8, "S": 0.9, "L": 0.65, "TD": 0.5, "K": 0.55,
|
||||
"R": 0.55}
|
||||
|
||||
# Articulation is a property of the *sound*, not of a language, so the table is
|
||||
# keyed on the 26 bare Latin letters and every script reaches it by reduction:
|
||||
# * diacritics are stripped by Unicode decomposition (é→e, ü→u, ế→e, ł→l …),
|
||||
# which covers every Latin-script language at once rather than one at a time;
|
||||
# * letters that do not decompose get a short explicit entry below;
|
||||
# * Cyrillic and Greek transliterate into the same 26 letters.
|
||||
# Anything else — CJK, Arabic, Devanagari, Hebrew, Thai — is not derivable from
|
||||
# its written form without a pronunciation dictionary, so those simply fall back
|
||||
# to the audio-only mouth. That is a clean degradation, not a missing language.
|
||||
_LETTER = {
|
||||
"a": "AA",
|
||||
"e": "E",
|
||||
"i": "I", "y": "I",
|
||||
"o": "O",
|
||||
"u": "U", "w": "U",
|
||||
"b": "MBP", "p": "MBP", "m": "MBP",
|
||||
"f": "FV", "v": "FV",
|
||||
"s": "S", "z": "S", "c": "S", "j": "S", "x": "S",
|
||||
"l": "L",
|
||||
"t": "TD", "d": "TD", "n": "TD",
|
||||
"k": "K", "g": "K", "h": "K", "q": "K",
|
||||
"r": "R",
|
||||
}
|
||||
|
||||
# Letters with no Unicode decomposition into a Latin base.
|
||||
_UNDECOMPOSED = {
|
||||
"ı": "i", "ø": "o", "đ": "d", "ħ": "h", "ŀ": "l", "ŧ": "t",
|
||||
"ß": "s", "æ": "a", "œ": "o", "þ": "t", "ð": "d", "ŋ": "n",
|
||||
"ł": "l",
|
||||
}
|
||||
|
||||
_CYRILLIC = {
|
||||
"а": "a", "б": "b", "в": "v", "г": "g", "д": "d", "е": "e", "ё": "e",
|
||||
"ж": "j", "з": "z", "и": "i", "й": "i", "к": "k", "л": "l", "м": "m",
|
||||
"н": "n", "о": "o", "п": "p", "р": "r", "с": "s", "т": "t", "у": "u",
|
||||
"ф": "f", "х": "h", "ц": "s", "ч": "s", "ш": "s", "щ": "s", "ъ": "",
|
||||
"ы": "i", "ь": "", "э": "e", "ю": "u", "я": "a",
|
||||
"і": "i", "ї": "i", "є": "e", "ґ": "g", "ў": "u",
|
||||
}
|
||||
|
||||
_GREEK = {
|
||||
"α": "a", "β": "v", "γ": "g", "δ": "d", "ε": "e", "ζ": "z", "η": "i",
|
||||
"θ": "t", "ι": "i", "κ": "k", "λ": "l", "μ": "m", "ν": "n", "ξ": "s",
|
||||
"ο": "o", "π": "p", "ρ": "r", "σ": "s", "ς": "s", "τ": "t", "υ": "i",
|
||||
"φ": "f", "χ": "h", "ψ": "s", "ω": "o",
|
||||
}
|
||||
|
||||
# Below this share of mappable letters the text is in a script we cannot read
|
||||
# phonetically, and forcing shapes onto it would be worse than not trying.
|
||||
_MIN_COVERAGE = 0.55
|
||||
|
||||
# English spellings that do not survive letter-by-letter reading.
|
||||
_DIGRAPH = {
|
||||
"sh": "S", "ch": "S", "ts": "S",
|
||||
"th": "TD", "ck": "K", "ng": "K", "gh": "K",
|
||||
"ph": "FV",
|
||||
"oo": "U", "ou": "O", "ow": "O", "wh": "U",
|
||||
"ee": "I", "ea": "I", "ie": "I",
|
||||
"qu": "K",
|
||||
}
|
||||
|
||||
_PAUSE = set(".,;:!?…\n")
|
||||
|
||||
|
||||
def to_latin(ch: str) -> str:
|
||||
"""Reduce any character to a bare Latin letter, or "" if it has none.
|
||||
|
||||
This is what makes the mouth language-agnostic: one reduction step replaces
|
||||
a per-language spelling table.
|
||||
"""
|
||||
c = ch.lower()
|
||||
if "a" <= c <= "z":
|
||||
return c
|
||||
if c in _UNDECOMPOSED:
|
||||
return _UNDECOMPOSED[c]
|
||||
if c in _CYRILLIC:
|
||||
return _CYRILLIC[c]
|
||||
if c in _GREEK:
|
||||
return _GREEK[c]
|
||||
# Strip combining marks: é→e, ü→u, ş→s, ğ→g, ế→e, ñ→n, å→a …
|
||||
base = "".join(k for k in unicodedata.normalize("NFD", c)
|
||||
if not unicodedata.combining(k))
|
||||
if len(base) == 1 and "a" <= base <= "z":
|
||||
return base
|
||||
if base and base != c: # e.g. fi → fi, take the first
|
||||
return to_latin(base[0])
|
||||
return ""
|
||||
|
||||
|
||||
def coverage(text: str) -> float:
|
||||
"""Fraction of the letters in `text` we can reduce to a Latin sound."""
|
||||
letters = [c for c in (text or "") if c.isalpha()]
|
||||
if not letters:
|
||||
return 0.0
|
||||
return sum(1 for c in letters if to_latin(c)) / len(letters)
|
||||
|
||||
|
||||
def text_to_visemes(text: str) -> list[tuple[str, float]]:
|
||||
"""Split a line of speech into (viseme, duration-weight) pairs.
|
||||
|
||||
Returns [] for scripts whose written form does not reveal pronunciation, so
|
||||
the caller falls back to the audio-only mouth instead of miming nonsense.
|
||||
"""
|
||||
s = (text or "").lower()
|
||||
if coverage(s) < _MIN_COVERAGE:
|
||||
return []
|
||||
|
||||
out: list[tuple[str, float]] = []
|
||||
i, n = 0, len(s)
|
||||
while i < n:
|
||||
ch = s[i]
|
||||
if ch in _PAUSE:
|
||||
out.append(("REST", 1.4))
|
||||
i += 1
|
||||
continue
|
||||
if ch.isspace():
|
||||
# A word gap is a beat, not a closed mouth — closing between every
|
||||
# word makes the avatar look like it is chewing.
|
||||
if out and out[-1][0] != "REST":
|
||||
out.append((out[-1][0], 0.35))
|
||||
i += 1
|
||||
continue
|
||||
|
||||
# Digraphs are an orthographic quirk of Latin spelling; check them on
|
||||
# the reduced letters so "SCH"/"Sch" and accented forms match too.
|
||||
two = to_latin(ch) + (to_latin(s[i + 1]) if i + 1 < n else "")
|
||||
if len(two) == 2 and two in _DIGRAPH:
|
||||
v = _DIGRAPH[two]
|
||||
i += 2
|
||||
else:
|
||||
base = to_latin(ch)
|
||||
i += 1
|
||||
if not base:
|
||||
continue
|
||||
v = _LETTER.get(base)
|
||||
if v is None:
|
||||
continue
|
||||
# A doubled letter is one sound in every orthography we handle here.
|
||||
if out and out[-1][0] == v:
|
||||
continue
|
||||
out.append((v, _DUR[v]))
|
||||
return out
|
||||
|
||||
|
||||
class VisemeStream:
|
||||
"""Fuses the transcript's shape sequence onto the audio's timing.
|
||||
|
||||
Thread note: `feed_text` runs on the receive coroutine and `frames` on the
|
||||
playback coroutine. Both live in the same asyncio loop, and `deque` append
|
||||
and popleft are atomic, so no lock is needed.
|
||||
"""
|
||||
|
||||
# Seconds a phoneme occupies at a normal speaking rate. The clock adapts
|
||||
# between these when the queue runs long (the model is talking fast) or
|
||||
# short (it is trailing off).
|
||||
_MIN_STEP = 0.045
|
||||
_MAX_STEP = 0.105
|
||||
|
||||
def __init__(self) -> None:
|
||||
self._q: deque[tuple[str, float]] = deque()
|
||||
self._cur = ("REST", 1.0)
|
||||
self._carry = 0.0
|
||||
|
||||
def reset(self) -> None:
|
||||
self._q.clear()
|
||||
self._cur = ("REST", 1.0)
|
||||
self._carry = 0.0
|
||||
|
||||
def feed_text(self, text: str) -> None:
|
||||
for item in text_to_visemes(text):
|
||||
self._q.append(item)
|
||||
# Never let a stalled turn pile up an unbounded backlog.
|
||||
while len(self._q) > 600:
|
||||
self._q.popleft()
|
||||
|
||||
@property
|
||||
def pending(self) -> int:
|
||||
return len(self._q)
|
||||
|
||||
def _step_seconds(self) -> float:
|
||||
# A long backlog means speech is outrunning the clock; shorten the step
|
||||
# so the mouth catches up instead of drifting further behind the voice.
|
||||
backlog = min(1.0, len(self._q) / 45.0)
|
||||
return self._MAX_STEP - (self._MAX_STEP - self._MIN_STEP) * backlog
|
||||
|
||||
def frames(self, audio, hop: float):
|
||||
"""Blend audio frames [(level, openness, width)] with the text queue."""
|
||||
out = []
|
||||
for level, a_open, a_wide in audio:
|
||||
if level <= 0.0:
|
||||
# Silence: let the queue wait rather than burning through it
|
||||
# during a pause, or the mouth ends up ahead of the voice.
|
||||
out.append((0.0, 0.0, 0.0))
|
||||
continue
|
||||
|
||||
self._carry += hop / max(1e-3, self._step_seconds() * self._cur[1])
|
||||
while self._carry >= 1.0 and self._q:
|
||||
self._cur = self._q.popleft()
|
||||
self._carry -= 1.0
|
||||
if self._carry >= 1.0:
|
||||
self._carry = 1.0 # queue empty — hold the last shape
|
||||
|
||||
t_open, t_wide, closure = VISEMES.get(self._cur[0], VISEMES["REST"])
|
||||
if self._q or self._cur[0] != "REST":
|
||||
# Text leads the shape; the audio keeps it honest so a bad
|
||||
# transcript alignment still tracks the real voice.
|
||||
o = 0.72 * t_open + 0.28 * a_open
|
||||
w = 0.78 * t_wide + 0.22 * a_wide
|
||||
else:
|
||||
o, w = a_open, a_wide
|
||||
closure = 0.0
|
||||
o *= 1.0 - closure
|
||||
out.append((level, max(0.0, min(1.0, o)), max(-1.0, min(1.0, w))))
|
||||
return out
|
||||
@@ -0,0 +1,211 @@
|
||||
"""
|
||||
Local wake-word detection for JARVIS ("Hey Jarvis").
|
||||
|
||||
Design goals:
|
||||
• ZERO cost when the feature is off — openwakeword is imported ONLY inside
|
||||
start()/install helpers, never at module load. If the user never enables
|
||||
wake word, none of this touches the app.
|
||||
• ZERO latency on the audio path — the microphone callback only ever does a
|
||||
cheap, non-blocking queue push (feed()); the actual model inference runs in
|
||||
this module's own background thread, so the real-time audio thread and the
|
||||
Gemini stream are never slowed.
|
||||
• Fully local & offline — audio fed here never leaves the machine; there is no
|
||||
network call except the one-time model download the user triggers from the UI.
|
||||
|
||||
openwakeword ships small ONNX models (a few MB each) and runs comfortably on a
|
||||
CPU. The pretrained wake phrase used here is "Hey Jarvis".
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import queue
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
from pathlib import Path
|
||||
from typing import Callable
|
||||
|
||||
# Pretrained openwakeword model that listens for "Hey Jarvis".
|
||||
WAKE_MODEL = "hey_jarvis"
|
||||
# Score in [0,1]; above this counts as a detection. Tunable per environment.
|
||||
DEFAULT_THRESHOLD = 0.5
|
||||
# Mic frames arrive at 16 kHz int16; this is just the detector's input rate.
|
||||
SAMPLE_RATE = 16000
|
||||
|
||||
|
||||
def is_installed() -> bool:
|
||||
"""True if the openwakeword package is importable (no model check)."""
|
||||
try:
|
||||
import importlib.util
|
||||
return importlib.util.find_spec("openwakeword") is not None
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def is_ready() -> bool:
|
||||
"""True if openwakeword is installed AND its model files are present on disk.
|
||||
|
||||
This is a cheap, DETERMINISTIC file-existence check. It deliberately does NOT
|
||||
construct a Model to probe readiness — doing that is slow and, worse, can clash
|
||||
with the detector's own Model when it's already running, which intermittently
|
||||
returned False and made the UI flicker to 'not downloaded'. Never raises.
|
||||
"""
|
||||
if not is_installed():
|
||||
return False
|
||||
try:
|
||||
import openwakeword
|
||||
models_dir = Path(openwakeword.__file__).resolve().parent / "resources" / "models"
|
||||
if not models_dir.is_dir():
|
||||
return False
|
||||
has_wake = (any(models_dir.glob(f"{WAKE_MODEL}*.onnx"))
|
||||
or any(models_dir.glob(f"{WAKE_MODEL}*.tflite")))
|
||||
has_mel = (any(models_dir.glob("melspectrogram*.onnx"))
|
||||
or any(models_dir.glob("melspectrogram*.tflite")))
|
||||
has_emb = (any(models_dir.glob("embedding_model*.onnx"))
|
||||
or any(models_dir.glob("embedding_model*.tflite")))
|
||||
return bool(has_wake and has_mel and has_emb)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def install_and_download(logger: Callable[[str], None] = print,
|
||||
notify: Callable[[str], None] | None = None) -> tuple[bool, str]:
|
||||
"""
|
||||
One-click setup for the UI button: pip-install openwakeword if missing, then
|
||||
download the wake model. Returns (ok, message). Never raises — every failure
|
||||
is reported through the returned message and the logger.
|
||||
"""
|
||||
_tell = notify or (lambda _msg: None)
|
||||
try:
|
||||
if not is_installed():
|
||||
logger("Wake word: installing openwakeword (one-time)…")
|
||||
_tell("Wake word: installing openwakeword (one-time)…")
|
||||
r = subprocess.run(
|
||||
[sys.executable, "-m", "pip", "install", "openwakeword"],
|
||||
capture_output=True, text=True,
|
||||
)
|
||||
if r.returncode != 0:
|
||||
tail = (r.stderr or r.stdout or "").strip().splitlines()[-1:] or [""]
|
||||
return False, f"pip install failed: {tail[0][:160]}"
|
||||
# Download the pretrained melspectrogram/embedding + wake models.
|
||||
logger("Wake word: downloading models…")
|
||||
_tell("Wake word: downloading models…")
|
||||
try:
|
||||
import openwakeword.utils as _u
|
||||
try:
|
||||
_u.download_models([WAKE_MODEL])
|
||||
except TypeError:
|
||||
_u.download_models() # older signature downloads the default set
|
||||
except Exception as e:
|
||||
return False, f"model download failed: {e}"
|
||||
|
||||
if not is_ready():
|
||||
return False, "installed, but the wake model could not be loaded."
|
||||
logger("Wake word: ready.")
|
||||
return True, "Wake word installed and ready."
|
||||
except Exception as e:
|
||||
return False, f"setup error: {e}"
|
||||
|
||||
|
||||
class WakeWordDetector:
|
||||
"""
|
||||
Runs the wake model in a dedicated thread. The mic thread calls feed() with
|
||||
raw int16 frames; detections invoke on_detect() (called from this thread —
|
||||
the callback must marshal to whatever loop/UI it needs).
|
||||
"""
|
||||
|
||||
def __init__(self, on_detect: Callable[[], None],
|
||||
threshold: float = DEFAULT_THRESHOLD,
|
||||
logger: Callable[[str], None] = print,
|
||||
notify: Callable[[str], None] | None = None):
|
||||
self._on_detect = on_detect
|
||||
self._threshold = threshold
|
||||
self._logger = logger
|
||||
# See PluginRegistry: `logger` is the console and gets everything,
|
||||
# `notify` is the activity log and gets only what the user must act on.
|
||||
self._notify = notify or (lambda _msg: None)
|
||||
self._queue: queue.Queue = queue.Queue(maxsize=50)
|
||||
self._thread: threading.Thread | None = None
|
||||
self._running = False
|
||||
self._model = None
|
||||
self._ready = False
|
||||
|
||||
def start(self) -> bool:
|
||||
"""Load the model and spawn the inference thread. Returns True on success.
|
||||
Safe to call again — a no-op if already running. Never raises."""
|
||||
if self._running:
|
||||
return True
|
||||
try:
|
||||
from openwakeword.model import Model
|
||||
self._model = Model(wakeword_models=[WAKE_MODEL], inference_framework="onnx")
|
||||
except Exception as e:
|
||||
self._logger(f"Wake word: could not load model — {e}")
|
||||
self._notify("Wake word unavailable — use the WAKE NOW button.")
|
||||
self._model = None
|
||||
return False
|
||||
self._running = True
|
||||
self._ready = True
|
||||
self._thread = threading.Thread(target=self._loop, daemon=True, name="WakeWordThread")
|
||||
self._thread.start()
|
||||
self._logger("Wake word: listening for 'Hey Jarvis'.")
|
||||
return True
|
||||
|
||||
def stop(self) -> None:
|
||||
self._running = False
|
||||
# unblock the thread if it's waiting on the queue
|
||||
try:
|
||||
self._queue.put_nowait(None)
|
||||
except Exception:
|
||||
pass
|
||||
self._model = None
|
||||
self._ready = False
|
||||
|
||||
@property
|
||||
def ready(self) -> bool:
|
||||
return self._ready
|
||||
|
||||
def feed(self, frame_int16) -> None:
|
||||
"""Called from the mic callback (real-time thread). Must stay cheap and
|
||||
never block — the frame is copied and dropped if the queue is backed up."""
|
||||
if not self._running:
|
||||
return
|
||||
try:
|
||||
# frame_int16 is a numpy int16 array (possibly 2-D mono) — flatten to 1-D
|
||||
data = frame_int16[:, 0].copy() if getattr(frame_int16, "ndim", 1) > 1 else frame_int16.copy()
|
||||
self._queue.put_nowait(data)
|
||||
except queue.Full:
|
||||
pass
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _loop(self) -> None:
|
||||
import numpy as np
|
||||
while self._running:
|
||||
try:
|
||||
frame = self._queue.get()
|
||||
if frame is None or not self._running:
|
||||
break
|
||||
scores = self._model.predict(np.asarray(frame, dtype=np.int16))
|
||||
score = 0.0
|
||||
if isinstance(scores, dict):
|
||||
# match the jarvis model regardless of exact key suffix
|
||||
for k, v in scores.items():
|
||||
if "jarvis" in k.lower():
|
||||
score = max(score, float(v))
|
||||
if score == 0.0 and scores:
|
||||
score = max(float(v) for v in scores.values())
|
||||
if score >= self._threshold:
|
||||
# drain any backlog so we don't double-fire on the same utterance
|
||||
self._drain()
|
||||
try:
|
||||
self._on_detect()
|
||||
except Exception as e:
|
||||
self._logger(f"Wake word: on_detect error — {e}")
|
||||
except Exception as e:
|
||||
self._logger(f"Wake word: inference error — {e}")
|
||||
|
||||
def _drain(self) -> None:
|
||||
try:
|
||||
while True:
|
||||
self._queue.get_nowait()
|
||||
except Exception:
|
||||
pass
|
||||
Reference in new issue
Block a user