Ricerca web senza chiavi (Bing/DuckDuckGo), documenti interrogabili locali, riserva cloud, consumi, avviso memoria, Tailscale, frasi d'attesa; motore interno più leggero (niente MCP esterni) e voce senza scatti

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
lucianoandClaude Opus 5.5 committed 2026-10-08 12:41:37 +02:00
1 parent 54895586c8
commit 0d4f4b7b37
13 files changed
+630 -28

No files matched your search

+117 -6
View File
@@ -3,6 +3,7 @@ from __future__ import annotations
import json
import queue
import sys
import threading
import time
@@ -56,6 +57,29 @@ def _diario(testo: str, strumenti: list[str], origine: str) -> None:
pass
PREZZI = {"claude api": (5.0, 25.0)} # $ per milione di token (input, output), Opus via API; ponytail: tabella fissa, aggiornare se cambiano i listini
def _consumi(ev: dict) -> None:
"""Riga in data/consumi.jsonl: motore, token in/out, costo (dichiarato da Claude Code o stimato per l'API)."""
try:
import json as _j
from avatar.settings import DATA_DIR as _D
costo = ev.get("costo")
if costo is None:
pin, pout = next((v for k, v in PREZZI.items() if str(ev.get("motore", "")).startswith(k)), (0.0, 0.0))
costo = (ev.get("input", 0) * pin + ev.get("output", 0) * pout) / 1e6
with open(_D / "consumi.jsonl", "a", encoding="utf-8") as f:
f.write(_j.dumps({"ts": time.strftime("%Y-%m-%d %H:%M"), "motore": ev.get("motore", "?"), "input": ev.get("input", 0),
"output": ev.get("output", 0), "costo": round(float(costo), 5), "abbonamento": bool(ev.get("abbonamento"))}) + "\n")
except Exception:
pass
FRASI_ATTESA = ["Un attimo che controllo.", "Fammi verificare.", "Dammi un secondo.", "Vediamo un po'.", "Ci guardo subito.",
"Aspetta un attimo, controllo.", "Mmh, fammi vedere.", "Un momento.", "Ok, ci penso un attimo.", "Fammi dare un'occhiata."]
class Assistant:
def __init__(self, ui, settings: Settings) -> None:
self.ui = ui
@@ -168,6 +192,7 @@ class Assistant:
except Exception as err:
self.ui.write_log(f"ERR: Riconoscimento vocale non disponibile — {err}")
self.reopen_audio()
self._prepare_fillers()
self._ensure_engine()
self._start_whatsapp_bridge()
try:
@@ -223,6 +248,18 @@ class Assistant:
self._engine = AnthropicEngine(api_key, self.name, self.user_name, s.get("effort"))
return self._engine
def _fallback_engine(self):
"""Riserva cloud per i motori locali: Claude Code (abbonamento), o l'API se c'è la chiave. Creata una volta."""
s = self.settings
if getattr(self, "_backup", None) is None:
if s.get_secret("anthropic_api_key"):
self._backup = AnthropicEngine(s.get_secret("anthropic_api_key"), self.name, self.user_name, s.get("effort"))
else:
self._backup = ClaudeCodeEngine(s.get("claudecode_model") or "sonnet", s.get("claudecode_access") or "chat",
s.get("claudecode_path") or "", s.get("claudecode_config_dir") or "",
self.name, self.user_name, s.get("effort"))
return self._backup
def reconfigure(self) -> None:
"""Dopo un salvataggio delle impostazioni."""
self.interrupt(silent=True)
@@ -241,6 +278,43 @@ class Assistant:
self.ui.write_log(f"ERR: Avatar 3D — {err}")
self.ui.write_log(f"SYS: Impostazioni applicate — motore {self.settings.get('provider')}.")
def _prepare_fillers(self) -> None:
"""Sintetizza una volta le frasi d'attesa con la voce attuale: al bisogno partono subito, senza attendere la sintesi."""
voice = self.voice
self._fillers = []
def work():
out = []
for f in FRASI_ATTESA:
try:
a = voice.synthesize(f)
if len(a):
out.append((f, a))
except Exception:
break
if voice is self.voice:
self._fillers = out
threading.Thread(target=work, daemon=True, name="frasi-attesa").start()
def _maybe_filler(self, turn: int, answered: list) -> None:
"""Dopo ~1,3 s senza risposta, una frase d'attesa breve e diversa dalla precedente."""
time.sleep(1.3)
if answered[0] or turn != self._turn or self._speaking or not self.settings.get("frasi_attesa", True):
return
choices = [x for x in getattr(self, "_fillers", []) if x[0] != getattr(self, "_last_filler", "")]
if not choices:
return
import random
text, audio = random.choice(choices)
self._last_filler = text
remote = getattr(self, "remote", None)
if remote is not None and remote.has_clients():
try:
remote.on_audio(text, audio)
except Exception:
pass
self._audio_q.put((turn, text, audio))
def reload_voice(self, from_customise: bool = False) -> None:
"""Ricarica la voce. `from_customise`: la scelta arriva dal pannello Customise
(Sara / Nicola / Sistema) e va copiata nelle impostazioni; altrimenti comandano
@@ -269,6 +343,7 @@ class Assistant:
self.voice = new
self.ui.write_log(f"SYS: Voce: {getattr(new, 'voice', '') or 'sistema'} ({'Kokoro' if self.settings.get('tts_engine') == 'kokoro' else 'macOS'}).")
threading.Thread(target=self._safe_load_voice, daemon=True).start()
self._prepare_fillers()
def _safe_load_voice(self) -> None:
try:
@@ -353,11 +428,17 @@ class Assistant:
splitter = SentenceSplitter()
used: list[str] = []
failed = [""]
answered = [False]
threading.Thread(target=self._maybe_filler, args=(turn, answered), daemon=True).start()
fallback_ok = [self.settings.get("provider") in ("mlx", "local") and bool(self.settings.get("riserva_cloud", True))]
def emit(ev: dict) -> None:
if turn != self._turn:
return
t = ev.get("type")
if t == "usage":
_consumi(ev); return
if t == "status" and ev.get("status") == "working" and ev.get("detail"):
used.append(str(ev["detail"]))
if t in ("done", "error"):
@@ -380,6 +461,7 @@ class Assistant:
elif ev["status"] == "loading" and detail:
self.ui.write_log(f"SYS: {detail}…")
elif t == "text":
answered[0] = True
for s in splitter.push(ev["delta"]):
self._speech_q.put((turn, s))
elif t == "memory_saved":
@@ -393,6 +475,9 @@ class Assistant:
if ev.get("sources"):
self.ui.show_content("Fonti", "\n".join(f"• {s['title']}\n {s['url']}" for s in ev["sources"]))
elif t == "error":
if not ev.get("aborted") and fallback_ok[0]:
failed[0] = ev["message"] # motore locale fallito: la riserva cloud ripete la domanda
return
if not ev.get("aborted"):
self.ui.write_log(f"ERR: {ev['message']}")
self._speech_q.put((turn, clean_for_speech("Scusa, c'è stato un problema: " + ev["message"])[:300]))
@@ -414,8 +499,16 @@ class Assistant:
attach_path = str(p)
except Exception as err:
self.ui.write_log(f"ERR: Allegato — {err}")
doc = sys.modules.get("avatar_plugins.documenti_rag")
if doc is not None:
doc.pausa(True)
try:
engine.send(engine_text, emit, self._abort, image=image, attach_path=attach_path)
if failed[0] and not self._abort.is_set():
backup = self._fallback_engine()
fallback_ok[0] = False
self.ui.write_log(f"SYS: Motore locale in errore ({failed[0][:80]}): rispondo con Claude.")
backup.send(engine_text, emit, self._abort, image=image, attach_path=attach_path)
finally:
self._busy = False
self._speech_q.put((turn, END))
@@ -431,18 +524,33 @@ class Assistant:
continue
if hasattr(self.voice, "stream"): # voce in streaming: i pezzi partono mentre la frase è ancora in generazione
try:
first = True
for chunk in self.voice.stream(item, cancel=lambda t=turn: t != self._turn):
if turn != self._turn:
break
first, buf = True, []
lead = getattr(self, "_tts_lead", 1.0) # secondi d'audio accumulati prima di iniziare a parlare
def push(audio):
nonlocal first
remote = getattr(self, "remote", None)
if remote is not None and remote.has_clients():
try:
remote.on_audio(item if first else "", chunk)
remote.on_audio(item if first else "", audio)
except Exception:
pass
self._audio_q.put((turn, item if first else "", chunk))
self._audio_q.put((turn, item if first else "", audio))
first = False
for chunk in self.voice.stream(item, cancel=lambda t=turn: t != self._turn):
if turn != self._turn:
break
if buf is not None:
buf.append(chunk)
if sum(len(c) for c in buf) / 24000 >= lead:
push(np.concatenate(buf)); buf = None
continue
if self._speaking and self.player._cursor < time.time(): # il lettore è rimasto a secco: scatto
self._tts_lead = min(4.0, getattr(self, "_tts_lead", 1.0) + 0.5)
push(chunk)
if buf:
push(np.concatenate(buf))
except Exception as err:
self.ui.write_log(f"ERR: Sintesi vocale — {err}")
continue
@@ -468,6 +576,9 @@ class Assistant:
if text is END:
self.player.drain()
self._speaking = False
doc = sys.modules.get("avatar_plugins.documenti_rag")
if doc is not None and not self._busy:
doc.pausa(False)
self._tail_until = time.monotonic() + 0.5
remote = getattr(self, "remote", None)
if remote is not None:
+4
View File
@@ -137,6 +137,10 @@ class AnthropicEngine:
full += event.delta.text
emit({"type": "text", "delta": event.delta.text})
message = stream.get_final_message()
u = getattr(message, "usage", None)
if u is not None:
inp = int(getattr(u, "input_tokens", 0) or 0) + int(getattr(u, "cache_read_input_tokens", 0) or 0) + int(getattr(u, "cache_creation_input_tokens", 0) or 0)
emit({"type": "usage", "motore": f"claude api ({MODEL})", "input": inp, "output": int(getattr(u, "output_tokens", 0) or 0)})
except anthropic.BadRequestError as err:
if self._compaction and not full:
print(f"[Claude] compattazione rifiutata, la disattivo: {err.message}")
+4
View File
@@ -206,6 +206,10 @@ class ClaudeCodeEngine:
emit({"type": "status", "status": "thinking"})
elif t == "result":
out["result"] = msg.get("result") or ""
u = msg.get("usage") or {}
emit({"type": "usage", "motore": f"claude code ({self.model})",
"input": int(u.get("input_tokens", 0)) + int(u.get("cache_read_input_tokens", 0)) + int(u.get("cache_creation_input_tokens", 0)),
"output": int(u.get("output_tokens", 0)), "costo": float(msg.get("total_cost_usd") or 0), "abbonamento": True})
if msg.get("is_error") or (msg.get("subtype") and msg["subtype"] != "success"):
errs = msg.get("errors") or []
out["error"] = "; ".join(map(str, errs)) or out["result"] or msg.get("subtype")
+56 -7
View File
@@ -15,7 +15,7 @@ from pathlib import Path
from avatar.memory_tools import memory_prompt, openai_tools, parse_args, run_tool
from avatar.plugins import registry
from avatar.websearch import brave_search
from avatar.websearch import web_search
from .base import Emit, History, Meter, compact_history, persona_text, summary_block, today_label, user_block
from .openai_compat import SEARCH_TOOL, Aborted
@@ -26,7 +26,7 @@ SAMPLING = {"temp": 0.6, "top_p": 0.8, "top_k": 20} # poco sotto i valori cons
DEBUG = bool(os.environ.get("MLX_DEBUG")) # stampa il testo grezzo generato (tag compresi)
DEFAULT_MODEL = "mlx-community/Qwen3-30B-A3B-Instruct-2507-4bit"
HUB = Path.home() / ".cache" / "huggingface" / "hub"
_EXCLUDE = ("whisper", "tts", "embed", "rerank", "clip", "vl-", "-vl", "vision", "diffusion", "parakeet")
_EXCLUDE = ("whisper", "tts", "kokoro", "embed", "rerank", "clip", "vl-", "-vl", "vision", "diffusion", "parakeet")
_lock = threading.Lock()
_loaded: dict = {"name": None, "model": None, "tokenizer": None, "cache": None, "tokens": [], "snap": None}
@@ -44,6 +44,31 @@ def cached_models() -> list[str]:
return out
def model_size_gb(name: str) -> float:
"""Peso su disco dei pesi di un modello nella cache (≈ memoria occupata una volta caricato); 0 se non scaricato."""
snaps = HUB / ("models--" + name.replace("/", "--")) / "snapshots"
if not snaps.exists():
return 0.0
return sum(f.stat().st_size for f in snaps.rglob("*.safetensors")) / 1e9
def memory_check(name: str) -> str:
"""Avviso se modello + voce + trascrizione superano ~75% della RAM del Mac."""
import subprocess
ram = int(subprocess.run(["sysctl", "-n", "hw.memsize"], capture_output=True, text=True).stdout.strip() or 0) / 1024 ** 3
gb = model_size_gb(name)
if not gb or not ram:
return "Peso non noto: verrà scaricato al primo uso."
extra = 2.4 + 0.5 # voce Qwen3-TTS 8 bit + Whisper
tot = gb + extra
msg = f"Modello {gb:.1f} GB + voce e trascrizione ≈ {tot:.0f} GB su {ram:.0f} GB di RAM."
if tot > ram * 0.75:
msg += " ⚠ Troppo pesante: il Mac rallenterà e le immagini potrebbero non avere spazio."
elif tot > ram * 0.6:
msg += " Va bene, ma con poca memoria libera per generare immagini."
return msg
def load_model(name: str):
"""Carica (o riusa) il modello. Un solo modello in memoria per volta."""
with _lock:
@@ -141,7 +166,7 @@ class TagFilter:
# "raw_args": il template riproduce gli argomenti come stringa grezza (prompt identico a ciò che il modello ha scritto).
FORMATS = {
"hermes": {"think": ("<think>", "</think>"), "call": ("<tool_call>", "</tool_call>"), "thinking": False, "raw_args": False},
"gemma4": {"think": ("<|channel>", "<channel|>"), "call": ("<|tool_call>", "<tool_call|>"), "thinking": False, "raw_args": True},
"gemma4": {"think": ("<|channel>", "<channel|>"), "call": ("<|tool_call>", "<tool_call|>"), "thinking": False, "raw_args": False},
"argkey": {"think": ("<think>", "</think>"), "call": ("<tool_call>", "</tool_call>"), "thinking": True, "raw_args": False},
}
GEMMA_ESC = '<|"|>'
@@ -214,6 +239,29 @@ def _parse_call(raw: str, fmt: str) -> tuple[str, dict, str] | None:
return str(d["name"]), dict(args or {}), json.dumps(args or {}, ensure_ascii=False)
def _dict_args(m: dict) -> dict:
"""Argomenti delle chiamate a strumenti sempre come dizionario: la cronologia è condivisa tra modelli e alcuni
template (Spark, Gemma heretic) rifiutano le stringhe salvate da altri modelli."""
calls = m.get("tool_calls")
if not calls or all(isinstance((c.get("function") or {}).get("arguments"), dict) for c in calls):
return m
fixed = []
for c in calls:
fn = dict(c.get("function") or {})
a = fn.get("arguments")
if not isinstance(a, dict):
try:
a = json.loads(a)
except Exception:
try:
a = json.loads(_gemma_to_json("{" + str(a) + "}"))
except Exception:
a = {}
fn["arguments"] = a if isinstance(a, dict) else {}
fixed.append({**c, "function": fn})
return {**m, "tool_calls": fixed}
class MLXEngine:
name = "mlx"
@@ -321,7 +369,7 @@ class MLXEngine:
"- Per azioni sul Mac (per esempio il calendario) usa gli strumenti dedicati. Se uno risponde con [CONFIRMATION_PENDING], chiedi all'utente di confermare sul pannello e non dire che è fatto.\n"
"- Non inventare mai dati reali: meteo, calendario, mail, messaggi, file, ora, server e memoria li ottieni SOLO chiamando lo strumento corrispondente, anche a metà conversazione. Rispondere senza averlo chiamato è un errore grave.\n"
"- Chiama gli strumenti solo nel formato previsto dal tuo template e mai descrivendoli a parole.")
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if self.search_api_key
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if True
else "\n- Non hai accesso al web: se ti chiedono informazioni aggiornate, dillo chiaramente.")
return f"{persona_text(self.assistant_name)}\n\n{user_block(self.user_name, memory_prompt())}{summary_block(self.history)}{note}"
@@ -335,11 +383,11 @@ class MLXEngine:
return msgs[start:]
def _tools(self) -> list:
return openai_tools() + registry.openai_tools() + ([SEARCH_TOOL] if self.search_api_key else [])
return openai_tools() + registry.openai_tools(local=True) + [SEARCH_TOOL]
def _prompt(self, generation: bool = True) -> str:
"""Prompt completo; con generation=False si ferma alla fine dell'ultimo messaggio (parte stabile)."""
messages = [{"role": "system", "content": self._system()}, *self._recent()]
messages = [{"role": "system", "content": self._system()}, *[_dict_args(m) for m in self._recent()]]
try:
return self.tokenizer.apply_chat_template(messages, tools=self._tools(), add_generation_prompt=generation, tokenize=False, enable_thinking=self._thinking())
except TypeError:
@@ -362,7 +410,7 @@ class MLXEngine:
if name == "cerca_web":
emit({"type": "status", "status": "searching"})
try:
hits = brave_search(str(args.get("query", "")), self.search_api_key)
hits = web_search(str(args.get("query", "")), self.search_api_key)
except Exception as err:
return f"Errore nella ricerca: {err}"
for h in hits[:5]:
@@ -415,6 +463,7 @@ class MLXEngine:
else:
# generate_step inserisce nella cache ogni token emesso (anche l'ultimo, anche quando si ferma per limite).
_loaded.update(cache=cache, tokens=tokens + generated)
emit({"type": "usage", "motore": f"locale ({self.model_name.split('/')[-1]})", "input": len(tokens), "output": len(generated)})
tail = tools.flush()
if tail:
text += tail
+4 -4
View File
@@ -8,7 +8,7 @@ import openai
from avatar.memory_tools import memory_prompt, openai_tools, parse_args, run_tool
from avatar.plugins import registry
from avatar.websearch import brave_search
from avatar.websearch import web_search
from .base import Emit, History, Meter, compact_history, persona_text, summary_block, today_label, user_block
MAX_ROUNDS = 8
@@ -94,7 +94,7 @@ class OpenAICompatEngine:
"- Quando l'utente ti chiede di ricordare qualcosa, o ti dice un fatto importante su di sé, DEVI chiamare salva_memoria prima di rispondere. Non dire mai di aver salvato senza averlo chiamato davvero.\n"
"- Per cancellare una memoria chiama dimentica_memoria; per cercarne una non presente nel prompt chiama cerca_memoria.\n"
"- Per azioni sul Mac (per esempio il calendario) usa gli strumenti dedicati. Se uno risponde con [CONFIRMATION_PENDING], chiedi all'utente di confermare sul pannello e non dire che è fatto.")
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if self.search_api_key
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if True
else "\n- Non hai accesso al web: se ti chiedono informazioni aggiornate, dillo chiaramente.")
else:
note = "\n\nNota: in questa modalità non hai strumenti (niente memoria automatica né ricerca web)."
@@ -121,14 +121,14 @@ class OpenAICompatEngine:
def _tools(self):
if not self._tools_supported:
return None
return openai_tools() + registry.openai_tools() + ([SEARCH_TOOL] if self.search_api_key else [])
return openai_tools() + registry.openai_tools(local=True) + [SEARCH_TOOL]
def _run_tool(self, name: str, raw_args: str, emit: Emit, sources: list) -> str:
args = parse_args(raw_args)
if name == "cerca_web":
emit({"type": "status", "status": "searching"})
try:
hits = brave_search(str(args.get("query", "")), self.search_api_key)
hits = web_search(str(args.get("query", "")), self.search_api_key)
except Exception as err:
return f"Errore nella ricerca: {err}"
for h in hits[:5]:
+5
View File
@@ -90,3 +90,8 @@ Sei **Ava**, l'assistente personale di chi ti parla. Vivi in un'app sul suo Mac
## Musica
- Per far suonare musica usa Spotify (spotify_riproduci, spotify_coda, spotify_controllo). Quando l'utente chiede consigli, "mettimi qualcosa", "qualcosa di nuovo che mi piaccia" o musica per un momento (cena, lavoro, festa), chiama spotify_gusti, scegli tu 6-10 brani reali coerenti con i suoi gusti (DJ, producer: conta anche il genere e il mood del momento), dillo in una frase e falli partire con spotify_coda.
## Documenti dell'utente
- Per domande sul contenuto dei documenti dell'utente (contratti, bollette, ricevute, manuali, appunti) usa documenti_cerca prima di dire che non lo sai; rispondi citando il file e, se i passaggi non contengono la risposta, dillo. Per trovare un file per nome resta file_cerca.
- Per "quanto ho speso / quanti token" usa consumi_riepilogo.
+12 -2
View File
@@ -127,8 +127,18 @@ class PluginRegistry:
def anthropic_tools(self) -> list[dict]:
return [{"name": t.name, "description": t.description, "eager_input_streaming": True, "input_schema": t.parameters} for t in self.active()]
def openai_tools(self) -> list[dict]:
return [{"type": "function", "function": {"name": t.name, "description": t.description, "parameters": t.parameters}} for t in self.active()]
def openai_tools(self, local: bool = False) -> list[dict]:
"""Per i modelli locali (local=True) esclude i server MCP esterni: le loro descrizioni (Cua Driver ~14k token)
rallentano troppo il primo turno. Si riattivano con l'impostazione mcp_locali."""
tools = self.active()
if local:
try:
from avatar.settings import Settings
if not Settings().get("mcp_locali"):
tools = [t for t in tools if not t.module.startswith("mcp:")]
except Exception:
pass
return [{"type": "function", "function": {"name": t.name, "description": t.description, "parameters": t.parameters}} for t in tools]
def mcp_tools(self) -> list[dict]:
return [{"name": t.name, "description": t.description, "inputSchema": t.parameters} for t in self.active()]
+24 -2
View File
@@ -40,6 +40,21 @@ def lan_ip() -> str:
return "127.0.0.1"
TAILSCALE = "/Applications/Tailscale.app/Contents/MacOS/Tailscale"
def tailscale() -> tuple[str, str]:
"""(indirizzo IPv4, nome MagicDNS) del Mac su Tailscale, oppure ("", "") se spento o non installato."""
import json as _j
try:
st = _j.loads(subprocess.run([TAILSCALE, "status", "--json"], capture_output=True, text=True, timeout=5).stdout or "{}")
me = st.get("Self") or {}
ip = next((a for a in me.get("TailscaleIPs") or [] if "." in a), "")
return (ip, (me.get("DNSName") or "").rstrip(".")) if st.get("BackendState") == "Running" else ("", "")
except Exception:
return "", ""
def ensure_certs(ip: str) -> tuple[Path, Path, Path]:
"""CA locale + certificato del server con SAN per IP e nome .local; rigenerato se l'IP cambia."""
CERT_DIR.mkdir(parents=True, exist_ok=True)
@@ -56,6 +71,9 @@ def ensure_certs(ip: str) -> tuple[Path, Path, Path]:
(CERT_DIR / "ca.cer").write_bytes(subprocess.run(["openssl", "x509", "-in", str(ca_crt), "-outform", "DER"], check=True, capture_output=True).stdout)
host = socket.gethostname()
san = f"IP:{ip},IP:127.0.0.1,DNS:{host},DNS:localhost"
ts_ip, ts_name = tailscale() # accesso da fuori casa: indirizzo e nome Tailscale nel certificato
if ts_ip:
san += f",IP:{ts_ip}" + (f",DNS:{ts_name}" if ts_name else "")
if not srv_crt.exists() or not srv_san.exists() or srv_san.read_text() != san:
csr = CERT_DIR / "server.csr"
ext = CERT_DIR / "server.ext"
@@ -108,7 +126,11 @@ class RemoteServer:
def urls(self) -> tuple[str, str, str, str]:
base = f"https://{self.ip}:{HTTPS_PORT}"
return base, self.new_pin(), f"{base}/?k={self.token}", f"{base} (CA: http://{self.ip}:{HTTP_PORT}/ca.cer)"
ts_ip, ts_name = tailscale()
if ts_ip:
ensure_certs(self.ip) # certificato aggiornato se Tailscale è stato acceso dopo l'avvio (vale al prossimo riavvio)
fuori = f" · fuori casa: https://{ts_name or ts_ip}:{HTTPS_PORT}" if ts_ip else " · fuori casa: accendi Tailscale"
return base, self.new_pin(), f"{base}/?k={self.token}", f"{base}{fuori} (CA: http://{self.ip}:{HTTP_PORT}/ca.cer)"
def has_clients(self) -> bool:
now = time.time()
@@ -461,7 +483,7 @@ class RemoteServer:
# ── verso i telefoni (chiamabili da qualunque thread) ──────────────────
def broadcast(self, payload: dict) -> None:
if not (self._clients or self._sse) or not self._loop:
if not self.has_clients() or not self._loop:
return
data = json.dumps(payload, ensure_ascii=False)
+6
View File
@@ -57,6 +57,12 @@ DEFAULTS: dict[str, Any] = {
"abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria
"umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente
"io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate
"frasi_attesa": True, # "Un attimo che controllo…" se la risposta tarda più di ~1,3 s
"riserva_cloud": True,
"mcp_locali": False, # strumenti dei server MCP esterni anche per i modelli locali (prompt molto più lungo) # se il motore locale fallisce, la risposta la dà Claude
"documenti_enabled": True, # indicizzazione oraria dei documenti per documenti_cerca
"documenti_cartelle": "~/Documents, ~/Desktop",
"documenti_escludi": "Progetti2026", # nomi di cartelle da saltare (codice, dati sensibili…)
"qwen_voce": "luza_voce",
"qwen_modello": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-8bit", # oppure ...-Base-bf16 (più lento, qualità piena) # voce clonata (file in data/voices) per il motore Qwen3-TTS
"ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth
+15 -1
View File
@@ -127,8 +127,13 @@ class SettingsDialog(QDialog):
self.mlx_model.addItems(cached_models() or [DEFAULT_MODEL])
self.mlx_model.setEditText(s.get("mlx_model") or DEFAULT_MODEL)
f.addRow("Modello (repo Hugging Face mlx-community)", self.mlx_model)
from .engines.mlx_engine import memory_check
self.mlx_mem = _note(memory_check(self.mlx_model.currentText())); f.addRow("", self.mlx_mem)
self.mlx_model.currentTextChanged.connect(lambda t: self.mlx_mem.setText(memory_check(t.strip())))
self.mlx_thinking = _combo([("auto", "Automatico (acceso solo dove serve, es. Spark)"), ("off", "Spento: risposte rapide"), ("on", "Acceso: più lento, meglio su domande complesse")], s.get("mlx_thinking") or "auto")
f.addRow("Ragionamento", self.mlx_thinking)
self.riserva = QCheckBox("Se il modello locale fallisce, rispondi con Claude"); self.riserva.setChecked(bool(s.get("riserva_cloud", True)))
f.addRow("Riserva cloud", self.riserva)
hint = QLabel("Elenco: modelli già scaricati. Un nome nuovo viene scaricato al primo uso. "
"Con 32 GB di RAM: modelli fino a ~20 GB in 4 bit (Qwen3-30B-A3B è veloce e supporta gli strumenti). "
"Non tenere aperto anche VLLMac con lo stesso modello.")
@@ -355,6 +360,14 @@ class SettingsDialog(QDialog):
form_sp.addRow("Client secret", self.sp_sec)
add(pg_tool, form_sp)
# ── Documenti interrogabili ───────────────────────────────────────
form_doc = QFormLayout()
form_doc.addRow(_note("Documenti interrogabili: LuZa legge il contenuto di PDF, Word e testi in queste cartelle e risponde a domande su di essi. Indice locale, aggiornato ogni ora."))
self.doc_on = QCheckBox("Attivo"); self.doc_on.setChecked(bool(s.get("documenti_enabled", True))); form_doc.addRow("Indicizzazione", self.doc_on)
self.doc_dirs = QLineEdit(str(s.get("documenti_cartelle") or "")); form_doc.addRow("Cartelle (separate da virgola)", self.doc_dirs)
self.doc_skip = QLineEdit(str(s.get("documenti_escludi") or "")); form_doc.addRow("Sottocartelle da escludere", self.doc_skip)
add(pg_tool, form_doc)
btns = QHBoxLayout(); btns.addStretch()
cancel = QPushButton("Annulla"); cancel.clicked.connect(self.reject); btns.addWidget(cancel)
save = QPushButton("Salva"); save.setObjectName("primary"); save.clicked.connect(self._save); btns.addWidget(save)
@@ -642,7 +655,7 @@ class SettingsDialog(QDialog):
values = {
"provider": self.provider.currentData(), "effort": self.effort.currentData(),
"local_base_url": self.local_url.text().strip().rstrip("/"), "local_model": self.local_model.currentText().strip(),
"mlx_model": self.mlx_model.currentText().strip(), "mlx_thinking": self.mlx_thinking.currentData() or "auto",
"mlx_model": self.mlx_model.currentText().strip(), "mlx_thinking": self.mlx_thinking.currentData() or "auto", "riserva_cloud": self.riserva.isChecked(),
"search_api_key": self.search_key.text().strip(),
"claudecode_model": self.cc_model.currentData(), "claudecode_access": self.cc_access.currentData(),
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
@@ -671,6 +684,7 @@ class SettingsDialog(QDialog):
"ipixel_enabled": self.px_on.isChecked(), "ipixel_stati": self.px_stati.isChecked(), "ipixel_notifiche": self.px_notif.isChecked(),
"ipixel_luminosita": int(self.px_lum.value()), "ipixel_intermezzi_min": int(self.px_fun.currentData() or 0), "ipixel_musica": self.px_mus.isChecked(),
"spotify_client_id": self.sp_id.text().strip(),
"documenti_enabled": self.doc_on.isChecked(), "documenti_cartelle": self.doc_dirs.text().strip(), "documenti_escludi": self.doc_skip.text().strip(),
"immagini_famiglia": self.img_fam.currentData() or "z-image-turbo", "immagini_modello": self.img_model.currentText().strip(),
}
if self.ha_token.text().strip():
+66 -6
View File
@@ -1,11 +1,13 @@
"""Ricerca web con Brave Search (per il motore locale)."""
"""Ricerca web per i motori locali: Brave (se c'è la chiave), altrimenti Bing e DuckDuckGo via HTML, senza chiavi né server."""
from __future__ import annotations
import html
import re
import requests
def brave_search(query: str, api_key: str, count: int = 6) -> list[dict]:
res = requests.get(
"https://api.search.brave.com/res/v1/web/search",
@@ -17,9 +19,67 @@ def brave_search(query: str, api_key: str, count: int = 6) -> list[dict]:
hits = []
for r in (res.json().get("web") or {}).get("results") or []:
if r.get("url"):
hits.append({
"title": r.get("title") or r["url"],
"url": r["url"],
"description": re.sub(r"<[^>]+>", "", r.get("description") or ""),
})
hits.append({"title": r.get("title") or r["url"], "url": r["url"],
"description": re.sub(r"<[^>]+>", "", r.get("description") or "")})
return hits
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_0) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Safari/605.1.15",
"Accept-Language": "it-IT,it;q=0.9"}
def _clean(t: str) -> str:
return " ".join(html.unescape(re.sub(r"<[^>]+>", "", t)).split())
def bing_search(query: str, count: int = 6) -> list[dict]:
r = requests.get("https://www.bing.com/search", params={"q": query, "setlang": "it", "cc": "IT"}, headers=UA, timeout=15)
r.raise_for_status()
hits = []
for block in re.findall(r'<li class="b_algo".*?</li>', r.text, re.S):
a = re.search(r'<h2[^>]*>\s*<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', block, re.S)
if not a or not a.group(1).startswith("http"):
continue
snip = re.search(r'<p[^>]*>(.*?)</p>', block, re.S)
hits.append({"title": _clean(a.group(2)), "url": html.unescape(a.group(1)), "description": _clean(snip.group(1)) if snip else ""})
if len(hits) >= count:
break
return hits
def ddg_search(query: str, count: int = 6) -> list[dict]:
r = requests.post("https://html.duckduckgo.com/html/", data={"q": query, "kl": "it-it"}, timeout=15, headers=UA)
r.raise_for_status()
hits = []
for m in re.finditer(r'class="result__a" href="([^"]+)"[^>]*>(.*?)</a>.*?class="result__snippet"[^>]*>(.*?)</a>', r.text, re.S):
url = html.unescape(m.group(1))
if "uddg=" in url:
url = requests.utils.unquote(url.split("uddg=")[1].split("&")[0])
hits.append({"title": html.unescape(re.sub(r"<[^>]+>", "", m.group(2))), "url": url,
"description": html.unescape(re.sub(r"<[^>]+>", "", m.group(3)))})
if len(hits) >= count:
break
return hits
def web_search(query: str, api_key: str = "", count: int = 6) -> list[dict]:
"""Brave (con chiave) → Bing → DuckDuckGo, senza chiavi né server. Errore solo se falliscono tutti."""
errors = []
for name, fn in (("Brave", (lambda: brave_search(query, api_key, count)) if api_key else None),
("Bing", lambda: bing_search(query, count)), ("DuckDuckGo", lambda: ddg_search(query, count))):
if fn is None:
continue
try:
hits = fn()
if hits:
return hits
errors.append(f"{name}: nessun risultato")
except Exception as e:
errors.append(f"{name}: {e}")
raise RuntimeError("; ".join(errors))
if __name__ == "__main__":
hits = web_search("meteo Ortona domani")
assert hits and hits[0]["url"].startswith("http") and hits[0]["title"], hits
print(len(hits), "risultati:", hits[0]["title"][:60], "|", hits[0]["description"][:80])
+58
View File
@@ -0,0 +1,58 @@
"""Consumi: token e costi per giorno e per motore, dal registro data/consumi.jsonl scritto a ogni risposta."""
from __future__ import annotations
import json
import sys
from collections import defaultdict
from datetime import datetime, timedelta
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from avatar.settings import DATA_DIR # noqa: E402
LOG = DATA_DIR / "consumi.jsonl"
def _n(x: int) -> str:
return f"{x:,}".replace(",", ".")
def riepilogo(giorni: int = 7) -> str:
since = (datetime.now() - timedelta(days=giorni - 1)).strftime("%Y-%m-%d")
try:
rows = [json.loads(l) for l in LOG.read_text(encoding="utf-8").splitlines() if l.strip()]
except FileNotFoundError:
rows = []
rows = [r for r in rows if r["ts"][:10] >= since]
if not rows:
return f"Nessun consumo registrato negli ultimi {giorni} giorni (il conteggio parte da oggi)."
per_m = defaultdict(lambda: [0, 0, 0, 0.0, False]); per_g = defaultdict(lambda: [0, 0.0])
for r in rows:
m = per_m[r["motore"]]; m[0] += 1; m[1] += r["input"]; m[2] += r["output"]; m[3] += r["costo"]; m[4] = m[4] or r.get("abbonamento", False)
g = per_g[r["ts"][:10]]; g[0] += r["input"] + r["output"]; g[1] += 0 if r.get("abbonamento") else r["costo"]
out = [f"Consumi ultimi {giorni} giorni:"]
for mot, (n, i, o, c, abb) in sorted(per_m.items(), key=lambda kv: -(kv[1][1] + kv[1][2])):
costo = "gratis (locale)" if mot.startswith("locale") else (f"≈ ${c:.2f} equivalente API, incluso nell'abbonamento" if abb else f"≈ ${c:.2f}")
out.append(f"- {mot}: {n} risposte, {_n(i)} token in ingresso, {_n(o)} in uscita, {costo}")
out.append("Per giorno: " + "; ".join(f"{g[8:10]}/{g[5:7]} {_n(t)} token" + (f" (${c:.2f})" if c else "") for g, (t, c) in sorted(per_g.items())))
return "\n".join(out)
def t_riepilogo(params: dict, ctx: dict) -> str:
return riepilogo(max(1, min(int(params.get("giorni") or 7), 90)))
TOOLS = [
{"name": "consumi_riepilogo", "description": "Token e costi delle risposte di LuZa per motore (locale, Claude Code, Claude API) e per giorno, negli ultimi N giorni (default 7). Per 'quanto ho speso?', 'quanti token ho usato?'.",
"parameters": {"type": "object", "properties": {"giorni": {"type": "integer"}}}, "run": t_riepilogo},
]
if __name__ == "__main__":
import tempfile
LOG = Path(tempfile.mkdtemp()) / "c.jsonl"
now = datetime.now().strftime("%Y-%m-%d %H:%M")
LOG.write_text("\n".join(json.dumps(r) for r in [
{"ts": now, "motore": "locale (gemma)", "input": 100, "output": 50, "costo": 0, "abbonamento": False},
{"ts": now, "motore": "claude api (x)", "input": 1000, "output": 200, "costo": 0.01, "abbonamento": False}]) + "\n")
out = riepilogo(7); assert "gratis (locale)" in out and "$0.01" in out, out; print(out)
+259
View File
@@ -0,0 +1,259 @@
"""Documenti interrogabili: indicizza le cartelle scelte (PDF, Word, testo, pagine…) in pezzi da ~800 caratteri,
con ricerca ibrida per parole (SQLite FTS5) e per significato (embeddings multilingual-e5-small). Tutto locale.
L'indice si aggiorna da solo ogni ora, solo per i file nuovi o modificati."""
from __future__ import annotations
import os
import re
import sqlite3
import sys
import threading
import time
from pathlib import Path
import numpy as np
sys.path.insert(0, str(Path(__file__).resolve().parent))
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
from avatar.settings import DATA_DIR, Settings # noqa: E402
DB = DATA_DIR / "documenti.sqlite"
EMB_MODEL = "intfloat/multilingual-e5-small"
EXT = {".pdf", ".docx", ".doc", ".txt", ".md", ".rtf", ".odt", ".pages", ".html", ".csv"}
SKIP_DIRS = {"node_modules", ".git", "Library", ".venv", "venv", "__pycache__", "build", "dist", ".Trash"}
MAX_MB = 30
_emb: dict = {}
_stato = {"in_corso": False, "ultimo": "", "errore": ""}
def _cartelle() -> list[Path]:
raw = str(Settings().get("documenti_cartelle") or "~/Documents, ~/Desktop")
return [Path(c.strip()).expanduser() for c in raw.split(",") if c.strip()]
def _con() -> sqlite3.Connection:
DB.parent.mkdir(parents=True, exist_ok=True)
con = sqlite3.connect(DB, timeout=30)
con.executescript("""
CREATE TABLE IF NOT EXISTS files (path TEXT PRIMARY KEY, mtime REAL, chunks INTEGER);
CREATE TABLE IF NOT EXISTS chunks (id INTEGER PRIMARY KEY, path TEXT, n INTEGER, text TEXT, emb BLOB);
CREATE INDEX IF NOT EXISTS chunks_path ON chunks(path);
CREATE VIRTUAL TABLE IF NOT EXISTS fts USING fts5(text, content='chunks', content_rowid='id', tokenize='unicode61 remove_diacritics 2');
CREATE TRIGGER IF NOT EXISTS chunks_ai AFTER INSERT ON chunks BEGIN INSERT INTO fts(rowid, text) VALUES (new.id, new.text); END;
CREATE TRIGGER IF NOT EXISTS chunks_ad AFTER DELETE ON chunks BEGIN INSERT INTO fts(fts, rowid, text) VALUES ('delete', old.id, old.text); END;
""")
return con
def _embed(texts: list[str], query: bool = False) -> np.ndarray:
"""Vettori normalizzati (384 dimensioni); prefissi 'query:'/'passage:' richiesti dal modello e5."""
import torch
from transformers import AutoModel, AutoTokenizer
if not _emb:
_emb["tok"] = AutoTokenizer.from_pretrained(EMB_MODEL)
dev = "mps" if torch.backends.mps.is_available() else "cpu" # GPU del Mac: ~10× più veloce della CPU
_emb["model"] = AutoModel.from_pretrained(EMB_MODEL).eval().to(dev)
_emb["dev"] = dev
tok, model = _emb["tok"], _emb["model"]
out = []
for i in range(0, len(texts), 32):
batch = [("query: " if query else "passage: ") + t for t in texts[i:i + 32]]
enc = tok(batch, padding=True, truncation=True, max_length=512, return_tensors="pt").to(_emb["dev"])
with torch.no_grad():
h = model(**enc).last_hidden_state
m = enc["attention_mask"].unsqueeze(-1).float()
v = (h * m).sum(1) / m.sum(1)
out.append(torch.nn.functional.normalize(v, dim=1).float().cpu().numpy().astype(np.float32))
return np.concatenate(out) if out else np.zeros((0, 384), np.float32)
def _testo(path: Path) -> str:
import file as fileplugin # stesso estrattore del plugin file (pdf, docx, testo, textutil per gli altri)
return fileplugin._read(path)
def _pezzi(testo: str, size: int = 800, overlap: int = 150) -> list[str]:
testo = re.sub(r"[ \t]+", " ", re.sub(r"\n{3,}", "\n\n", testo)).strip()
out, i = [], 0
while i < len(testo):
out.append(testo[i:i + size]); i += size - overlap
return [p for p in out if len(p.strip()) > 40]
def _escludi() -> set[str]:
raw = str(Settings().get("documenti_escludi") or "Progetti2026")
return {x.strip() for x in raw.split(",") if x.strip()}
def _files():
skip = SKIP_DIRS | _escludi()
for root in _cartelle():
if not root.exists():
continue
for dirpath, dirnames, filenames in os.walk(root):
dirnames[:] = [d for d in dirnames if d not in skip and not d.startswith(".") and not d.endswith(".app")]
for f in filenames:
p = Path(dirpath) / f
if p.suffix.lower() in EXT and not f.startswith("~$"):
try:
if p.stat().st_size <= MAX_MB * 1e6:
yield p
except OSError:
pass
def aggiorna(log=None) -> str:
"""Indicizzazione incrementale: aggiunge i file nuovi/modificati, toglie quelli spariti."""
import fcntl
DB.parent.mkdir(parents=True, exist_ok=True)
lockf = open(DATA_DIR / "documenti.lock", "w")
try: # blocco tra processi: app e indicizzazioni manuali non si sovrappongono
fcntl.flock(lockf, fcntl.LOCK_EX | fcntl.LOCK_NB)
except OSError:
lockf.close()
return "Indicizzazione già in corso."
_stato["in_corso"] = True
try:
con = _con()
noti = dict(con.execute("SELECT path, mtime FROM files").fetchall())
visti, nuovi, pezzi_tot = set(), 0, 0
for p in _files():
sp = str(p); visti.add(sp)
mt = p.stat().st_mtime
if noti.get(sp) == mt:
continue
try:
pezzi = _pezzi(_testo(p))[:400]
except Exception:
pezzi = []
vec = _embed(pezzi) if pezzi else np.zeros((0, 384), np.float32)
con.execute("DELETE FROM chunks WHERE path=?", (sp,))
con.executemany("INSERT INTO chunks(path, n, text, emb) VALUES (?,?,?,?)", [(sp, i, t, vec[i].tobytes()) for i, t in enumerate(pezzi)])
con.execute("INSERT OR REPLACE INTO files(path, mtime, chunks) VALUES (?,?,?)", (sp, mt, len(pezzi)))
con.commit()
nuovi += 1; pezzi_tot += len(pezzi)
if log and nuovi % 25 == 0:
log(f"[documenti] indicizzati {nuovi} file…")
spariti = [p for p in noti if p not in visti]
for sp in spariti:
con.execute("DELETE FROM chunks WHERE path=?", (sp,)); con.execute("DELETE FROM files WHERE path=?", (sp,))
con.commit()
n_files, n_chunks = con.execute("SELECT COUNT(*), COALESCE(SUM(chunks),0) FROM files").fetchone()
con.close()
_stato["ultimo"] = time.strftime("%d/%m %H:%M")
return f"Indice aggiornato: {nuovi} file nuovi o modificati, {len(spariti)} rimossi. Totale {n_files} documenti, {n_chunks} passaggi."
finally:
_stato["in_corso"] = False
fcntl.flock(lockf, fcntl.LOCK_UN); lockf.close()
def cerca_testo(domanda: str, k: int = 6) -> list[dict]:
con = _con()
rows = con.execute("SELECT id, path, text, emb FROM chunks").fetchall()
if not rows:
con.close(); return []
ids = np.array([r[0] for r in rows]); M = np.frombuffer(b"".join(r[3] for r in rows), dtype=np.float32).reshape(len(rows), -1)
q = _embed([domanda], query=True)[0]
sem = M @ q # coseno (vettori normalizzati)
score = {int(i): float(s) for i, s in zip(ids, sem)}
words = [w for w in re.findall(r"\w{3,}", domanda.lower())]
if words:
try:
for (rid,) in con.execute("SELECT rowid FROM fts WHERE fts MATCH ? ORDER BY rank LIMIT 30", (" OR ".join(w + "*" for w in words),)):
score[rid] = score.get(rid, 0) + 0.15 # bonus per le parole esatte
except sqlite3.OperationalError:
pass
best = sorted(score.items(), key=lambda kv: -kv[1])[:k * 3]
by_id = {r[0]: r for r in rows}
out, per_file = [], {}
for rid, sc in best:
_i, path, text, _e = by_id[rid]
if per_file.get(path, 0) >= 2:
continue
per_file[path] = per_file.get(path, 0) + 1
out.append({"path": path, "testo": text, "score": round(sc, 3)})
if len(out) >= k:
break
con.close()
return out
# ── strumenti ────────────────────────────────────────────────────────────────
def t_cerca(params: dict, ctx: dict) -> str:
domanda = str(params.get("domanda") or "").strip()
if not domanda:
return "Errore: serve la domanda."
if not DB.exists():
return "Non ho ancora indicizzato i documenti: usa documenti_aggiorna (ci vuole qualche minuto la prima volta)."
res = cerca_testo(domanda, max(3, min(int(params.get("quanti") or 6), 12)))
if not res:
return "Nessun passaggio pertinente nei documenti indicizzati."
home = str(Path.home())
righe = [f"[{i + 1}] {r['path'].replace(home, '~')} (pertinenza {r['score']}):\n{r['testo']}" for i, r in enumerate(res)]
return ("Passaggi più pertinenti dai documenti dell'utente. Rispondi basandoti su questi, cita il nome del file, "
"e di' chiaramente se l'informazione non c'è:\n\n" + "\n\n".join(righe))
def t_aggiorna(params: dict, ctx: dict) -> str:
p = ctx.get("player")
if p is not None and hasattr(p, "set_state"):
p.set_state("PROCESSING · indicizzo i documenti")
return aggiorna(ctx.get("log"))
def t_stato(params: dict, ctx: dict) -> str:
if not DB.exists():
return "Indice non ancora creato. Cartelle: " + ", ".join(str(c) for c in _cartelle())
con = _con(); n_files, n_chunks = con.execute("SELECT COUNT(*), COALESCE(SUM(chunks),0) FROM files").fetchone(); con.close()
return (f"Documenti indicizzati: {n_files} file, {n_chunks} passaggi. Cartelle: {', '.join(str(c) for c in _cartelle())}. "
+ ("Indicizzazione in corso." if _stato["in_corso"] else f"Ultimo aggiornamento: {_stato['ultimo'] or 'all\'avvio'}."))
_proc: dict = {"p": None}
def pausa(attiva: bool) -> None:
"""Sospende/riprende l'indicizzazione mentre LuZa risponde: niente contesa di GPU e CPU con il modello."""
import signal
p = _proc["p"]
if p is not None and p.poll() is None:
try:
p.send_signal(signal.SIGSTOP if attiva else signal.SIGCONT)
except Exception:
pass
def _servizio() -> None:
"""Ogni ora indicizza in un processo separato a priorità bassa (niente blocchi del GIL nell'app)."""
import subprocess
time.sleep(120)
while True:
if Settings().get("documenti_enabled", True):
code = ("import sys; sys.argv=['memory_mcp.py']; sys.path.insert(0, %r); import documenti_rag as d; print('[documenti]', d.aggiorna(), flush=True)"
% str(Path(__file__).resolve().parent))
_proc["p"] = subprocess.Popen(["nice", "-n", "15", sys.executable, "-c", code], cwd=str(Path(__file__).resolve().parent.parent))
_proc["p"].wait()
_proc["p"] = None
time.sleep(3600)
if Path(sys.argv[0]).name != "memory_mcp.py": # solo nell'app
threading.Thread(target=_servizio, daemon=True, name="documenti").start()
TOOLS = [
{"name": "documenti_cerca", "description": "Cerca nel CONTENUTO dei documenti dell'utente (PDF, Word, testi nelle cartelle indicizzate: Documenti, Scrivania) per rispondere a domande come 'cosa dice il contratto sulle penali?', 'quando scade l'assicurazione?', 'trova la ricevuta del dentista'. Restituisce i passaggi più pertinenti con il file di provenienza.",
"parameters": {"type": "object", "properties": {"domanda": {"type": "string"}, "quanti": {"type": "integer"}}, "required": ["domanda"]}, "run": t_cerca},
{"name": "documenti_aggiorna", "description": "Aggiorna subito l'indice dei documenti (file nuovi o modificati). Normalmente avviene da solo ogni ora.",
"parameters": {"type": "object", "properties": {}}, "run": t_aggiorna},
{"name": "documenti_stato", "description": "Quanti documenti sono indicizzati, quali cartelle e quando è stato l'ultimo aggiornamento.",
"parameters": {"type": "object", "properties": {}}, "run": t_stato},
]
if __name__ == "__main__":
assert _pezzi("a" * 2000)[1].startswith("a") and len(_pezzi("a" * 2000)) == 4
v = _embed(["Il contratto prevede una penale del 5% per ritardo."]); q = _embed(["quanto è la penale?"], query=True)[0]
w = _embed(["La ricetta della carbonara usa guanciale."])
assert float(v[0] @ q) > float(w[0] @ q), "il significato deve battere il testo non pertinente"
print("ok: pertinente", round(float(v[0] @ q), 3), "> non pertinente", round(float(w[0] @ q), 3))