Ricerca web senza chiavi (Bing/DuckDuckGo), documenti interrogabili locali, riserva cloud, consumi, avviso memoria, Tailscale, frasi d'attesa; motore interno più leggero (niente MCP esterni) e voce senza scatti
Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
This commit is contained in:
1 parent
54895586c8
commit
0d4f4b7b37
13 files changed
+630
-28
No files matched your search
+117
-6
@@ -3,6 +3,7 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import queue
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
@@ -56,6 +57,29 @@ def _diario(testo: str, strumenti: list[str], origine: str) -> None:
|
||||
pass
|
||||
|
||||
|
||||
PREZZI = {"claude api": (5.0, 25.0)} # $ per milione di token (input, output), Opus via API; ponytail: tabella fissa, aggiornare se cambiano i listini
|
||||
|
||||
|
||||
def _consumi(ev: dict) -> None:
|
||||
"""Riga in data/consumi.jsonl: motore, token in/out, costo (dichiarato da Claude Code o stimato per l'API)."""
|
||||
try:
|
||||
import json as _j
|
||||
from avatar.settings import DATA_DIR as _D
|
||||
costo = ev.get("costo")
|
||||
if costo is None:
|
||||
pin, pout = next((v for k, v in PREZZI.items() if str(ev.get("motore", "")).startswith(k)), (0.0, 0.0))
|
||||
costo = (ev.get("input", 0) * pin + ev.get("output", 0) * pout) / 1e6
|
||||
with open(_D / "consumi.jsonl", "a", encoding="utf-8") as f:
|
||||
f.write(_j.dumps({"ts": time.strftime("%Y-%m-%d %H:%M"), "motore": ev.get("motore", "?"), "input": ev.get("input", 0),
|
||||
"output": ev.get("output", 0), "costo": round(float(costo), 5), "abbonamento": bool(ev.get("abbonamento"))}) + "\n")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
FRASI_ATTESA = ["Un attimo che controllo.", "Fammi verificare.", "Dammi un secondo.", "Vediamo un po'.", "Ci guardo subito.",
|
||||
"Aspetta un attimo, controllo.", "Mmh, fammi vedere.", "Un momento.", "Ok, ci penso un attimo.", "Fammi dare un'occhiata."]
|
||||
|
||||
|
||||
class Assistant:
|
||||
def __init__(self, ui, settings: Settings) -> None:
|
||||
self.ui = ui
|
||||
@@ -168,6 +192,7 @@ class Assistant:
|
||||
except Exception as err:
|
||||
self.ui.write_log(f"ERR: Riconoscimento vocale non disponibile — {err}")
|
||||
self.reopen_audio()
|
||||
self._prepare_fillers()
|
||||
self._ensure_engine()
|
||||
self._start_whatsapp_bridge()
|
||||
try:
|
||||
@@ -223,6 +248,18 @@ class Assistant:
|
||||
self._engine = AnthropicEngine(api_key, self.name, self.user_name, s.get("effort"))
|
||||
return self._engine
|
||||
|
||||
def _fallback_engine(self):
|
||||
"""Riserva cloud per i motori locali: Claude Code (abbonamento), o l'API se c'è la chiave. Creata una volta."""
|
||||
s = self.settings
|
||||
if getattr(self, "_backup", None) is None:
|
||||
if s.get_secret("anthropic_api_key"):
|
||||
self._backup = AnthropicEngine(s.get_secret("anthropic_api_key"), self.name, self.user_name, s.get("effort"))
|
||||
else:
|
||||
self._backup = ClaudeCodeEngine(s.get("claudecode_model") or "sonnet", s.get("claudecode_access") or "chat",
|
||||
s.get("claudecode_path") or "", s.get("claudecode_config_dir") or "",
|
||||
self.name, self.user_name, s.get("effort"))
|
||||
return self._backup
|
||||
|
||||
def reconfigure(self) -> None:
|
||||
"""Dopo un salvataggio delle impostazioni."""
|
||||
self.interrupt(silent=True)
|
||||
@@ -241,6 +278,43 @@ class Assistant:
|
||||
self.ui.write_log(f"ERR: Avatar 3D — {err}")
|
||||
self.ui.write_log(f"SYS: Impostazioni applicate — motore {self.settings.get('provider')}.")
|
||||
|
||||
def _prepare_fillers(self) -> None:
|
||||
"""Sintetizza una volta le frasi d'attesa con la voce attuale: al bisogno partono subito, senza attendere la sintesi."""
|
||||
voice = self.voice
|
||||
self._fillers = []
|
||||
|
||||
def work():
|
||||
out = []
|
||||
for f in FRASI_ATTESA:
|
||||
try:
|
||||
a = voice.synthesize(f)
|
||||
if len(a):
|
||||
out.append((f, a))
|
||||
except Exception:
|
||||
break
|
||||
if voice is self.voice:
|
||||
self._fillers = out
|
||||
threading.Thread(target=work, daemon=True, name="frasi-attesa").start()
|
||||
|
||||
def _maybe_filler(self, turn: int, answered: list) -> None:
|
||||
"""Dopo ~1,3 s senza risposta, una frase d'attesa breve e diversa dalla precedente."""
|
||||
time.sleep(1.3)
|
||||
if answered[0] or turn != self._turn or self._speaking or not self.settings.get("frasi_attesa", True):
|
||||
return
|
||||
choices = [x for x in getattr(self, "_fillers", []) if x[0] != getattr(self, "_last_filler", "")]
|
||||
if not choices:
|
||||
return
|
||||
import random
|
||||
text, audio = random.choice(choices)
|
||||
self._last_filler = text
|
||||
remote = getattr(self, "remote", None)
|
||||
if remote is not None and remote.has_clients():
|
||||
try:
|
||||
remote.on_audio(text, audio)
|
||||
except Exception:
|
||||
pass
|
||||
self._audio_q.put((turn, text, audio))
|
||||
|
||||
def reload_voice(self, from_customise: bool = False) -> None:
|
||||
"""Ricarica la voce. `from_customise`: la scelta arriva dal pannello Customise
|
||||
(Sara / Nicola / Sistema) e va copiata nelle impostazioni; altrimenti comandano
|
||||
@@ -269,6 +343,7 @@ class Assistant:
|
||||
self.voice = new
|
||||
self.ui.write_log(f"SYS: Voce: {getattr(new, 'voice', '') or 'sistema'} ({'Kokoro' if self.settings.get('tts_engine') == 'kokoro' else 'macOS'}).")
|
||||
threading.Thread(target=self._safe_load_voice, daemon=True).start()
|
||||
self._prepare_fillers()
|
||||
|
||||
def _safe_load_voice(self) -> None:
|
||||
try:
|
||||
@@ -353,11 +428,17 @@ class Assistant:
|
||||
splitter = SentenceSplitter()
|
||||
|
||||
used: list[str] = []
|
||||
failed = [""]
|
||||
answered = [False]
|
||||
threading.Thread(target=self._maybe_filler, args=(turn, answered), daemon=True).start()
|
||||
fallback_ok = [self.settings.get("provider") in ("mlx", "local") and bool(self.settings.get("riserva_cloud", True))]
|
||||
|
||||
def emit(ev: dict) -> None:
|
||||
if turn != self._turn:
|
||||
return
|
||||
t = ev.get("type")
|
||||
if t == "usage":
|
||||
_consumi(ev); return
|
||||
if t == "status" and ev.get("status") == "working" and ev.get("detail"):
|
||||
used.append(str(ev["detail"]))
|
||||
if t in ("done", "error"):
|
||||
@@ -380,6 +461,7 @@ class Assistant:
|
||||
elif ev["status"] == "loading" and detail:
|
||||
self.ui.write_log(f"SYS: {detail}…")
|
||||
elif t == "text":
|
||||
answered[0] = True
|
||||
for s in splitter.push(ev["delta"]):
|
||||
self._speech_q.put((turn, s))
|
||||
elif t == "memory_saved":
|
||||
@@ -393,6 +475,9 @@ class Assistant:
|
||||
if ev.get("sources"):
|
||||
self.ui.show_content("Fonti", "\n".join(f"• {s['title']}\n {s['url']}" for s in ev["sources"]))
|
||||
elif t == "error":
|
||||
if not ev.get("aborted") and fallback_ok[0]:
|
||||
failed[0] = ev["message"] # motore locale fallito: la riserva cloud ripete la domanda
|
||||
return
|
||||
if not ev.get("aborted"):
|
||||
self.ui.write_log(f"ERR: {ev['message']}")
|
||||
self._speech_q.put((turn, clean_for_speech("Scusa, c'è stato un problema: " + ev["message"])[:300]))
|
||||
@@ -414,8 +499,16 @@ class Assistant:
|
||||
attach_path = str(p)
|
||||
except Exception as err:
|
||||
self.ui.write_log(f"ERR: Allegato — {err}")
|
||||
doc = sys.modules.get("avatar_plugins.documenti_rag")
|
||||
if doc is not None:
|
||||
doc.pausa(True)
|
||||
try:
|
||||
engine.send(engine_text, emit, self._abort, image=image, attach_path=attach_path)
|
||||
if failed[0] and not self._abort.is_set():
|
||||
backup = self._fallback_engine()
|
||||
fallback_ok[0] = False
|
||||
self.ui.write_log(f"SYS: Motore locale in errore ({failed[0][:80]}): rispondo con Claude.")
|
||||
backup.send(engine_text, emit, self._abort, image=image, attach_path=attach_path)
|
||||
finally:
|
||||
self._busy = False
|
||||
self._speech_q.put((turn, END))
|
||||
@@ -431,18 +524,33 @@ class Assistant:
|
||||
continue
|
||||
if hasattr(self.voice, "stream"): # voce in streaming: i pezzi partono mentre la frase è ancora in generazione
|
||||
try:
|
||||
first = True
|
||||
for chunk in self.voice.stream(item, cancel=lambda t=turn: t != self._turn):
|
||||
if turn != self._turn:
|
||||
break
|
||||
first, buf = True, []
|
||||
lead = getattr(self, "_tts_lead", 1.0) # secondi d'audio accumulati prima di iniziare a parlare
|
||||
|
||||
def push(audio):
|
||||
nonlocal first
|
||||
remote = getattr(self, "remote", None)
|
||||
if remote is not None and remote.has_clients():
|
||||
try:
|
||||
remote.on_audio(item if first else "", chunk)
|
||||
remote.on_audio(item if first else "", audio)
|
||||
except Exception:
|
||||
pass
|
||||
self._audio_q.put((turn, item if first else "", chunk))
|
||||
self._audio_q.put((turn, item if first else "", audio))
|
||||
first = False
|
||||
|
||||
for chunk in self.voice.stream(item, cancel=lambda t=turn: t != self._turn):
|
||||
if turn != self._turn:
|
||||
break
|
||||
if buf is not None:
|
||||
buf.append(chunk)
|
||||
if sum(len(c) for c in buf) / 24000 >= lead:
|
||||
push(np.concatenate(buf)); buf = None
|
||||
continue
|
||||
if self._speaking and self.player._cursor < time.time(): # il lettore è rimasto a secco: scatto
|
||||
self._tts_lead = min(4.0, getattr(self, "_tts_lead", 1.0) + 0.5)
|
||||
push(chunk)
|
||||
if buf:
|
||||
push(np.concatenate(buf))
|
||||
except Exception as err:
|
||||
self.ui.write_log(f"ERR: Sintesi vocale — {err}")
|
||||
continue
|
||||
@@ -468,6 +576,9 @@ class Assistant:
|
||||
if text is END:
|
||||
self.player.drain()
|
||||
self._speaking = False
|
||||
doc = sys.modules.get("avatar_plugins.documenti_rag")
|
||||
if doc is not None and not self._busy:
|
||||
doc.pausa(False)
|
||||
self._tail_until = time.monotonic() + 0.5
|
||||
remote = getattr(self, "remote", None)
|
||||
if remote is not None:
|
||||
|
||||
@@ -137,6 +137,10 @@ class AnthropicEngine:
|
||||
full += event.delta.text
|
||||
emit({"type": "text", "delta": event.delta.text})
|
||||
message = stream.get_final_message()
|
||||
u = getattr(message, "usage", None)
|
||||
if u is not None:
|
||||
inp = int(getattr(u, "input_tokens", 0) or 0) + int(getattr(u, "cache_read_input_tokens", 0) or 0) + int(getattr(u, "cache_creation_input_tokens", 0) or 0)
|
||||
emit({"type": "usage", "motore": f"claude api ({MODEL})", "input": inp, "output": int(getattr(u, "output_tokens", 0) or 0)})
|
||||
except anthropic.BadRequestError as err:
|
||||
if self._compaction and not full:
|
||||
print(f"[Claude] compattazione rifiutata, la disattivo: {err.message}")
|
||||
|
||||
@@ -206,6 +206,10 @@ class ClaudeCodeEngine:
|
||||
emit({"type": "status", "status": "thinking"})
|
||||
elif t == "result":
|
||||
out["result"] = msg.get("result") or ""
|
||||
u = msg.get("usage") or {}
|
||||
emit({"type": "usage", "motore": f"claude code ({self.model})",
|
||||
"input": int(u.get("input_tokens", 0)) + int(u.get("cache_read_input_tokens", 0)) + int(u.get("cache_creation_input_tokens", 0)),
|
||||
"output": int(u.get("output_tokens", 0)), "costo": float(msg.get("total_cost_usd") or 0), "abbonamento": True})
|
||||
if msg.get("is_error") or (msg.get("subtype") and msg["subtype"] != "success"):
|
||||
errs = msg.get("errors") or []
|
||||
out["error"] = "; ".join(map(str, errs)) or out["result"] or msg.get("subtype")
|
||||
|
||||
@@ -15,7 +15,7 @@ from pathlib import Path
|
||||
|
||||
from avatar.memory_tools import memory_prompt, openai_tools, parse_args, run_tool
|
||||
from avatar.plugins import registry
|
||||
from avatar.websearch import brave_search
|
||||
from avatar.websearch import web_search
|
||||
from .base import Emit, History, Meter, compact_history, persona_text, summary_block, today_label, user_block
|
||||
from .openai_compat import SEARCH_TOOL, Aborted
|
||||
|
||||
@@ -26,7 +26,7 @@ SAMPLING = {"temp": 0.6, "top_p": 0.8, "top_k": 20} # poco sotto i valori cons
|
||||
DEBUG = bool(os.environ.get("MLX_DEBUG")) # stampa il testo grezzo generato (tag compresi)
|
||||
DEFAULT_MODEL = "mlx-community/Qwen3-30B-A3B-Instruct-2507-4bit"
|
||||
HUB = Path.home() / ".cache" / "huggingface" / "hub"
|
||||
_EXCLUDE = ("whisper", "tts", "embed", "rerank", "clip", "vl-", "-vl", "vision", "diffusion", "parakeet")
|
||||
_EXCLUDE = ("whisper", "tts", "kokoro", "embed", "rerank", "clip", "vl-", "-vl", "vision", "diffusion", "parakeet")
|
||||
|
||||
_lock = threading.Lock()
|
||||
_loaded: dict = {"name": None, "model": None, "tokenizer": None, "cache": None, "tokens": [], "snap": None}
|
||||
@@ -44,6 +44,31 @@ def cached_models() -> list[str]:
|
||||
return out
|
||||
|
||||
|
||||
def model_size_gb(name: str) -> float:
|
||||
"""Peso su disco dei pesi di un modello nella cache (≈ memoria occupata una volta caricato); 0 se non scaricato."""
|
||||
snaps = HUB / ("models--" + name.replace("/", "--")) / "snapshots"
|
||||
if not snaps.exists():
|
||||
return 0.0
|
||||
return sum(f.stat().st_size for f in snaps.rglob("*.safetensors")) / 1e9
|
||||
|
||||
|
||||
def memory_check(name: str) -> str:
|
||||
"""Avviso se modello + voce + trascrizione superano ~75% della RAM del Mac."""
|
||||
import subprocess
|
||||
ram = int(subprocess.run(["sysctl", "-n", "hw.memsize"], capture_output=True, text=True).stdout.strip() or 0) / 1024 ** 3
|
||||
gb = model_size_gb(name)
|
||||
if not gb or not ram:
|
||||
return "Peso non noto: verrà scaricato al primo uso."
|
||||
extra = 2.4 + 0.5 # voce Qwen3-TTS 8 bit + Whisper
|
||||
tot = gb + extra
|
||||
msg = f"Modello {gb:.1f} GB + voce e trascrizione ≈ {tot:.0f} GB su {ram:.0f} GB di RAM."
|
||||
if tot > ram * 0.75:
|
||||
msg += " ⚠ Troppo pesante: il Mac rallenterà e le immagini potrebbero non avere spazio."
|
||||
elif tot > ram * 0.6:
|
||||
msg += " Va bene, ma con poca memoria libera per generare immagini."
|
||||
return msg
|
||||
|
||||
|
||||
def load_model(name: str):
|
||||
"""Carica (o riusa) il modello. Un solo modello in memoria per volta."""
|
||||
with _lock:
|
||||
@@ -141,7 +166,7 @@ class TagFilter:
|
||||
# "raw_args": il template riproduce gli argomenti come stringa grezza (prompt identico a ciò che il modello ha scritto).
|
||||
FORMATS = {
|
||||
"hermes": {"think": ("<think>", "</think>"), "call": ("<tool_call>", "</tool_call>"), "thinking": False, "raw_args": False},
|
||||
"gemma4": {"think": ("<|channel>", "<channel|>"), "call": ("<|tool_call>", "<tool_call|>"), "thinking": False, "raw_args": True},
|
||||
"gemma4": {"think": ("<|channel>", "<channel|>"), "call": ("<|tool_call>", "<tool_call|>"), "thinking": False, "raw_args": False},
|
||||
"argkey": {"think": ("<think>", "</think>"), "call": ("<tool_call>", "</tool_call>"), "thinking": True, "raw_args": False},
|
||||
}
|
||||
GEMMA_ESC = '<|"|>'
|
||||
@@ -214,6 +239,29 @@ def _parse_call(raw: str, fmt: str) -> tuple[str, dict, str] | None:
|
||||
return str(d["name"]), dict(args or {}), json.dumps(args or {}, ensure_ascii=False)
|
||||
|
||||
|
||||
def _dict_args(m: dict) -> dict:
|
||||
"""Argomenti delle chiamate a strumenti sempre come dizionario: la cronologia è condivisa tra modelli e alcuni
|
||||
template (Spark, Gemma heretic) rifiutano le stringhe salvate da altri modelli."""
|
||||
calls = m.get("tool_calls")
|
||||
if not calls or all(isinstance((c.get("function") or {}).get("arguments"), dict) for c in calls):
|
||||
return m
|
||||
fixed = []
|
||||
for c in calls:
|
||||
fn = dict(c.get("function") or {})
|
||||
a = fn.get("arguments")
|
||||
if not isinstance(a, dict):
|
||||
try:
|
||||
a = json.loads(a)
|
||||
except Exception:
|
||||
try:
|
||||
a = json.loads(_gemma_to_json("{" + str(a) + "}"))
|
||||
except Exception:
|
||||
a = {}
|
||||
fn["arguments"] = a if isinstance(a, dict) else {}
|
||||
fixed.append({**c, "function": fn})
|
||||
return {**m, "tool_calls": fixed}
|
||||
|
||||
|
||||
class MLXEngine:
|
||||
name = "mlx"
|
||||
|
||||
@@ -321,7 +369,7 @@ class MLXEngine:
|
||||
"- Per azioni sul Mac (per esempio il calendario) usa gli strumenti dedicati. Se uno risponde con [CONFIRMATION_PENDING], chiedi all'utente di confermare sul pannello e non dire che è fatto.\n"
|
||||
"- Non inventare mai dati reali: meteo, calendario, mail, messaggi, file, ora, server e memoria li ottieni SOLO chiamando lo strumento corrispondente, anche a metà conversazione. Rispondere senza averlo chiamato è un errore grave.\n"
|
||||
"- Chiama gli strumenti solo nel formato previsto dal tuo template e mai descrivendoli a parole.")
|
||||
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if self.search_api_key
|
||||
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if True
|
||||
else "\n- Non hai accesso al web: se ti chiedono informazioni aggiornate, dillo chiaramente.")
|
||||
return f"{persona_text(self.assistant_name)}\n\n{user_block(self.user_name, memory_prompt())}{summary_block(self.history)}{note}"
|
||||
|
||||
@@ -335,11 +383,11 @@ class MLXEngine:
|
||||
return msgs[start:]
|
||||
|
||||
def _tools(self) -> list:
|
||||
return openai_tools() + registry.openai_tools() + ([SEARCH_TOOL] if self.search_api_key else [])
|
||||
return openai_tools() + registry.openai_tools(local=True) + [SEARCH_TOOL]
|
||||
|
||||
def _prompt(self, generation: bool = True) -> str:
|
||||
"""Prompt completo; con generation=False si ferma alla fine dell'ultimo messaggio (parte stabile)."""
|
||||
messages = [{"role": "system", "content": self._system()}, *self._recent()]
|
||||
messages = [{"role": "system", "content": self._system()}, *[_dict_args(m) for m in self._recent()]]
|
||||
try:
|
||||
return self.tokenizer.apply_chat_template(messages, tools=self._tools(), add_generation_prompt=generation, tokenize=False, enable_thinking=self._thinking())
|
||||
except TypeError:
|
||||
@@ -362,7 +410,7 @@ class MLXEngine:
|
||||
if name == "cerca_web":
|
||||
emit({"type": "status", "status": "searching"})
|
||||
try:
|
||||
hits = brave_search(str(args.get("query", "")), self.search_api_key)
|
||||
hits = web_search(str(args.get("query", "")), self.search_api_key)
|
||||
except Exception as err:
|
||||
return f"Errore nella ricerca: {err}"
|
||||
for h in hits[:5]:
|
||||
@@ -415,6 +463,7 @@ class MLXEngine:
|
||||
else:
|
||||
# generate_step inserisce nella cache ogni token emesso (anche l'ultimo, anche quando si ferma per limite).
|
||||
_loaded.update(cache=cache, tokens=tokens + generated)
|
||||
emit({"type": "usage", "motore": f"locale ({self.model_name.split('/')[-1]})", "input": len(tokens), "output": len(generated)})
|
||||
tail = tools.flush()
|
||||
if tail:
|
||||
text += tail
|
||||
|
||||
@@ -8,7 +8,7 @@ import openai
|
||||
|
||||
from avatar.memory_tools import memory_prompt, openai_tools, parse_args, run_tool
|
||||
from avatar.plugins import registry
|
||||
from avatar.websearch import brave_search
|
||||
from avatar.websearch import web_search
|
||||
from .base import Emit, History, Meter, compact_history, persona_text, summary_block, today_label, user_block
|
||||
|
||||
MAX_ROUNDS = 8
|
||||
@@ -94,7 +94,7 @@ class OpenAICompatEngine:
|
||||
"- Quando l'utente ti chiede di ricordare qualcosa, o ti dice un fatto importante su di sé, DEVI chiamare salva_memoria prima di rispondere. Non dire mai di aver salvato senza averlo chiamato davvero.\n"
|
||||
"- Per cancellare una memoria chiama dimentica_memoria; per cercarne una non presente nel prompt chiama cerca_memoria.\n"
|
||||
"- Per azioni sul Mac (per esempio il calendario) usa gli strumenti dedicati. Se uno risponde con [CONFIRMATION_PENDING], chiedi all'utente di confermare sul pannello e non dire che è fatto.")
|
||||
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if self.search_api_key
|
||||
note += ("\n- Per informazioni aggiornate chiama cerca_web e rispondi in base ai risultati." if True
|
||||
else "\n- Non hai accesso al web: se ti chiedono informazioni aggiornate, dillo chiaramente.")
|
||||
else:
|
||||
note = "\n\nNota: in questa modalità non hai strumenti (niente memoria automatica né ricerca web)."
|
||||
@@ -121,14 +121,14 @@ class OpenAICompatEngine:
|
||||
def _tools(self):
|
||||
if not self._tools_supported:
|
||||
return None
|
||||
return openai_tools() + registry.openai_tools() + ([SEARCH_TOOL] if self.search_api_key else [])
|
||||
return openai_tools() + registry.openai_tools(local=True) + [SEARCH_TOOL]
|
||||
|
||||
def _run_tool(self, name: str, raw_args: str, emit: Emit, sources: list) -> str:
|
||||
args = parse_args(raw_args)
|
||||
if name == "cerca_web":
|
||||
emit({"type": "status", "status": "searching"})
|
||||
try:
|
||||
hits = brave_search(str(args.get("query", "")), self.search_api_key)
|
||||
hits = web_search(str(args.get("query", "")), self.search_api_key)
|
||||
except Exception as err:
|
||||
return f"Errore nella ricerca: {err}"
|
||||
for h in hits[:5]:
|
||||
|
||||
@@ -90,3 +90,8 @@ Sei **Ava**, l'assistente personale di chi ti parla. Vivi in un'app sul suo Mac
|
||||
## Musica
|
||||
|
||||
- Per far suonare musica usa Spotify (spotify_riproduci, spotify_coda, spotify_controllo). Quando l'utente chiede consigli, "mettimi qualcosa", "qualcosa di nuovo che mi piaccia" o musica per un momento (cena, lavoro, festa), chiama spotify_gusti, scegli tu 6-10 brani reali coerenti con i suoi gusti (DJ, producer: conta anche il genere e il mood del momento), dillo in una frase e falli partire con spotify_coda.
|
||||
|
||||
## Documenti dell'utente
|
||||
|
||||
- Per domande sul contenuto dei documenti dell'utente (contratti, bollette, ricevute, manuali, appunti) usa documenti_cerca prima di dire che non lo sai; rispondi citando il file e, se i passaggi non contengono la risposta, dillo. Per trovare un file per nome resta file_cerca.
|
||||
- Per "quanto ho speso / quanti token" usa consumi_riepilogo.
|
||||
+12
-2
@@ -127,8 +127,18 @@ class PluginRegistry:
|
||||
def anthropic_tools(self) -> list[dict]:
|
||||
return [{"name": t.name, "description": t.description, "eager_input_streaming": True, "input_schema": t.parameters} for t in self.active()]
|
||||
|
||||
def openai_tools(self) -> list[dict]:
|
||||
return [{"type": "function", "function": {"name": t.name, "description": t.description, "parameters": t.parameters}} for t in self.active()]
|
||||
def openai_tools(self, local: bool = False) -> list[dict]:
|
||||
"""Per i modelli locali (local=True) esclude i server MCP esterni: le loro descrizioni (Cua Driver ~14k token)
|
||||
rallentano troppo il primo turno. Si riattivano con l'impostazione mcp_locali."""
|
||||
tools = self.active()
|
||||
if local:
|
||||
try:
|
||||
from avatar.settings import Settings
|
||||
if not Settings().get("mcp_locali"):
|
||||
tools = [t for t in tools if not t.module.startswith("mcp:")]
|
||||
except Exception:
|
||||
pass
|
||||
return [{"type": "function", "function": {"name": t.name, "description": t.description, "parameters": t.parameters}} for t in tools]
|
||||
|
||||
def mcp_tools(self) -> list[dict]:
|
||||
return [{"name": t.name, "description": t.description, "inputSchema": t.parameters} for t in self.active()]
|
||||
|
||||
+24
-2
@@ -40,6 +40,21 @@ def lan_ip() -> str:
|
||||
return "127.0.0.1"
|
||||
|
||||
|
||||
TAILSCALE = "/Applications/Tailscale.app/Contents/MacOS/Tailscale"
|
||||
|
||||
|
||||
def tailscale() -> tuple[str, str]:
|
||||
"""(indirizzo IPv4, nome MagicDNS) del Mac su Tailscale, oppure ("", "") se spento o non installato."""
|
||||
import json as _j
|
||||
try:
|
||||
st = _j.loads(subprocess.run([TAILSCALE, "status", "--json"], capture_output=True, text=True, timeout=5).stdout or "{}")
|
||||
me = st.get("Self") or {}
|
||||
ip = next((a for a in me.get("TailscaleIPs") or [] if "." in a), "")
|
||||
return (ip, (me.get("DNSName") or "").rstrip(".")) if st.get("BackendState") == "Running" else ("", "")
|
||||
except Exception:
|
||||
return "", ""
|
||||
|
||||
|
||||
def ensure_certs(ip: str) -> tuple[Path, Path, Path]:
|
||||
"""CA locale + certificato del server con SAN per IP e nome .local; rigenerato se l'IP cambia."""
|
||||
CERT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
@@ -56,6 +71,9 @@ def ensure_certs(ip: str) -> tuple[Path, Path, Path]:
|
||||
(CERT_DIR / "ca.cer").write_bytes(subprocess.run(["openssl", "x509", "-in", str(ca_crt), "-outform", "DER"], check=True, capture_output=True).stdout)
|
||||
host = socket.gethostname()
|
||||
san = f"IP:{ip},IP:127.0.0.1,DNS:{host},DNS:localhost"
|
||||
ts_ip, ts_name = tailscale() # accesso da fuori casa: indirizzo e nome Tailscale nel certificato
|
||||
if ts_ip:
|
||||
san += f",IP:{ts_ip}" + (f",DNS:{ts_name}" if ts_name else "")
|
||||
if not srv_crt.exists() or not srv_san.exists() or srv_san.read_text() != san:
|
||||
csr = CERT_DIR / "server.csr"
|
||||
ext = CERT_DIR / "server.ext"
|
||||
@@ -108,7 +126,11 @@ class RemoteServer:
|
||||
|
||||
def urls(self) -> tuple[str, str, str, str]:
|
||||
base = f"https://{self.ip}:{HTTPS_PORT}"
|
||||
return base, self.new_pin(), f"{base}/?k={self.token}", f"{base} (CA: http://{self.ip}:{HTTP_PORT}/ca.cer)"
|
||||
ts_ip, ts_name = tailscale()
|
||||
if ts_ip:
|
||||
ensure_certs(self.ip) # certificato aggiornato se Tailscale è stato acceso dopo l'avvio (vale al prossimo riavvio)
|
||||
fuori = f" · fuori casa: https://{ts_name or ts_ip}:{HTTPS_PORT}" if ts_ip else " · fuori casa: accendi Tailscale"
|
||||
return base, self.new_pin(), f"{base}/?k={self.token}", f"{base}{fuori} (CA: http://{self.ip}:{HTTP_PORT}/ca.cer)"
|
||||
|
||||
def has_clients(self) -> bool:
|
||||
now = time.time()
|
||||
@@ -461,7 +483,7 @@ class RemoteServer:
|
||||
|
||||
# ── verso i telefoni (chiamabili da qualunque thread) ──────────────────
|
||||
def broadcast(self, payload: dict) -> None:
|
||||
if not (self._clients or self._sse) or not self._loop:
|
||||
if not self.has_clients() or not self._loop:
|
||||
return
|
||||
data = json.dumps(payload, ensure_ascii=False)
|
||||
|
||||
|
||||
@@ -57,6 +57,12 @@ DEFAULTS: dict[str, Any] = {
|
||||
"abitudini_enabled": True, # distillazione settimanale delle abitudini nella memoria
|
||||
"umore_enabled": False, # stima oraria dell'umore dai messaggi scritti dall'utente
|
||||
"io_nomi": "Luciano", # come compare l'utente come mittente nelle chat esportate
|
||||
"frasi_attesa": True, # "Un attimo che controllo…" se la risposta tarda più di ~1,3 s
|
||||
"riserva_cloud": True,
|
||||
"mcp_locali": False, # strumenti dei server MCP esterni anche per i modelli locali (prompt molto più lungo) # se il motore locale fallisce, la risposta la dà Claude
|
||||
"documenti_enabled": True, # indicizzazione oraria dei documenti per documenti_cerca
|
||||
"documenti_cartelle": "~/Documents, ~/Desktop",
|
||||
"documenti_escludi": "Progetti2026", # nomi di cartelle da saltare (codice, dati sensibili…)
|
||||
"qwen_voce": "luza_voce",
|
||||
"qwen_modello": "mlx-community/Qwen3-TTS-12Hz-1.7B-Base-8bit", # oppure ...-Base-bf16 (più lento, qualità piena) # voce clonata (file in data/voices) per il motore Qwen3-TTS
|
||||
"ipixel_enabled": False, # pannello LED iPIXEL via Bluetooth
|
||||
|
||||
@@ -127,8 +127,13 @@ class SettingsDialog(QDialog):
|
||||
self.mlx_model.addItems(cached_models() or [DEFAULT_MODEL])
|
||||
self.mlx_model.setEditText(s.get("mlx_model") or DEFAULT_MODEL)
|
||||
f.addRow("Modello (repo Hugging Face mlx-community)", self.mlx_model)
|
||||
from .engines.mlx_engine import memory_check
|
||||
self.mlx_mem = _note(memory_check(self.mlx_model.currentText())); f.addRow("", self.mlx_mem)
|
||||
self.mlx_model.currentTextChanged.connect(lambda t: self.mlx_mem.setText(memory_check(t.strip())))
|
||||
self.mlx_thinking = _combo([("auto", "Automatico (acceso solo dove serve, es. Spark)"), ("off", "Spento: risposte rapide"), ("on", "Acceso: più lento, meglio su domande complesse")], s.get("mlx_thinking") or "auto")
|
||||
f.addRow("Ragionamento", self.mlx_thinking)
|
||||
self.riserva = QCheckBox("Se il modello locale fallisce, rispondi con Claude"); self.riserva.setChecked(bool(s.get("riserva_cloud", True)))
|
||||
f.addRow("Riserva cloud", self.riserva)
|
||||
hint = QLabel("Elenco: modelli già scaricati. Un nome nuovo viene scaricato al primo uso. "
|
||||
"Con 32 GB di RAM: modelli fino a ~20 GB in 4 bit (Qwen3-30B-A3B è veloce e supporta gli strumenti). "
|
||||
"Non tenere aperto anche VLLMac con lo stesso modello.")
|
||||
@@ -355,6 +360,14 @@ class SettingsDialog(QDialog):
|
||||
form_sp.addRow("Client secret", self.sp_sec)
|
||||
add(pg_tool, form_sp)
|
||||
|
||||
# ── Documenti interrogabili ───────────────────────────────────────
|
||||
form_doc = QFormLayout()
|
||||
form_doc.addRow(_note("Documenti interrogabili: LuZa legge il contenuto di PDF, Word e testi in queste cartelle e risponde a domande su di essi. Indice locale, aggiornato ogni ora."))
|
||||
self.doc_on = QCheckBox("Attivo"); self.doc_on.setChecked(bool(s.get("documenti_enabled", True))); form_doc.addRow("Indicizzazione", self.doc_on)
|
||||
self.doc_dirs = QLineEdit(str(s.get("documenti_cartelle") or "")); form_doc.addRow("Cartelle (separate da virgola)", self.doc_dirs)
|
||||
self.doc_skip = QLineEdit(str(s.get("documenti_escludi") or "")); form_doc.addRow("Sottocartelle da escludere", self.doc_skip)
|
||||
add(pg_tool, form_doc)
|
||||
|
||||
btns = QHBoxLayout(); btns.addStretch()
|
||||
cancel = QPushButton("Annulla"); cancel.clicked.connect(self.reject); btns.addWidget(cancel)
|
||||
save = QPushButton("Salva"); save.setObjectName("primary"); save.clicked.connect(self._save); btns.addWidget(save)
|
||||
@@ -642,7 +655,7 @@ class SettingsDialog(QDialog):
|
||||
values = {
|
||||
"provider": self.provider.currentData(), "effort": self.effort.currentData(),
|
||||
"local_base_url": self.local_url.text().strip().rstrip("/"), "local_model": self.local_model.currentText().strip(),
|
||||
"mlx_model": self.mlx_model.currentText().strip(), "mlx_thinking": self.mlx_thinking.currentData() or "auto",
|
||||
"mlx_model": self.mlx_model.currentText().strip(), "mlx_thinking": self.mlx_thinking.currentData() or "auto", "riserva_cloud": self.riserva.isChecked(),
|
||||
"search_api_key": self.search_key.text().strip(),
|
||||
"claudecode_model": self.cc_model.currentData(), "claudecode_access": self.cc_access.currentData(),
|
||||
"claudecode_config_dir": self.cc_config.currentText().strip(), "claudecode_path": self.cc_path.text().strip(),
|
||||
@@ -671,6 +684,7 @@ class SettingsDialog(QDialog):
|
||||
"ipixel_enabled": self.px_on.isChecked(), "ipixel_stati": self.px_stati.isChecked(), "ipixel_notifiche": self.px_notif.isChecked(),
|
||||
"ipixel_luminosita": int(self.px_lum.value()), "ipixel_intermezzi_min": int(self.px_fun.currentData() or 0), "ipixel_musica": self.px_mus.isChecked(),
|
||||
"spotify_client_id": self.sp_id.text().strip(),
|
||||
"documenti_enabled": self.doc_on.isChecked(), "documenti_cartelle": self.doc_dirs.text().strip(), "documenti_escludi": self.doc_skip.text().strip(),
|
||||
"immagini_famiglia": self.img_fam.currentData() or "z-image-turbo", "immagini_modello": self.img_model.currentText().strip(),
|
||||
}
|
||||
if self.ha_token.text().strip():
|
||||
|
||||
+66
-6
@@ -1,11 +1,13 @@
|
||||
"""Ricerca web con Brave Search (per il motore locale)."""
|
||||
"""Ricerca web per i motori locali: Brave (se c'è la chiave), altrimenti Bing e DuckDuckGo via HTML, senza chiavi né server."""
|
||||
from __future__ import annotations
|
||||
|
||||
import html
|
||||
import re
|
||||
|
||||
import requests
|
||||
|
||||
|
||||
|
||||
def brave_search(query: str, api_key: str, count: int = 6) -> list[dict]:
|
||||
res = requests.get(
|
||||
"https://api.search.brave.com/res/v1/web/search",
|
||||
@@ -17,9 +19,67 @@ def brave_search(query: str, api_key: str, count: int = 6) -> list[dict]:
|
||||
hits = []
|
||||
for r in (res.json().get("web") or {}).get("results") or []:
|
||||
if r.get("url"):
|
||||
hits.append({
|
||||
"title": r.get("title") or r["url"],
|
||||
"url": r["url"],
|
||||
"description": re.sub(r"<[^>]+>", "", r.get("description") or ""),
|
||||
})
|
||||
hits.append({"title": r.get("title") or r["url"], "url": r["url"],
|
||||
"description": re.sub(r"<[^>]+>", "", r.get("description") or "")})
|
||||
return hits
|
||||
|
||||
|
||||
UA = {"User-Agent": "Mozilla/5.0 (Macintosh; Intel Mac OS X 14_0) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.0 Safari/605.1.15",
|
||||
"Accept-Language": "it-IT,it;q=0.9"}
|
||||
|
||||
|
||||
def _clean(t: str) -> str:
|
||||
return " ".join(html.unescape(re.sub(r"<[^>]+>", "", t)).split())
|
||||
|
||||
|
||||
def bing_search(query: str, count: int = 6) -> list[dict]:
|
||||
r = requests.get("https://www.bing.com/search", params={"q": query, "setlang": "it", "cc": "IT"}, headers=UA, timeout=15)
|
||||
r.raise_for_status()
|
||||
hits = []
|
||||
for block in re.findall(r'<li class="b_algo".*?</li>', r.text, re.S):
|
||||
a = re.search(r'<h2[^>]*>\s*<a[^>]+href="([^"]+)"[^>]*>(.*?)</a>', block, re.S)
|
||||
if not a or not a.group(1).startswith("http"):
|
||||
continue
|
||||
snip = re.search(r'<p[^>]*>(.*?)</p>', block, re.S)
|
||||
hits.append({"title": _clean(a.group(2)), "url": html.unescape(a.group(1)), "description": _clean(snip.group(1)) if snip else ""})
|
||||
if len(hits) >= count:
|
||||
break
|
||||
return hits
|
||||
|
||||
|
||||
def ddg_search(query: str, count: int = 6) -> list[dict]:
|
||||
r = requests.post("https://html.duckduckgo.com/html/", data={"q": query, "kl": "it-it"}, timeout=15, headers=UA)
|
||||
r.raise_for_status()
|
||||
hits = []
|
||||
for m in re.finditer(r'class="result__a" href="([^"]+)"[^>]*>(.*?)</a>.*?class="result__snippet"[^>]*>(.*?)</a>', r.text, re.S):
|
||||
url = html.unescape(m.group(1))
|
||||
if "uddg=" in url:
|
||||
url = requests.utils.unquote(url.split("uddg=")[1].split("&")[0])
|
||||
hits.append({"title": html.unescape(re.sub(r"<[^>]+>", "", m.group(2))), "url": url,
|
||||
"description": html.unescape(re.sub(r"<[^>]+>", "", m.group(3)))})
|
||||
if len(hits) >= count:
|
||||
break
|
||||
return hits
|
||||
|
||||
|
||||
def web_search(query: str, api_key: str = "", count: int = 6) -> list[dict]:
|
||||
"""Brave (con chiave) → Bing → DuckDuckGo, senza chiavi né server. Errore solo se falliscono tutti."""
|
||||
errors = []
|
||||
for name, fn in (("Brave", (lambda: brave_search(query, api_key, count)) if api_key else None),
|
||||
("Bing", lambda: bing_search(query, count)), ("DuckDuckGo", lambda: ddg_search(query, count))):
|
||||
if fn is None:
|
||||
continue
|
||||
try:
|
||||
hits = fn()
|
||||
if hits:
|
||||
return hits
|
||||
errors.append(f"{name}: nessun risultato")
|
||||
except Exception as e:
|
||||
errors.append(f"{name}: {e}")
|
||||
raise RuntimeError("; ".join(errors))
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
hits = web_search("meteo Ortona domani")
|
||||
assert hits and hits[0]["url"].startswith("http") and hits[0]["title"], hits
|
||||
print(len(hits), "risultati:", hits[0]["title"][:60], "|", hits[0]["description"][:80])
|
||||
@@ -0,0 +1,58 @@
|
||||
"""Consumi: token e costi per giorno e per motore, dal registro data/consumi.jsonl scritto a ogni risposta."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
from collections import defaultdict
|
||||
from datetime import datetime, timedelta
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
from avatar.settings import DATA_DIR # noqa: E402
|
||||
|
||||
LOG = DATA_DIR / "consumi.jsonl"
|
||||
|
||||
|
||||
def _n(x: int) -> str:
|
||||
return f"{x:,}".replace(",", ".")
|
||||
|
||||
|
||||
def riepilogo(giorni: int = 7) -> str:
|
||||
since = (datetime.now() - timedelta(days=giorni - 1)).strftime("%Y-%m-%d")
|
||||
try:
|
||||
rows = [json.loads(l) for l in LOG.read_text(encoding="utf-8").splitlines() if l.strip()]
|
||||
except FileNotFoundError:
|
||||
rows = []
|
||||
rows = [r for r in rows if r["ts"][:10] >= since]
|
||||
if not rows:
|
||||
return f"Nessun consumo registrato negli ultimi {giorni} giorni (il conteggio parte da oggi)."
|
||||
per_m = defaultdict(lambda: [0, 0, 0, 0.0, False]); per_g = defaultdict(lambda: [0, 0.0])
|
||||
for r in rows:
|
||||
m = per_m[r["motore"]]; m[0] += 1; m[1] += r["input"]; m[2] += r["output"]; m[3] += r["costo"]; m[4] = m[4] or r.get("abbonamento", False)
|
||||
g = per_g[r["ts"][:10]]; g[0] += r["input"] + r["output"]; g[1] += 0 if r.get("abbonamento") else r["costo"]
|
||||
out = [f"Consumi ultimi {giorni} giorni:"]
|
||||
for mot, (n, i, o, c, abb) in sorted(per_m.items(), key=lambda kv: -(kv[1][1] + kv[1][2])):
|
||||
costo = "gratis (locale)" if mot.startswith("locale") else (f"≈ ${c:.2f} equivalente API, incluso nell'abbonamento" if abb else f"≈ ${c:.2f}")
|
||||
out.append(f"- {mot}: {n} risposte, {_n(i)} token in ingresso, {_n(o)} in uscita, {costo}")
|
||||
out.append("Per giorno: " + "; ".join(f"{g[8:10]}/{g[5:7]} {_n(t)} token" + (f" (${c:.2f})" if c else "") for g, (t, c) in sorted(per_g.items())))
|
||||
return "\n".join(out)
|
||||
|
||||
|
||||
def t_riepilogo(params: dict, ctx: dict) -> str:
|
||||
return riepilogo(max(1, min(int(params.get("giorni") or 7), 90)))
|
||||
|
||||
|
||||
TOOLS = [
|
||||
{"name": "consumi_riepilogo", "description": "Token e costi delle risposte di LuZa per motore (locale, Claude Code, Claude API) e per giorno, negli ultimi N giorni (default 7). Per 'quanto ho speso?', 'quanti token ho usato?'.",
|
||||
"parameters": {"type": "object", "properties": {"giorni": {"type": "integer"}}}, "run": t_riepilogo},
|
||||
]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import tempfile
|
||||
LOG = Path(tempfile.mkdtemp()) / "c.jsonl"
|
||||
now = datetime.now().strftime("%Y-%m-%d %H:%M")
|
||||
LOG.write_text("\n".join(json.dumps(r) for r in [
|
||||
{"ts": now, "motore": "locale (gemma)", "input": 100, "output": 50, "costo": 0, "abbonamento": False},
|
||||
{"ts": now, "motore": "claude api (x)", "input": 1000, "output": 200, "costo": 0.01, "abbonamento": False}]) + "\n")
|
||||
out = riepilogo(7); assert "gratis (locale)" in out and "$0.01" in out, out; print(out)
|
||||
@@ -0,0 +1,259 @@
|
||||
"""Documenti interrogabili: indicizza le cartelle scelte (PDF, Word, testo, pagine…) in pezzi da ~800 caratteri,
|
||||
con ricerca ibrida per parole (SQLite FTS5) e per significato (embeddings multilingual-e5-small). Tutto locale.
|
||||
L'indice si aggiorna da solo ogni ora, solo per i file nuovi o modificati."""
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import re
|
||||
import sqlite3
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from pathlib import Path
|
||||
|
||||
import numpy as np
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
|
||||
from avatar.settings import DATA_DIR, Settings # noqa: E402
|
||||
|
||||
DB = DATA_DIR / "documenti.sqlite"
|
||||
EMB_MODEL = "intfloat/multilingual-e5-small"
|
||||
EXT = {".pdf", ".docx", ".doc", ".txt", ".md", ".rtf", ".odt", ".pages", ".html", ".csv"}
|
||||
SKIP_DIRS = {"node_modules", ".git", "Library", ".venv", "venv", "__pycache__", "build", "dist", ".Trash"}
|
||||
MAX_MB = 30
|
||||
_emb: dict = {}
|
||||
_stato = {"in_corso": False, "ultimo": "", "errore": ""}
|
||||
|
||||
|
||||
def _cartelle() -> list[Path]:
|
||||
raw = str(Settings().get("documenti_cartelle") or "~/Documents, ~/Desktop")
|
||||
return [Path(c.strip()).expanduser() for c in raw.split(",") if c.strip()]
|
||||
|
||||
|
||||
def _con() -> sqlite3.Connection:
|
||||
DB.parent.mkdir(parents=True, exist_ok=True)
|
||||
con = sqlite3.connect(DB, timeout=30)
|
||||
con.executescript("""
|
||||
CREATE TABLE IF NOT EXISTS files (path TEXT PRIMARY KEY, mtime REAL, chunks INTEGER);
|
||||
CREATE TABLE IF NOT EXISTS chunks (id INTEGER PRIMARY KEY, path TEXT, n INTEGER, text TEXT, emb BLOB);
|
||||
CREATE INDEX IF NOT EXISTS chunks_path ON chunks(path);
|
||||
CREATE VIRTUAL TABLE IF NOT EXISTS fts USING fts5(text, content='chunks', content_rowid='id', tokenize='unicode61 remove_diacritics 2');
|
||||
CREATE TRIGGER IF NOT EXISTS chunks_ai AFTER INSERT ON chunks BEGIN INSERT INTO fts(rowid, text) VALUES (new.id, new.text); END;
|
||||
CREATE TRIGGER IF NOT EXISTS chunks_ad AFTER DELETE ON chunks BEGIN INSERT INTO fts(fts, rowid, text) VALUES ('delete', old.id, old.text); END;
|
||||
""")
|
||||
return con
|
||||
|
||||
|
||||
def _embed(texts: list[str], query: bool = False) -> np.ndarray:
|
||||
"""Vettori normalizzati (384 dimensioni); prefissi 'query:'/'passage:' richiesti dal modello e5."""
|
||||
import torch
|
||||
from transformers import AutoModel, AutoTokenizer
|
||||
if not _emb:
|
||||
_emb["tok"] = AutoTokenizer.from_pretrained(EMB_MODEL)
|
||||
dev = "mps" if torch.backends.mps.is_available() else "cpu" # GPU del Mac: ~10× più veloce della CPU
|
||||
_emb["model"] = AutoModel.from_pretrained(EMB_MODEL).eval().to(dev)
|
||||
_emb["dev"] = dev
|
||||
tok, model = _emb["tok"], _emb["model"]
|
||||
out = []
|
||||
for i in range(0, len(texts), 32):
|
||||
batch = [("query: " if query else "passage: ") + t for t in texts[i:i + 32]]
|
||||
enc = tok(batch, padding=True, truncation=True, max_length=512, return_tensors="pt").to(_emb["dev"])
|
||||
with torch.no_grad():
|
||||
h = model(**enc).last_hidden_state
|
||||
m = enc["attention_mask"].unsqueeze(-1).float()
|
||||
v = (h * m).sum(1) / m.sum(1)
|
||||
out.append(torch.nn.functional.normalize(v, dim=1).float().cpu().numpy().astype(np.float32))
|
||||
return np.concatenate(out) if out else np.zeros((0, 384), np.float32)
|
||||
|
||||
|
||||
def _testo(path: Path) -> str:
|
||||
import file as fileplugin # stesso estrattore del plugin file (pdf, docx, testo, textutil per gli altri)
|
||||
return fileplugin._read(path)
|
||||
|
||||
|
||||
def _pezzi(testo: str, size: int = 800, overlap: int = 150) -> list[str]:
|
||||
testo = re.sub(r"[ \t]+", " ", re.sub(r"\n{3,}", "\n\n", testo)).strip()
|
||||
out, i = [], 0
|
||||
while i < len(testo):
|
||||
out.append(testo[i:i + size]); i += size - overlap
|
||||
return [p for p in out if len(p.strip()) > 40]
|
||||
|
||||
|
||||
def _escludi() -> set[str]:
|
||||
raw = str(Settings().get("documenti_escludi") or "Progetti2026")
|
||||
return {x.strip() for x in raw.split(",") if x.strip()}
|
||||
|
||||
|
||||
def _files():
|
||||
skip = SKIP_DIRS | _escludi()
|
||||
for root in _cartelle():
|
||||
if not root.exists():
|
||||
continue
|
||||
for dirpath, dirnames, filenames in os.walk(root):
|
||||
dirnames[:] = [d for d in dirnames if d not in skip and not d.startswith(".") and not d.endswith(".app")]
|
||||
for f in filenames:
|
||||
p = Path(dirpath) / f
|
||||
if p.suffix.lower() in EXT and not f.startswith("~$"):
|
||||
try:
|
||||
if p.stat().st_size <= MAX_MB * 1e6:
|
||||
yield p
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
def aggiorna(log=None) -> str:
|
||||
"""Indicizzazione incrementale: aggiunge i file nuovi/modificati, toglie quelli spariti."""
|
||||
import fcntl
|
||||
DB.parent.mkdir(parents=True, exist_ok=True)
|
||||
lockf = open(DATA_DIR / "documenti.lock", "w")
|
||||
try: # blocco tra processi: app e indicizzazioni manuali non si sovrappongono
|
||||
fcntl.flock(lockf, fcntl.LOCK_EX | fcntl.LOCK_NB)
|
||||
except OSError:
|
||||
lockf.close()
|
||||
return "Indicizzazione già in corso."
|
||||
_stato["in_corso"] = True
|
||||
try:
|
||||
con = _con()
|
||||
noti = dict(con.execute("SELECT path, mtime FROM files").fetchall())
|
||||
visti, nuovi, pezzi_tot = set(), 0, 0
|
||||
for p in _files():
|
||||
sp = str(p); visti.add(sp)
|
||||
mt = p.stat().st_mtime
|
||||
if noti.get(sp) == mt:
|
||||
continue
|
||||
try:
|
||||
pezzi = _pezzi(_testo(p))[:400]
|
||||
except Exception:
|
||||
pezzi = []
|
||||
vec = _embed(pezzi) if pezzi else np.zeros((0, 384), np.float32)
|
||||
con.execute("DELETE FROM chunks WHERE path=?", (sp,))
|
||||
con.executemany("INSERT INTO chunks(path, n, text, emb) VALUES (?,?,?,?)", [(sp, i, t, vec[i].tobytes()) for i, t in enumerate(pezzi)])
|
||||
con.execute("INSERT OR REPLACE INTO files(path, mtime, chunks) VALUES (?,?,?)", (sp, mt, len(pezzi)))
|
||||
con.commit()
|
||||
nuovi += 1; pezzi_tot += len(pezzi)
|
||||
if log and nuovi % 25 == 0:
|
||||
log(f"[documenti] indicizzati {nuovi} file…")
|
||||
spariti = [p for p in noti if p not in visti]
|
||||
for sp in spariti:
|
||||
con.execute("DELETE FROM chunks WHERE path=?", (sp,)); con.execute("DELETE FROM files WHERE path=?", (sp,))
|
||||
con.commit()
|
||||
n_files, n_chunks = con.execute("SELECT COUNT(*), COALESCE(SUM(chunks),0) FROM files").fetchone()
|
||||
con.close()
|
||||
_stato["ultimo"] = time.strftime("%d/%m %H:%M")
|
||||
return f"Indice aggiornato: {nuovi} file nuovi o modificati, {len(spariti)} rimossi. Totale {n_files} documenti, {n_chunks} passaggi."
|
||||
finally:
|
||||
_stato["in_corso"] = False
|
||||
fcntl.flock(lockf, fcntl.LOCK_UN); lockf.close()
|
||||
|
||||
|
||||
def cerca_testo(domanda: str, k: int = 6) -> list[dict]:
|
||||
con = _con()
|
||||
rows = con.execute("SELECT id, path, text, emb FROM chunks").fetchall()
|
||||
if not rows:
|
||||
con.close(); return []
|
||||
ids = np.array([r[0] for r in rows]); M = np.frombuffer(b"".join(r[3] for r in rows), dtype=np.float32).reshape(len(rows), -1)
|
||||
q = _embed([domanda], query=True)[0]
|
||||
sem = M @ q # coseno (vettori normalizzati)
|
||||
score = {int(i): float(s) for i, s in zip(ids, sem)}
|
||||
words = [w for w in re.findall(r"\w{3,}", domanda.lower())]
|
||||
if words:
|
||||
try:
|
||||
for (rid,) in con.execute("SELECT rowid FROM fts WHERE fts MATCH ? ORDER BY rank LIMIT 30", (" OR ".join(w + "*" for w in words),)):
|
||||
score[rid] = score.get(rid, 0) + 0.15 # bonus per le parole esatte
|
||||
except sqlite3.OperationalError:
|
||||
pass
|
||||
best = sorted(score.items(), key=lambda kv: -kv[1])[:k * 3]
|
||||
by_id = {r[0]: r for r in rows}
|
||||
out, per_file = [], {}
|
||||
for rid, sc in best:
|
||||
_i, path, text, _e = by_id[rid]
|
||||
if per_file.get(path, 0) >= 2:
|
||||
continue
|
||||
per_file[path] = per_file.get(path, 0) + 1
|
||||
out.append({"path": path, "testo": text, "score": round(sc, 3)})
|
||||
if len(out) >= k:
|
||||
break
|
||||
con.close()
|
||||
return out
|
||||
|
||||
|
||||
# ── strumenti ────────────────────────────────────────────────────────────────
|
||||
def t_cerca(params: dict, ctx: dict) -> str:
|
||||
domanda = str(params.get("domanda") or "").strip()
|
||||
if not domanda:
|
||||
return "Errore: serve la domanda."
|
||||
if not DB.exists():
|
||||
return "Non ho ancora indicizzato i documenti: usa documenti_aggiorna (ci vuole qualche minuto la prima volta)."
|
||||
res = cerca_testo(domanda, max(3, min(int(params.get("quanti") or 6), 12)))
|
||||
if not res:
|
||||
return "Nessun passaggio pertinente nei documenti indicizzati."
|
||||
home = str(Path.home())
|
||||
righe = [f"[{i + 1}] {r['path'].replace(home, '~')} (pertinenza {r['score']}):\n{r['testo']}" for i, r in enumerate(res)]
|
||||
return ("Passaggi più pertinenti dai documenti dell'utente. Rispondi basandoti su questi, cita il nome del file, "
|
||||
"e di' chiaramente se l'informazione non c'è:\n\n" + "\n\n".join(righe))
|
||||
|
||||
|
||||
def t_aggiorna(params: dict, ctx: dict) -> str:
|
||||
p = ctx.get("player")
|
||||
if p is not None and hasattr(p, "set_state"):
|
||||
p.set_state("PROCESSING · indicizzo i documenti")
|
||||
return aggiorna(ctx.get("log"))
|
||||
|
||||
|
||||
def t_stato(params: dict, ctx: dict) -> str:
|
||||
if not DB.exists():
|
||||
return "Indice non ancora creato. Cartelle: " + ", ".join(str(c) for c in _cartelle())
|
||||
con = _con(); n_files, n_chunks = con.execute("SELECT COUNT(*), COALESCE(SUM(chunks),0) FROM files").fetchone(); con.close()
|
||||
return (f"Documenti indicizzati: {n_files} file, {n_chunks} passaggi. Cartelle: {', '.join(str(c) for c in _cartelle())}. "
|
||||
+ ("Indicizzazione in corso." if _stato["in_corso"] else f"Ultimo aggiornamento: {_stato['ultimo'] or 'all\'avvio'}."))
|
||||
|
||||
|
||||
_proc: dict = {"p": None}
|
||||
|
||||
|
||||
def pausa(attiva: bool) -> None:
|
||||
"""Sospende/riprende l'indicizzazione mentre LuZa risponde: niente contesa di GPU e CPU con il modello."""
|
||||
import signal
|
||||
p = _proc["p"]
|
||||
if p is not None and p.poll() is None:
|
||||
try:
|
||||
p.send_signal(signal.SIGSTOP if attiva else signal.SIGCONT)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _servizio() -> None:
|
||||
"""Ogni ora indicizza in un processo separato a priorità bassa (niente blocchi del GIL nell'app)."""
|
||||
import subprocess
|
||||
time.sleep(120)
|
||||
while True:
|
||||
if Settings().get("documenti_enabled", True):
|
||||
code = ("import sys; sys.argv=['memory_mcp.py']; sys.path.insert(0, %r); import documenti_rag as d; print('[documenti]', d.aggiorna(), flush=True)"
|
||||
% str(Path(__file__).resolve().parent))
|
||||
_proc["p"] = subprocess.Popen(["nice", "-n", "15", sys.executable, "-c", code], cwd=str(Path(__file__).resolve().parent.parent))
|
||||
_proc["p"].wait()
|
||||
_proc["p"] = None
|
||||
time.sleep(3600)
|
||||
|
||||
|
||||
if Path(sys.argv[0]).name != "memory_mcp.py": # solo nell'app
|
||||
threading.Thread(target=_servizio, daemon=True, name="documenti").start()
|
||||
|
||||
|
||||
TOOLS = [
|
||||
{"name": "documenti_cerca", "description": "Cerca nel CONTENUTO dei documenti dell'utente (PDF, Word, testi nelle cartelle indicizzate: Documenti, Scrivania) per rispondere a domande come 'cosa dice il contratto sulle penali?', 'quando scade l'assicurazione?', 'trova la ricevuta del dentista'. Restituisce i passaggi più pertinenti con il file di provenienza.",
|
||||
"parameters": {"type": "object", "properties": {"domanda": {"type": "string"}, "quanti": {"type": "integer"}}, "required": ["domanda"]}, "run": t_cerca},
|
||||
{"name": "documenti_aggiorna", "description": "Aggiorna subito l'indice dei documenti (file nuovi o modificati). Normalmente avviene da solo ogni ora.",
|
||||
"parameters": {"type": "object", "properties": {}}, "run": t_aggiorna},
|
||||
{"name": "documenti_stato", "description": "Quanti documenti sono indicizzati, quali cartelle e quando è stato l'ultimo aggiornamento.",
|
||||
"parameters": {"type": "object", "properties": {}}, "run": t_stato},
|
||||
]
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
assert _pezzi("a" * 2000)[1].startswith("a") and len(_pezzi("a" * 2000)) == 4
|
||||
v = _embed(["Il contratto prevede una penale del 5% per ritardo."]); q = _embed(["quanto è la penale?"], query=True)[0]
|
||||
w = _embed(["La ricetta della carbonara usa guanciale."])
|
||||
assert float(v[0] @ q) > float(w[0] @ q), "il significato deve battere il testo non pertinente"
|
||||
print("ok: pertinente", round(float(v[0] @ q), 3), "> non pertinente", round(float(w[0] @ q), 3))
|
||||
Reference in new issue
Block a user