import json import re from datetime import datetime from threading import Lock from pathlib import Path import sys def get_base_dir() -> Path: if getattr(sys, "frozen", False): return Path(sys.executable).parent return Path(__file__).resolve().parent.parent BASE_DIR = get_base_dir() MEMORY_PATH = BASE_DIR / "memory" / "long_term.json" _lock = Lock() MAX_VALUE_LENGTH = 380 # ── Why there are two very different numbers here ──────────────────────────── # # There used to be one: MEMORY_MAX_CHARS = 2200, applied to the whole store. It # was a *storage* limit, and it existed only because the entire memory was # pasted into the system prompt on every connect — so growing the memory grew # every single request. When it filled, _trim_to_limit() deleted the oldest # entries and printed one line to a console nobody reads. A memory described as # "deeply remembers projects, preferences and personal context" was in practice # two pages long, and quietly forgot your sister's name after a few weeks. # # Storage and prompt budget are now separate concerns: # # MEMORY_MAX_CHARS — a runaway guard, not a feature limit. Nothing normal # reaches it; a bug writing in a loop does. # PROMPT_CORE_CHARS — what actually rides in the system prompt every session. # Smaller than the old whole-memory dump, so sessions # start *faster* than before, not slower. # # Everything above the core stays on disk and is fetched on demand by the # recall_memory tool — see search_memory() and format_memory_for_prompt(). MEMORY_MAX_CHARS = 200_000 PROMPT_CORE_CHARS = 900 PROMPT_INDEX_CHARS = 420 # Most entries any one category may contribute to the core block, so a person # with forty stored preferences still gets their sister into the prompt. PROMPT_MAX_PER_CATEGORY = 6 def _empty_memory() -> dict: return { "identity": {}, "preferences": {}, "projects": {}, "relationships": {}, "wishes": {}, "notes": {}, } def load_memory() -> dict: if not MEMORY_PATH.exists(): return _empty_memory() with _lock: try: data = json.loads(MEMORY_PATH.read_text(encoding="utf-8")) if isinstance(data, dict): base = _empty_memory() for key in base: if key not in data: data[key] = {} return data return _empty_memory() except Exception as e: print(f"[Memory] ⚠️ Load error: {e}") return _empty_memory() def _all_entries(memory: dict) -> list[tuple]: entries = [] for cat, items in memory.items(): if not isinstance(items, dict): continue for key, entry in items.items(): if isinstance(entry, dict) and "value" in entry: entries.append((cat, key, entry)) return entries # Set by main.py so a trim can reach the activity log. Deleting something a # person told you and mentioning it only on stdout is how a memory loses trust. _trim_notifier = None def set_trim_notifier(fn) -> None: """Register a callable(str) that surfaces trims to the user.""" global _trim_notifier _trim_notifier = fn def _trim_to_limit(memory: dict) -> dict: if len(json.dumps(memory, ensure_ascii=False)) <= MEMORY_MAX_CHARS: return memory entries = _all_entries(memory) entries.sort(key=lambda t: t[2].get("updated", "0000-00-00")) dropped = [] for cat, key, _ in entries: if len(json.dumps(memory, ensure_ascii=False)) <= MEMORY_MAX_CHARS: break del memory[cat][key] dropped.append(f"{cat}/{key}") print(f"[Memory] 🗑️ Trimmed {cat}/{key}") if dropped and _trim_notifier: try: _trim_notifier( f"SYS: Memory full — forgot {len(dropped)} oldest entries " f"({', '.join(dropped[:3])}{'…' if len(dropped) > 3 else ''})" ) except Exception: pass return memory def save_memory(memory: dict) -> None: if not isinstance(memory, dict): return memory = _trim_to_limit(memory) MEMORY_PATH.parent.mkdir(parents=True, exist_ok=True) with _lock: MEMORY_PATH.write_text( json.dumps(memory, indent=2, ensure_ascii=False), encoding="utf-8", ) try: export_markdown(memory) except Exception as err: print(f"[Memory] esportazione Markdown fallita: {err}") EXPORT_DIR = get_base_dir() / "data" / "memoria" _EXPORT_LABELS = {"identity": "Identità", "preferences": "Preferenze", "projects": "Progetti", "relationships": "Persone", "wishes": "Desideri e obiettivi", "notes": "Note"} def export_markdown(memory: dict, folder: Path | None = None) -> Path: """Specchia la memoria in una cartella di file Markdown (apribile come vault Obsidian): un file per categoria più un indice.""" folder = folder or EXPORT_DIR folder.mkdir(parents=True, exist_ok=True) index = ["# Memoria di LuZa", "", f"Aggiornata: {datetime.now():%Y-%m-%d %H:%M}", ""] for cat, label in _EXPORT_LABELS.items(): entries = memory.get(cat) or {} if not isinstance(entries, dict): continue lines = ["---", f"categoria: {cat}", f"voci: {len(entries)}", "---", "", f"# {label}", ""] for key, entry in sorted(entries.items(), key=lambda kv: (kv[1].get("updated", "") if isinstance(kv[1], dict) else ""), reverse=True): val = _entry_value(entry) upd = entry.get("updated", "") if isinstance(entry, dict) else "" lines.append(f"- **{_pretty(key)}**: {val}" + (f" _(aggiornato {upd})_" if upd else "")) (folder / f"{label}.md").write_text("\n".join(lines) + "\n", encoding="utf-8") index.append(f"- [[{label}]] — {len(entries)} voci") sessions = memory.get("sessions") or [] if isinstance(sessions, list) and sessions: lines = ["# Sessioni recenti", ""] + [f"- {e.get('date', '')}: {e.get('summary', '')}" for e in sessions] (folder / "Sessioni recenti.md").write_text("\n".join(lines) + "\n", encoding="utf-8") index.append("- [[Sessioni recenti]]") (folder / "Memoria.md").write_text("\n".join(index) + "\n", encoding="utf-8") return folder def _truncate_value(val: str) -> str: if isinstance(val, str) and len(val) > MAX_VALUE_LENGTH: return val[:MAX_VALUE_LENGTH].rstrip() + "…" return val def _recursive_update(target: dict, updates: dict) -> bool: changed = False for key, value in updates.items(): if value is None: continue if isinstance(value, str) and not value.strip(): continue if isinstance(value, dict) and "value" not in value: if key not in target or not isinstance(target[key], dict): target[key] = {} changed = True if _recursive_update(target[key], value): changed = True else: new_val = _truncate_value(str(value["value"] if isinstance(value, dict) else value)) entry = {"value": new_val, "updated": datetime.now().strftime("%Y-%m-%d")} existing = target.get(key, {}) if not isinstance(existing, dict) or existing.get("value") != new_val: target[key] = entry changed = True return changed def update_memory(memory_update: dict) -> dict: if not isinstance(memory_update, dict) or not memory_update: return load_memory() memory = load_memory() if _recursive_update(memory, memory_update): save_memory(memory) print(f"[Memory] 💾 Saved: {list(memory_update.keys())}") return memory def _entry_value(entry) -> str: """Accept both the {'value': ..., 'updated': ...} shape and a bare string, because early versions of the store wrote plain strings.""" if isinstance(entry, dict): return str(entry.get("value", "") or "").strip() return str(entry or "").strip() def _pretty(key: str) -> str: return key.replace("_", " ").strip() # Identity is always in the prompt; these categories compete for the remaining # budget by recency. _CATEGORY_LABELS = { "preferences": "Preferences", "projects": "Active projects / goals", "relationships": "People in their life", "wishes": "Wishes / plans", "notes": "Notes", } _IDENTITY_FIELDS = ["name", "age", "birthday", "city", "job", "language", "school", "nationality"] def format_memory_for_prompt(memory: dict | None) -> str: """Build the memory block that goes into the system prompt. This used to dump everything. It now sends three things: 1. IDENTITY - always, in full. It is small, and it is wrong for the assistant to have to look up your name. 2. RECENT - the most recently updated entries from every other category, up to PROMPT_CORE_CHARS. Recency is the cheapest useful relevance signal available without embeddings. 3. AN INDEX - the *keys* of everything else, values omitted. Point 3 is what makes recall work at all. A model cannot decide to look something up if it does not know the thing exists: with only points 1 and 2, "who is Ayse?" would get "I don't know" while ayse_sister sat on disk unread. The index costs a few hundred characters and turns recall from a gamble into a lookup. Net effect on latency: this block is SMALLER than the old full dump, so every session connects with fewer tokens. Occasionally the model spends one extra round trip on recall_memory - covered by the acknowledgment it already speaks before any slow step.""" if not memory: return "" core_lines: list[str] = [] # 1. Identity - always, in full identity = memory.get("identity", {}) or {} for field in _IDENTITY_FIELDS: val = _entry_value(identity.get(field)) if not val: continue if field == "language": # Labelled as an observation, not a setting. A bare "Language: # English" line written months ago reads like a standing order and # was one of the reasons a Turkish question came back in English. core_lines.append( f"Has spoken to you in: {val} (an observation about the past — " f"always answer in the language of their CURRENT message)") else: core_lines.append(f"{field.title()}: {val}") for key, entry in identity.items(): if key in _IDENTITY_FIELDS: continue val = _entry_value(entry) if val: core_lines.append(f"{_pretty(key).title()}: {val}") # 2. Everything else, most recently updated first rest: list[tuple[str, str, str, str]] = [] # (updated, cat, key, value) for cat in _CATEGORY_LABELS: for key, entry in (memory.get(cat, {}) or {}).items(): val = _entry_value(entry) if not val: continue updated = (entry.get("updated", "") if isinstance(entry, dict) else "") or "0000-00-00" rest.append((updated, cat, key, val)) rest.sort(key=lambda t: t[0], reverse=True) used = sum(len(l) + 1 for l in core_lines) shown: dict[str, list[str]] = {} overflow: dict[str, list[str]] = {} # Recency decides order, but no single category may take the whole budget. # Without the cap, someone with forty stored preferences gets a prompt that # is forty preferences and not one person's name — the categories that # matter most in conversation are also the ones that change least often, so # pure recency systematically buries them. per_cat_used: dict[str, int] = {} for _updated, cat, key, val in rest: line = f" - {_pretty(key).title()}: {val}" if (per_cat_used.get(cat, 0) < PROMPT_MAX_PER_CATEGORY and used + len(line) + 1 <= PROMPT_CORE_CHARS): shown.setdefault(cat, []).append(line) per_cat_used[cat] = per_cat_used.get(cat, 0) + 1 used += len(line) + 1 else: overflow.setdefault(cat, []).append(_pretty(key)) # The index is a table of contents, so it is interleaved across categories # rather than continuing in recency order. Sorted by recency it would list # twenty-four preferences before the first relationship, and the one entry # the index exists for — the old fact the model has no other way to know # about — would fall off the end. indexed: list[str] = [] if overflow: cats = [c for c in _CATEGORY_LABELS if overflow.get(c)] cursor = {c: 0 for c in cats} while cats: for cat in list(cats): i = cursor[cat] if i >= len(overflow[cat]): cats.remove(cat) continue indexed.append(overflow[cat][i]) cursor[cat] = i + 1 for cat, label in _CATEGORY_LABELS.items(): if shown.get(cat): core_lines.append("") core_lines.append(f"{label}:") core_lines.extend(shown[cat]) if not core_lines and not indexed: return "" out = [ "[WHAT YOU KNOW ABOUT THIS PERSON — use naturally, never recite like a list]", *core_lines, ] # 3. The index of what is on disk but not in this prompt if indexed: budget, names = PROMPT_INDEX_CHARS, [] for n in indexed: if budget - len(n) - 2 < 0: break names.append(n) budget -= len(n) + 2 if names: out.append("") out.append( "[ALSO REMEMBERED — values not shown here. Call recall_memory " "with a keyword to read any of these before saying you do not know]" ) out.append(", ".join(names) + (f" (+{len(indexed) - len(names)} more)" if len(indexed) > len(names) else "")) return "\n".join(out) + "\n" # ── Recall ──────────────────────────────────────────────────────────────────── def _score(query_words: list[str], cat: str, key: str, value: str) -> int: """Cheap lexical relevance. No embeddings, no network, no model call - this runs in well under a millisecond, which is the entire point: recall must cost one model round trip, never two.""" hay_key = _pretty(key).lower() hay_val = value.lower() score = 0 for w in query_words: if not w: continue if w == hay_key: score += 10 elif w in hay_key: score += 6 if w in hay_val: score += 3 if w in cat: score += 1 return score def search_memory(query: str, limit: int = 8) -> str: """Find stored facts matching `query`. Backs the recall_memory tool. An empty query is treated as "show me everything you know", capped - the model asks that when the user says "what do you remember about me?".""" memory = load_memory() words = [w for w in re.split(r"[^\w]+", (query or "").lower()) if len(w) > 1] rows: list[tuple[int, str, str, str]] = [] for cat, items in memory.items(): if not isinstance(items, dict): continue # skip 'sessions', which is a list for key, entry in items.items(): val = _entry_value(entry) if not val: continue s = _score(words, cat, key, val) if words else 1 if s > 0: rows.append((s, cat, key, val)) if not rows: return (f"Nothing stored about '{query}'." if query else "I have not stored anything about this person yet.") rows.sort(key=lambda r: (-r[0], r[2])) lines = [f"{cat}/{_pretty(key)}: {val}" for _s, cat, key, val in rows[:max(1, limit)]] head = (f"Stored facts matching '{query}':" if query else "Everything currently stored:") more = (f"\n(+{len(rows) - len(lines)} more — search with a narrower keyword)" if len(rows) > len(lines) else "") return head + "\n" + "\n".join(lines) + more def all_entries_for_ui() -> list[dict]: """Flat list for the memory panel: what JARVIS knows, and when it learned it. Sorted newest first so the panel opens on what changed most recently.""" memory = load_memory() rows = [] for cat, items in memory.items(): if not isinstance(items, dict): continue for key, entry in items.items(): val = _entry_value(entry) if not val: continue rows.append({ "category": cat, "key": key, "value": val, "updated": (entry.get("updated", "") if isinstance(entry, dict) else ""), }) rows.sort(key=lambda r: (r["updated"] or "0000-00-00"), reverse=True) return rows def remember(key: str, value: str, category: str = "notes") -> str: valid = {"identity", "preferences", "projects", "relationships", "wishes", "notes"} if category not in valid: category = "notes" update_memory({category: {key: {"value": value}}}) return f"Remembered: {category}/{key} = {value}" def forget(key: str, category: str = "notes") -> str: memory = load_memory() cat = memory.get(category, {}) if key in cat: del cat[key] memory[category] = cat save_memory(memory) return f"Forgotten: {category}/{key}" return f"Not found: {category}/{key}" forget_memory = forget # ── Session memory ───────────────────────────────────────────────────────────── _SESSION_MAX = 3 # safety cap — in practice 0-1 entries after pop def save_session_summary(summary: str, language: str = "") -> None: """Append a 1-2 sentence session summary to long_term.json['sessions'].""" summary = (summary or "").strip() if not summary: return memory = load_memory() sessions = memory.get("sessions", []) if not isinstance(sessions, list): sessions = [] entry: dict = { "date": datetime.now().strftime("%Y-%m-%d"), "summary": summary[:280], } if language: entry["language"] = language sessions.append(entry) memory["sessions"] = sessions[-_SESSION_MAX:] with _lock: MEMORY_PATH.parent.mkdir(parents=True, exist_ok=True) MEMORY_PATH.write_text( json.dumps(memory, indent=2, ensure_ascii=False), encoding="utf-8", ) print(f"[Memory] 📝 Session saved ({entry['date']}): {summary[:60]}…") def pop_last_session() -> dict | None: """ Return AND remove the most recent session entry. Calling this consumes the entry so it is never repeated in future briefings. """ with _lock: if not MEMORY_PATH.exists(): return None try: memory = json.loads(MEMORY_PATH.read_text(encoding="utf-8")) sessions = memory.get("sessions", []) if not isinstance(sessions, list) or not sessions: return None entry = sessions.pop() # remove the last entry memory["sessions"] = sessions MEMORY_PATH.write_text( json.dumps(memory, indent=2, ensure_ascii=False), encoding="utf-8", ) return entry except Exception as e: print(f"[Memory] ⚠️ pop_last_session error: {e}") return None