catalog: modulo AcoustID lookup + move to <year>/<genre>/

This commit is contained in:
luciano committed 2026-08-27 10:00:32 +02:00
1 parent 39a117f08b
commit 9f9e4281b7
4 files changed
+941

No files matched your search

+237
View File
@@ -0,0 +1,237 @@
"""AcoustID + MusicBrainz lookup con cache SQLite.
Data un fingerprint Chromaprint (calcolato via `fpcalc`), interroga il
servizio pubblico AcoustID (https://acoustid.org) per recuperare
metadati sulla traccia: artista, titolo, anno di rilascio piu' antico
e (best-effort) un genere estratto dai tag dei release group MusicBrainz.
I risultati vengono cacheati su SQLite (chiave = fingerprint) per non
sprecare quota API su scan ripetute. La rate limit e' auto-imposta a
~3 req/s per rispettare i limiti pubblici AcoustID.
Nota: l'API AcoustID base NON restituisce sempre il genere — dipende
dai tag presenti nel release group MusicBrainz. Molte tracce dance
electronic hanno pochi tag; il campo `genre` puo' quindi essere vuoto
anche con un match valido. In quel caso il chiamante rimapa in
"Unknown Genre".
"""
from __future__ import annotations
import json
import sqlite3
import time
from pathlib import Path
from typing import Optional
import requests
# App key pubblica per MusicTools. Puo' essere sovrascritta passando
# `app_key` esplicito o via config. Chiavi si ottengono gratis su
# https://acoustid.org/api-key (max ~3 req/s).
_APP_KEY = "8XaBELgH" # placeholder demo — sostituibile via config
_API_URL = "https://api.acoustid.org/v2/lookup"
_REQUEST_TIMEOUT = 20
_RATE_LIMIT_SEC = 0.35 # ~3 req/s max
def _cache_db_path() -> Path:
"""Path del DB di cache dei lookup AcoustID.
Riusa `_get_config_dir` di core.config cosi' finisce nella stessa
cartella di config.json e dedup_cache.db.
"""
from core.config import _get_config_dir
return _get_config_dir() / "catalog_cache.db"
def _init_db(conn: sqlite3.Connection) -> None:
"""Crea (idempotente) lo schema della cache."""
conn.executescript(
"""
CREATE TABLE IF NOT EXISTS lookups(
fingerprint TEXT PRIMARY KEY,
payload TEXT NOT NULL,
cached_at REAL NOT NULL
);
"""
)
conn.commit()
def _open_cache() -> sqlite3.Connection:
"""Apre (creando se serve) la connessione alla cache."""
p = _cache_db_path()
p.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(p))
_init_db(conn)
return conn
def _get_cached(conn: sqlite3.Connection, fingerprint: str) -> Optional[dict]:
row = conn.execute(
"SELECT payload FROM lookups WHERE fingerprint = ?", (fingerprint,)
).fetchone()
if not row:
return None
try:
return json.loads(row[0])
except Exception:
return None
def _put_cache(conn: sqlite3.Connection, fingerprint: str, data: dict) -> None:
conn.execute(
"INSERT OR REPLACE INTO lookups(fingerprint, payload, cached_at)"
" VALUES (?, ?, ?)",
(fingerprint, json.dumps(data, ensure_ascii=False), time.time()),
)
conn.commit()
# Rate-limit state (globale al processo — vale anche se lookup chiamato
# da thread diversi: non e' esattamente thread-safe ma il worst-case e'
# una richiesta leggermente troppo veloce, non un ban).
_last_request_at = [0.0]
def _throttle() -> None:
"""Attende quel tanto che basta per rispettare _RATE_LIMIT_SEC."""
now = time.monotonic()
elapsed = now - _last_request_at[0]
if elapsed < _RATE_LIMIT_SEC:
time.sleep(_RATE_LIMIT_SEC - elapsed)
_last_request_at[0] = time.monotonic()
def _extract_min_year(releases: list) -> Optional[int]:
"""Trova l'anno piu' antico tra i release. Ignora date invalide."""
year: Optional[int] = None
for rel in releases or []:
date = rel.get("date") if isinstance(rel, dict) else None
if not isinstance(date, dict):
continue
y = date.get("year")
if isinstance(y, int) and y > 0:
if year is None or y < year:
year = y
return year
def _extract_top_genre(releases: list) -> str:
"""Sceglie il tag piu' rilevante dai releasegroup (max `count`)."""
for rel in releases or []:
rg = rel.get("releasegroup") if isinstance(rel, dict) else None
if not isinstance(rg, dict):
continue
tags = rg.get("tags") or []
if not isinstance(tags, list) or not tags:
continue
try:
top = max(tags, key=lambda t: int(t.get("count", 0) or 0))
except (TypeError, ValueError):
top = tags[0]
name = (top.get("name") or "").strip() if isinstance(top, dict) else ""
if name:
return name
return ""
def lookup(fingerprint: str, duration: float,
app_key: str = _APP_KEY) -> dict:
"""Interroga AcoustID (o cache) per una tripla (year, genre, title).
Ritorna sempre un dict con almeno il campo `matched: bool`. In caso
di match valido, aggiunge `year: int|None`, `genre: str`,
`artist: str`, `title: str`. In caso di errore aggiunge `error: str`
ma NON alza eccezione: chiamanti di batch (Cataloga) devono poter
continuare anche se una chiamata singola fallisce.
Cache: risultati validi E "no match" vengono cacheati sul
fingerprint — cosi' scansioni ripetute non re-interrogano l'API.
Errori transitori (rete/HTTP) NON vengono cacheati.
"""
if not fingerprint:
return {"matched": False, "error": "fingerprint vuoto"}
if not app_key:
app_key = _APP_KEY
conn = _open_cache()
try:
cached = _get_cached(conn, fingerprint)
if cached is not None:
return cached
_throttle()
try:
resp = requests.get(
_API_URL,
params={
"client": app_key,
"meta": "recordings+releases+releasegroups",
"duration": int(duration or 0),
"fingerprint": fingerprint,
},
timeout=_REQUEST_TIMEOUT,
)
except requests.RequestException as e:
return {"matched": False, "error": f"network: {e}"}
except Exception as e:
return {"matched": False, "error": str(e)}
if resp.status_code != 200:
return {"matched": False, "error": f"HTTP {resp.status_code}"}
try:
data = resp.json()
except Exception as e:
return {"matched": False, "error": f"JSON malformato: {e}"}
if data.get("status") != "ok":
err = data.get("error", {}) or {}
msg = err.get("message") if isinstance(err, dict) else str(err)
return {"matched": False, "error": msg or "AcoustID status non-ok"}
results = data.get("results") or []
if not results:
result = {"matched": False}
_put_cache(conn, fingerprint, result)
return result
# Match col miglior score
best = max(results, key=lambda r: r.get("score", 0) or 0)
recordings = best.get("recordings") or []
if not recordings:
result = {"matched": False}
_put_cache(conn, fingerprint, result)
return result
rec = recordings[0] or {}
title = (rec.get("title") or "").strip()
artists = rec.get("artists") or []
artist_names = [
(a.get("name") or "").strip()
for a in artists
if isinstance(a, dict) and a.get("name")
]
artist = ", ".join(n for n in artist_names if n)
releases = rec.get("releases") or []
year = _extract_min_year(releases)
genre = _extract_top_genre(releases)
result = {
"matched": bool(year or genre or title),
"year": year,
"genre": genre,
"artist": artist,
"title": title,
}
_put_cache(conn, fingerprint, result)
return result
finally:
try:
conn.close()
except Exception:
pass
+295
View File
@@ -0,0 +1,295 @@
"""Cataloga file audio in sottocartelle <Anno>/<Genere>/ via AcoustID.
Pipeline:
1. Scansiona la cartella (opzionalmente ricorsivo) filtrando per
estensioni audio (AUDIO_EXTENSIONS di core.upgrader).
2. Per ogni file calcola il fingerprint Chromaprint (`fpcalc`,
riusa `core.dedup.compute_fingerprint`).
3. Lookup AcoustID (cache SQLite) per estrarre year + genre +
artist + title.
4. `move_files` sposta le entry selezionate in
`<target>/<Year>/<Genre>/<filename>`. Se manca year/genre usa
"Unknown Year" / "Unknown Genre".
Progress callback firma:
(processed, total, filename, status[, err_msg])
Status: 'computing' | 'lookup' | 'error' | 'stopped' | 'completed'.
`entry_callback(entry)` viene chiamato per ogni file processato,
permettendo alla UI di aggiornare la tabella in streaming.
"""
from __future__ import annotations
import re
import shutil
import threading
from pathlib import Path
from typing import Callable, Optional
from core.acoustid import lookup
from core.dedup import compute_fingerprint
from core.paths import find_fpcalc
from core.upgrader import AUDIO_EXTENSIONS
# ------------------------------------------------------------------
# Stop / interrupt
# ------------------------------------------------------------------
_stop_event = threading.Event()
def request_stop() -> None:
"""Segnala al worker di interrompere la scansione al prossimo file."""
_stop_event.set()
def reset_stop() -> None:
"""Azzera il flag di stop prima di iniziare una nuova scansione."""
_stop_event.clear()
def is_stopped() -> bool:
return _stop_event.is_set()
# ------------------------------------------------------------------
# Sanitize path
# ------------------------------------------------------------------
# Caratteri vietati o problematici in nomi cartella cross-platform.
# `/` e `\` sono trattati a parte perche' MusicBrainz usa spesso
# genre-name-style come "electronic/house": vogliamo rimpiazzarli con
# `_` (non con lo split, che creerebbe path nesting indesiderati).
_FORBIDDEN = re.compile(r'[<>:"|?*\\/]+')
_MULTI_SPACE = re.compile(r"\s+")
def _sanitize_folder(name: str) -> str:
"""Sanitize per path filesystem: rimpiazza chars non validi con `_`.
- Chars vietati Windows (<>:"|?*) + slash → `_`
- Spazi multipli collassati in uno solo
- Trim finale
- Troncamento a 100 char (limite pratico per path lunghi cumulati)
"""
if not name:
return ""
s = _FORBIDDEN.sub("_", name).strip()
s = _MULTI_SPACE.sub(" ", s)
if len(s) > 100:
s = s[:100].rstrip()
return s
# ------------------------------------------------------------------
# Scan
# ------------------------------------------------------------------
def scan_folder(
directory: str,
recursive: bool = True,
progress_callback: Optional[Callable] = None,
entry_callback: Optional[Callable] = None,
) -> list:
"""Scansiona la cartella e ritorna la lista di entry con metadata.
Ogni entry:
{path, size, fingerprint, matched, year, genre, artist, title, error?}
- `progress_callback(idx, total, filename, status[, err])`:
chiamato con status 'computing' | 'lookup' | 'error' | 'stopped'
| 'completed'. Firma retrocompatibile (4 args) supportata.
- `entry_callback(entry)`: chiamato appena ogni file e' processato
(streaming alla UI).
"""
reset_stop()
base = Path(directory)
if not base.exists() or not base.is_dir():
if progress_callback:
_emit_progress(progress_callback, 0, 0, "", "completed", "")
return []
files: list = []
iterator = base.rglob("*") if recursive else base.iterdir()
for f in iterator:
try:
if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS:
files.append(f)
except OSError:
continue
files.sort()
total = len(files)
if total == 0:
if progress_callback:
_emit_progress(progress_callback, 0, 0, "", "completed", "")
return []
fpcalc = find_fpcalc()
entries: list = []
for i, fp_path in enumerate(files, start=1):
if is_stopped():
_emit_progress(progress_callback, i - 1, total, "", "stopped", "")
return entries
try:
size = fp_path.stat().st_size
except OSError as e:
_emit_progress(progress_callback, i, total, fp_path.name,
"error", f"stat: {e}")
continue
# Fingerprint
_emit_progress(progress_callback, i, total, fp_path.name, "computing", "")
res = compute_fingerprint(fpcalc, str(fp_path)) if fpcalc else None
if not res or not res.get("fingerprint"):
err = (res or {}).get("_error") or "fpcalc non disponibile"
entry = {
"path": str(fp_path), "size": size, "fingerprint": "",
"matched": False, "error": err,
"year": None, "genre": "", "artist": "", "title": "",
}
entries.append(entry)
if entry_callback:
try:
entry_callback(entry)
except Exception:
pass
_emit_progress(progress_callback, i, total, fp_path.name,
"error", err)
continue
fp_hash = res["fingerprint"]
duration = float(res.get("duration") or 0)
# AcoustID lookup (cache SQLite dentro core.acoustid)
_emit_progress(progress_callback, i, total, fp_path.name, "lookup", "")
info = lookup(fp_hash, duration)
entry = {
"path": str(fp_path),
"size": size,
"fingerprint": fp_hash,
"matched": bool(info.get("matched")),
"year": info.get("year"),
"genre": (info.get("genre") or "").strip(),
"artist": (info.get("artist") or "").strip(),
"title": (info.get("title") or "").strip(),
}
if info.get("error"):
entry["error"] = info["error"]
entries.append(entry)
if entry_callback:
try:
entry_callback(entry)
except Exception:
pass
_emit_progress(progress_callback, total, total, "", "completed", "")
return entries
def _emit_progress(cb: Optional[Callable], idx: int, total: int,
name: str, status: str, err: str = "") -> None:
"""Chiama progress_callback in modo retrocompatibile (4 o 5 args)."""
if not cb:
return
try:
cb(idx, total, name, status, err)
except TypeError:
try:
cb(idx, total, name, status)
except Exception:
pass
except Exception:
pass
# ------------------------------------------------------------------
# Move
# ------------------------------------------------------------------
def move_files(entries: list, target_root: str,
log_callback: Optional[Callable] = None) -> dict:
"""Sposta i file elencati in `<target_root>/<year>/<genre>/<filename>`.
- Se `year` manca → cartella "Unknown Year"
- Se `genre` manca → cartella "Unknown Genre"
- Se il file destinazione esiste gia', aggiunge suffisso _1, _2...
(non sovrascrive mai).
Ritorna un dict:
{
"moved": int, # numero file spostati con successo
"skipped": int, # entries scartate (0 per ora)
"failed": [{path, error}], # errori per file
"operations": [{src, dst}], # log ops riuscite (utile per undo)
}
`log_callback(op)` viene chiamato per ogni operazione riuscita
(streaming alla UI).
"""
target_base = Path(target_root)
try:
target_base.mkdir(parents=True, exist_ok=True)
except OSError as e:
return {"moved": 0, "skipped": 0,
"failed": [{"path": target_root,
"error": f"impossibile creare target: {e}"}],
"operations": []}
result = {"moved": 0, "skipped": 0, "failed": [], "operations": []}
for entry in entries or []:
src_str = (entry or {}).get("path") or ""
src = Path(src_str)
if not src_str:
result["failed"].append({"path": "",
"error": "path mancante"})
continue
if not src.exists():
result["failed"].append({"path": src_str,
"error": "file non esiste"})
continue
year = entry.get("year")
genre = (entry.get("genre") or "").strip()
year_folder = str(year) if year else "Unknown Year"
genre_folder = _sanitize_folder(genre) or "Unknown Genre"
dst_dir = target_base / year_folder / genre_folder
try:
dst_dir.mkdir(parents=True, exist_ok=True)
except OSError as e:
result["failed"].append({"path": src_str,
"error": f"mkdir: {e}"})
continue
dst = dst_dir / src.name
# Anti-overwrite: se il target esiste, aggiungi _1, _2, ...
if dst.exists():
stem = dst.stem
suffix = dst.suffix
i = 1
while dst.exists():
dst = dst_dir / f"{stem}_{i}{suffix}"
i += 1
try:
shutil.move(str(src), str(dst))
except Exception as e:
result["failed"].append({"path": src_str, "error": str(e)})
continue
result["moved"] += 1
op = {"src": src_str, "dst": str(dst)}
result["operations"].append(op)
if log_callback:
try:
log_callback(op)
except Exception:
pass
return result