catalog: modulo AcoustID lookup + move to <year>/<genre>/

This commit is contained in:
luciano committed 2026-08-27 10:00:32 +02:00
1 parent 39a117f08b
commit 9f9e4281b7
4 files changed
+941

No files matched your search

+237
View File
@@ -0,0 +1,237 @@
"""AcoustID + MusicBrainz lookup con cache SQLite.
Data un fingerprint Chromaprint (calcolato via `fpcalc`), interroga il
servizio pubblico AcoustID (https://acoustid.org) per recuperare
metadati sulla traccia: artista, titolo, anno di rilascio piu' antico
e (best-effort) un genere estratto dai tag dei release group MusicBrainz.
I risultati vengono cacheati su SQLite (chiave = fingerprint) per non
sprecare quota API su scan ripetute. La rate limit e' auto-imposta a
~3 req/s per rispettare i limiti pubblici AcoustID.
Nota: l'API AcoustID base NON restituisce sempre il genere — dipende
dai tag presenti nel release group MusicBrainz. Molte tracce dance
electronic hanno pochi tag; il campo `genre` puo' quindi essere vuoto
anche con un match valido. In quel caso il chiamante rimapa in
"Unknown Genre".
"""
from __future__ import annotations
import json
import sqlite3
import time
from pathlib import Path
from typing import Optional
import requests
# App key pubblica per MusicTools. Puo' essere sovrascritta passando
# `app_key` esplicito o via config. Chiavi si ottengono gratis su
# https://acoustid.org/api-key (max ~3 req/s).
_APP_KEY = "8XaBELgH" # placeholder demo — sostituibile via config
_API_URL = "https://api.acoustid.org/v2/lookup"
_REQUEST_TIMEOUT = 20
_RATE_LIMIT_SEC = 0.35 # ~3 req/s max
def _cache_db_path() -> Path:
"""Path del DB di cache dei lookup AcoustID.
Riusa `_get_config_dir` di core.config cosi' finisce nella stessa
cartella di config.json e dedup_cache.db.
"""
from core.config import _get_config_dir
return _get_config_dir() / "catalog_cache.db"
def _init_db(conn: sqlite3.Connection) -> None:
"""Crea (idempotente) lo schema della cache."""
conn.executescript(
"""
CREATE TABLE IF NOT EXISTS lookups(
fingerprint TEXT PRIMARY KEY,
payload TEXT NOT NULL,
cached_at REAL NOT NULL
);
"""
)
conn.commit()
def _open_cache() -> sqlite3.Connection:
"""Apre (creando se serve) la connessione alla cache."""
p = _cache_db_path()
p.parent.mkdir(parents=True, exist_ok=True)
conn = sqlite3.connect(str(p))
_init_db(conn)
return conn
def _get_cached(conn: sqlite3.Connection, fingerprint: str) -> Optional[dict]:
row = conn.execute(
"SELECT payload FROM lookups WHERE fingerprint = ?", (fingerprint,)
).fetchone()
if not row:
return None
try:
return json.loads(row[0])
except Exception:
return None
def _put_cache(conn: sqlite3.Connection, fingerprint: str, data: dict) -> None:
conn.execute(
"INSERT OR REPLACE INTO lookups(fingerprint, payload, cached_at)"
" VALUES (?, ?, ?)",
(fingerprint, json.dumps(data, ensure_ascii=False), time.time()),
)
conn.commit()
# Rate-limit state (globale al processo — vale anche se lookup chiamato
# da thread diversi: non e' esattamente thread-safe ma il worst-case e'
# una richiesta leggermente troppo veloce, non un ban).
_last_request_at = [0.0]
def _throttle() -> None:
"""Attende quel tanto che basta per rispettare _RATE_LIMIT_SEC."""
now = time.monotonic()
elapsed = now - _last_request_at[0]
if elapsed < _RATE_LIMIT_SEC:
time.sleep(_RATE_LIMIT_SEC - elapsed)
_last_request_at[0] = time.monotonic()
def _extract_min_year(releases: list) -> Optional[int]:
"""Trova l'anno piu' antico tra i release. Ignora date invalide."""
year: Optional[int] = None
for rel in releases or []:
date = rel.get("date") if isinstance(rel, dict) else None
if not isinstance(date, dict):
continue
y = date.get("year")
if isinstance(y, int) and y > 0:
if year is None or y < year:
year = y
return year
def _extract_top_genre(releases: list) -> str:
"""Sceglie il tag piu' rilevante dai releasegroup (max `count`)."""
for rel in releases or []:
rg = rel.get("releasegroup") if isinstance(rel, dict) else None
if not isinstance(rg, dict):
continue
tags = rg.get("tags") or []
if not isinstance(tags, list) or not tags:
continue
try:
top = max(tags, key=lambda t: int(t.get("count", 0) or 0))
except (TypeError, ValueError):
top = tags[0]
name = (top.get("name") or "").strip() if isinstance(top, dict) else ""
if name:
return name
return ""
def lookup(fingerprint: str, duration: float,
app_key: str = _APP_KEY) -> dict:
"""Interroga AcoustID (o cache) per una tripla (year, genre, title).
Ritorna sempre un dict con almeno il campo `matched: bool`. In caso
di match valido, aggiunge `year: int|None`, `genre: str`,
`artist: str`, `title: str`. In caso di errore aggiunge `error: str`
ma NON alza eccezione: chiamanti di batch (Cataloga) devono poter
continuare anche se una chiamata singola fallisce.
Cache: risultati validi E "no match" vengono cacheati sul
fingerprint — cosi' scansioni ripetute non re-interrogano l'API.
Errori transitori (rete/HTTP) NON vengono cacheati.
"""
if not fingerprint:
return {"matched": False, "error": "fingerprint vuoto"}
if not app_key:
app_key = _APP_KEY
conn = _open_cache()
try:
cached = _get_cached(conn, fingerprint)
if cached is not None:
return cached
_throttle()
try:
resp = requests.get(
_API_URL,
params={
"client": app_key,
"meta": "recordings+releases+releasegroups",
"duration": int(duration or 0),
"fingerprint": fingerprint,
},
timeout=_REQUEST_TIMEOUT,
)
except requests.RequestException as e:
return {"matched": False, "error": f"network: {e}"}
except Exception as e:
return {"matched": False, "error": str(e)}
if resp.status_code != 200:
return {"matched": False, "error": f"HTTP {resp.status_code}"}
try:
data = resp.json()
except Exception as e:
return {"matched": False, "error": f"JSON malformato: {e}"}
if data.get("status") != "ok":
err = data.get("error", {}) or {}
msg = err.get("message") if isinstance(err, dict) else str(err)
return {"matched": False, "error": msg or "AcoustID status non-ok"}
results = data.get("results") or []
if not results:
result = {"matched": False}
_put_cache(conn, fingerprint, result)
return result
# Match col miglior score
best = max(results, key=lambda r: r.get("score", 0) or 0)
recordings = best.get("recordings") or []
if not recordings:
result = {"matched": False}
_put_cache(conn, fingerprint, result)
return result
rec = recordings[0] or {}
title = (rec.get("title") or "").strip()
artists = rec.get("artists") or []
artist_names = [
(a.get("name") or "").strip()
for a in artists
if isinstance(a, dict) and a.get("name")
]
artist = ", ".join(n for n in artist_names if n)
releases = rec.get("releases") or []
year = _extract_min_year(releases)
genre = _extract_top_genre(releases)
result = {
"matched": bool(year or genre or title),
"year": year,
"genre": genre,
"artist": artist,
"title": title,
}
_put_cache(conn, fingerprint, result)
return result
finally:
try:
conn.close()
except Exception:
pass
+295
View File
@@ -0,0 +1,295 @@
"""Cataloga file audio in sottocartelle <Anno>/<Genere>/ via AcoustID.
Pipeline:
1. Scansiona la cartella (opzionalmente ricorsivo) filtrando per
estensioni audio (AUDIO_EXTENSIONS di core.upgrader).
2. Per ogni file calcola il fingerprint Chromaprint (`fpcalc`,
riusa `core.dedup.compute_fingerprint`).
3. Lookup AcoustID (cache SQLite) per estrarre year + genre +
artist + title.
4. `move_files` sposta le entry selezionate in
`<target>/<Year>/<Genre>/<filename>`. Se manca year/genre usa
"Unknown Year" / "Unknown Genre".
Progress callback firma:
(processed, total, filename, status[, err_msg])
Status: 'computing' | 'lookup' | 'error' | 'stopped' | 'completed'.
`entry_callback(entry)` viene chiamato per ogni file processato,
permettendo alla UI di aggiornare la tabella in streaming.
"""
from __future__ import annotations
import re
import shutil
import threading
from pathlib import Path
from typing import Callable, Optional
from core.acoustid import lookup
from core.dedup import compute_fingerprint
from core.paths import find_fpcalc
from core.upgrader import AUDIO_EXTENSIONS
# ------------------------------------------------------------------
# Stop / interrupt
# ------------------------------------------------------------------
_stop_event = threading.Event()
def request_stop() -> None:
"""Segnala al worker di interrompere la scansione al prossimo file."""
_stop_event.set()
def reset_stop() -> None:
"""Azzera il flag di stop prima di iniziare una nuova scansione."""
_stop_event.clear()
def is_stopped() -> bool:
return _stop_event.is_set()
# ------------------------------------------------------------------
# Sanitize path
# ------------------------------------------------------------------
# Caratteri vietati o problematici in nomi cartella cross-platform.
# `/` e `\` sono trattati a parte perche' MusicBrainz usa spesso
# genre-name-style come "electronic/house": vogliamo rimpiazzarli con
# `_` (non con lo split, che creerebbe path nesting indesiderati).
_FORBIDDEN = re.compile(r'[<>:"|?*\\/]+')
_MULTI_SPACE = re.compile(r"\s+")
def _sanitize_folder(name: str) -> str:
"""Sanitize per path filesystem: rimpiazza chars non validi con `_`.
- Chars vietati Windows (<>:"|?*) + slash → `_`
- Spazi multipli collassati in uno solo
- Trim finale
- Troncamento a 100 char (limite pratico per path lunghi cumulati)
"""
if not name:
return ""
s = _FORBIDDEN.sub("_", name).strip()
s = _MULTI_SPACE.sub(" ", s)
if len(s) > 100:
s = s[:100].rstrip()
return s
# ------------------------------------------------------------------
# Scan
# ------------------------------------------------------------------
def scan_folder(
directory: str,
recursive: bool = True,
progress_callback: Optional[Callable] = None,
entry_callback: Optional[Callable] = None,
) -> list:
"""Scansiona la cartella e ritorna la lista di entry con metadata.
Ogni entry:
{path, size, fingerprint, matched, year, genre, artist, title, error?}
- `progress_callback(idx, total, filename, status[, err])`:
chiamato con status 'computing' | 'lookup' | 'error' | 'stopped'
| 'completed'. Firma retrocompatibile (4 args) supportata.
- `entry_callback(entry)`: chiamato appena ogni file e' processato
(streaming alla UI).
"""
reset_stop()
base = Path(directory)
if not base.exists() or not base.is_dir():
if progress_callback:
_emit_progress(progress_callback, 0, 0, "", "completed", "")
return []
files: list = []
iterator = base.rglob("*") if recursive else base.iterdir()
for f in iterator:
try:
if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS:
files.append(f)
except OSError:
continue
files.sort()
total = len(files)
if total == 0:
if progress_callback:
_emit_progress(progress_callback, 0, 0, "", "completed", "")
return []
fpcalc = find_fpcalc()
entries: list = []
for i, fp_path in enumerate(files, start=1):
if is_stopped():
_emit_progress(progress_callback, i - 1, total, "", "stopped", "")
return entries
try:
size = fp_path.stat().st_size
except OSError as e:
_emit_progress(progress_callback, i, total, fp_path.name,
"error", f"stat: {e}")
continue
# Fingerprint
_emit_progress(progress_callback, i, total, fp_path.name, "computing", "")
res = compute_fingerprint(fpcalc, str(fp_path)) if fpcalc else None
if not res or not res.get("fingerprint"):
err = (res or {}).get("_error") or "fpcalc non disponibile"
entry = {
"path": str(fp_path), "size": size, "fingerprint": "",
"matched": False, "error": err,
"year": None, "genre": "", "artist": "", "title": "",
}
entries.append(entry)
if entry_callback:
try:
entry_callback(entry)
except Exception:
pass
_emit_progress(progress_callback, i, total, fp_path.name,
"error", err)
continue
fp_hash = res["fingerprint"]
duration = float(res.get("duration") or 0)
# AcoustID lookup (cache SQLite dentro core.acoustid)
_emit_progress(progress_callback, i, total, fp_path.name, "lookup", "")
info = lookup(fp_hash, duration)
entry = {
"path": str(fp_path),
"size": size,
"fingerprint": fp_hash,
"matched": bool(info.get("matched")),
"year": info.get("year"),
"genre": (info.get("genre") or "").strip(),
"artist": (info.get("artist") or "").strip(),
"title": (info.get("title") or "").strip(),
}
if info.get("error"):
entry["error"] = info["error"]
entries.append(entry)
if entry_callback:
try:
entry_callback(entry)
except Exception:
pass
_emit_progress(progress_callback, total, total, "", "completed", "")
return entries
def _emit_progress(cb: Optional[Callable], idx: int, total: int,
name: str, status: str, err: str = "") -> None:
"""Chiama progress_callback in modo retrocompatibile (4 o 5 args)."""
if not cb:
return
try:
cb(idx, total, name, status, err)
except TypeError:
try:
cb(idx, total, name, status)
except Exception:
pass
except Exception:
pass
# ------------------------------------------------------------------
# Move
# ------------------------------------------------------------------
def move_files(entries: list, target_root: str,
log_callback: Optional[Callable] = None) -> dict:
"""Sposta i file elencati in `<target_root>/<year>/<genre>/<filename>`.
- Se `year` manca → cartella "Unknown Year"
- Se `genre` manca → cartella "Unknown Genre"
- Se il file destinazione esiste gia', aggiunge suffisso _1, _2...
(non sovrascrive mai).
Ritorna un dict:
{
"moved": int, # numero file spostati con successo
"skipped": int, # entries scartate (0 per ora)
"failed": [{path, error}], # errori per file
"operations": [{src, dst}], # log ops riuscite (utile per undo)
}
`log_callback(op)` viene chiamato per ogni operazione riuscita
(streaming alla UI).
"""
target_base = Path(target_root)
try:
target_base.mkdir(parents=True, exist_ok=True)
except OSError as e:
return {"moved": 0, "skipped": 0,
"failed": [{"path": target_root,
"error": f"impossibile creare target: {e}"}],
"operations": []}
result = {"moved": 0, "skipped": 0, "failed": [], "operations": []}
for entry in entries or []:
src_str = (entry or {}).get("path") or ""
src = Path(src_str)
if not src_str:
result["failed"].append({"path": "",
"error": "path mancante"})
continue
if not src.exists():
result["failed"].append({"path": src_str,
"error": "file non esiste"})
continue
year = entry.get("year")
genre = (entry.get("genre") or "").strip()
year_folder = str(year) if year else "Unknown Year"
genre_folder = _sanitize_folder(genre) or "Unknown Genre"
dst_dir = target_base / year_folder / genre_folder
try:
dst_dir.mkdir(parents=True, exist_ok=True)
except OSError as e:
result["failed"].append({"path": src_str,
"error": f"mkdir: {e}"})
continue
dst = dst_dir / src.name
# Anti-overwrite: se il target esiste, aggiungi _1, _2, ...
if dst.exists():
stem = dst.stem
suffix = dst.suffix
i = 1
while dst.exists():
dst = dst_dir / f"{stem}_{i}{suffix}"
i += 1
try:
shutil.move(str(src), str(dst))
except Exception as e:
result["failed"].append({"path": src_str, "error": str(e)})
continue
result["moved"] += 1
op = {"src": src_str, "dst": str(dst)}
result["operations"].append(op)
if log_callback:
try:
log_callback(op)
except Exception:
pass
return result
+204
View File
@@ -0,0 +1,204 @@
"""Test per core.acoustid — lookup AcoustID + cache SQLite.
Mock su `requests.get` in modo che i test siano offline. Ogni test
isola la cache SQLite in tmp_path via monkeypatch di
`_cache_db_path`.
"""
from __future__ import annotations
from unittest import mock
import pytest
from core import acoustid
# ------------------------------------------------------------------
# Fixture: cache isolata + reset del rate-limit
# ------------------------------------------------------------------
@pytest.fixture
def patched_cache(tmp_path, monkeypatch):
db = tmp_path / "acoustid_cache_test.db"
monkeypatch.setattr(acoustid, "_cache_db_path", lambda: db)
# Rate-limit: azzera cosi' i test non aspettano throttle
monkeypatch.setattr(acoustid, "_RATE_LIMIT_SEC", 0.0)
monkeypatch.setattr(acoustid, "_last_request_at", [0.0])
return db
def _mock_resp(status_code: int = 200, json_data: dict = None):
"""Costruisce un mock di response `requests`."""
m = mock.Mock()
m.status_code = status_code
m.json.return_value = json_data or {}
return m
# ------------------------------------------------------------------
# lookup — cache
# ------------------------------------------------------------------
class TestLookupCache:
def test_lookup_uses_cache(self, patched_cache):
"""Seconda chiamata con lo stesso fingerprint riusa la cache."""
payload = {
"status": "ok",
"results": [{
"score": 0.99,
"recordings": [{
"title": "Some Song",
"artists": [{"name": "Artist X"}],
"releases": [
{"date": {"year": 2005},
"releasegroup": {"tags": [{"name": "House", "count": 5}]}},
],
}],
}],
}
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(200, payload)) as m:
r1 = acoustid.lookup("FP-1", 180.0)
r2 = acoustid.lookup("FP-1", 180.0)
assert m.call_count == 1, "la seconda chiamata deve venire dalla cache"
assert r1 == r2
assert r1["matched"] is True
assert r1["year"] == 2005
assert r1["genre"].lower() == "house"
assert r1["title"] == "Some Song"
assert r1["artist"] == "Artist X"
# ------------------------------------------------------------------
# lookup — no results / errori
# ------------------------------------------------------------------
class TestLookupNoResults:
def test_lookup_no_results_returns_unmatched(self, patched_cache):
payload = {"status": "ok", "results": []}
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(200, payload)):
r = acoustid.lookup("FP-NORESULT", 100.0)
assert r["matched"] is False
# deve essere cacheato
with mock.patch.object(acoustid.requests, "get") as m:
r2 = acoustid.lookup("FP-NORESULT", 100.0)
assert m.call_count == 0
assert r2 == r
def test_lookup_no_recordings_returns_unmatched(self, patched_cache):
"""results presenti ma senza recordings -> matched=False."""
payload = {
"status": "ok",
"results": [{"score": 0.5, "recordings": []}],
}
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(200, payload)):
r = acoustid.lookup("FP-EMPTYREC", 200)
assert r["matched"] is False
# ------------------------------------------------------------------
# lookup — anno minimo
# ------------------------------------------------------------------
class TestLookupYearExtraction:
def test_lookup_extracts_min_year(self, patched_cache):
"""3 releases (2003, 1998, 2010) -> year=1998."""
payload = {
"status": "ok",
"results": [{
"score": 0.95,
"recordings": [{
"title": "Classic",
"artists": [{"name": "Artist"}],
"releases": [
{"date": {"year": 2003}},
{"date": {"year": 1998}},
{"date": {"year": 2010}},
],
}],
}],
}
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(200, payload)):
r = acoustid.lookup("FP-YEAR", 180)
assert r["matched"] is True
assert r["year"] == 1998
def test_lookup_missing_year_becomes_none(self, patched_cache):
"""Nessuna release con year valido -> year=None (matched se ha titolo)."""
payload = {
"status": "ok",
"results": [{
"score": 0.9,
"recordings": [{
"title": "T",
"artists": [{"name": "A"}],
"releases": [{"date": {}}, {"other": 1}],
}],
}],
}
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(200, payload)):
r = acoustid.lookup("FP-NOY", 100)
# matched puo' essere True se ha almeno un titolo
assert r["year"] is None
assert r["title"] == "T"
# ------------------------------------------------------------------
# lookup — errori HTTP / rete
# ------------------------------------------------------------------
class TestLookupErrors:
def test_lookup_returns_error_on_http_failure(self, patched_cache):
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(500, {})):
r = acoustid.lookup("FP-HTTP", 100)
assert r["matched"] is False
assert "HTTP 500" in r.get("error", "")
def test_lookup_returns_error_on_network_failure(self, patched_cache):
with mock.patch.object(acoustid.requests, "get",
side_effect=acoustid.requests.ConnectionError("boom")):
r = acoustid.lookup("FP-NET", 100)
assert r["matched"] is False
assert "network" in r.get("error", "").lower() or "boom" in r.get("error", "")
def test_lookup_empty_fingerprint_returns_unmatched(self, patched_cache):
r = acoustid.lookup("", 100)
assert r["matched"] is False
assert "fingerprint" in r.get("error", "").lower()
def test_lookup_api_status_error(self, patched_cache):
payload = {"status": "error",
"error": {"message": "invalid fingerprint"}}
with mock.patch.object(acoustid.requests, "get",
return_value=_mock_resp(200, payload)):
r = acoustid.lookup("FP-BAD", 100)
assert r["matched"] is False
assert "invalid" in r.get("error", "").lower()
# ------------------------------------------------------------------
# helper: _extract_top_genre
# ------------------------------------------------------------------
class TestGenreExtraction:
def test_top_genre_picks_highest_count(self):
releases = [{
"releasegroup": {
"tags": [
{"name": "electronic", "count": 3},
{"name": "house", "count": 12},
{"name": "dance", "count": 7},
]
}
}]
assert acoustid._extract_top_genre(releases).lower() == "house"
def test_top_genre_no_tags_returns_empty(self):
releases = [{"releasegroup": {"tags": []}}]
assert acoustid._extract_top_genre(releases) == ""
def test_top_genre_no_releasegroup_returns_empty(self):
assert acoustid._extract_top_genre([{}]) == ""
+205
View File
@@ -0,0 +1,205 @@
"""Test per core.catalog — sanitize + move_files + scan_folder.
Focus principale sui casi in cui il piano ha promesso comportamento
esplicito: sanitize dei chars vietati, struttura year/genre, gestione
di anno/genere mancanti, conflitto di filename.
"""
from __future__ import annotations
from pathlib import Path
from unittest import mock
import pytest
from core import catalog
# ------------------------------------------------------------------
# Sanitize
# ------------------------------------------------------------------
class TestSanitizeFolder:
def test_sanitize_folder_removes_forbidden_chars(self):
# `/` (dai tag musicbrainz), `:` (Windows), `?`, `*`, `|`, ecc.
s = catalog._sanitize_folder("electronic/house")
assert "/" not in s
assert "electronic" in s and "house" in s
s2 = catalog._sanitize_folder("prog:rock?")
for ch in '<>:"|?*\\/':
assert ch not in s2
def test_sanitize_folder_collapses_spaces(self):
assert catalog._sanitize_folder(" tech house ") == "tech house"
def test_sanitize_folder_empty_returns_empty(self):
assert catalog._sanitize_folder("") == ""
assert catalog._sanitize_folder(None) == ""
def test_sanitize_folder_truncates_long(self):
s = catalog._sanitize_folder("a" * 300)
assert len(s) <= 100
# ------------------------------------------------------------------
# move_files
# ------------------------------------------------------------------
def _touch(path: Path, size: int = 8) -> Path:
path.parent.mkdir(parents=True, exist_ok=True)
path.write_bytes(b"\x00" * size)
return path
class TestMoveFiles:
def test_move_files_creates_year_genre_structure(self, tmp_path):
"""Un file matched → finisce in <target>/<year>/<genre>/<filename>."""
src = _touch(tmp_path / "source" / "song.mp3")
target = tmp_path / "cat"
entries = [{
"path": str(src), "size": src.stat().st_size,
"matched": True, "year": 2005, "genre": "House",
"artist": "X", "title": "T", "fingerprint": "FP",
}]
res = catalog.move_files(entries, str(target))
assert res["moved"] == 1
assert res["failed"] == []
dst = target / "2005" / "House" / "song.mp3"
assert dst.exists()
assert not src.exists()
# Log operations popolato
assert res["operations"] and res["operations"][0]["src"] == str(src)
assert res["operations"][0]["dst"] == str(dst)
def test_move_files_handles_missing_year_or_genre(self, tmp_path):
"""Year/genre mancanti → Unknown Year / Unknown Genre."""
no_year = _touch(tmp_path / "src" / "no_year.mp3")
no_genre = _touch(tmp_path / "src" / "no_genre.mp3")
no_both = _touch(tmp_path / "src" / "no_both.mp3")
target = tmp_path / "cat"
entries = [
{"path": str(no_year), "matched": True, "year": None,
"genre": "House"},
{"path": str(no_genre), "matched": True, "year": 2010,
"genre": ""},
{"path": str(no_both), "matched": False, "year": None,
"genre": ""},
]
res = catalog.move_files(entries, str(target))
assert res["moved"] == 3
assert (target / "Unknown Year" / "House" / "no_year.mp3").exists()
assert (target / "2010" / "Unknown Genre" / "no_genre.mp3").exists()
assert (target / "Unknown Year" / "Unknown Genre"
/ "no_both.mp3").exists()
def test_move_files_dedup_conflicting_filenames(self, tmp_path):
"""Se un file con lo stesso nome esiste gia' nel target, aggiungi
suffisso _1, _2, ... (mai overwrite)."""
src1 = _touch(tmp_path / "srcA" / "song.mp3", size=10)
src2 = _touch(tmp_path / "srcB" / "song.mp3", size=20)
target = tmp_path / "cat"
entries = [
{"path": str(src1), "matched": True, "year": 2000,
"genre": "Rock"},
{"path": str(src2), "matched": True, "year": 2000,
"genre": "Rock"},
]
res = catalog.move_files(entries, str(target))
assert res["moved"] == 2
d1 = target / "2000" / "Rock" / "song.mp3"
d2 = target / "2000" / "Rock" / "song_1.mp3"
assert d1.exists() and d2.exists()
# Contenuto preservato dal move (src1 = 10 byte, src2 = 20 byte).
# L'ordine di iterazione garantisce che song.mp3 = src1.
assert d1.stat().st_size == 10
assert d2.stat().st_size == 20
def test_move_files_sanitizes_forbidden_genre(self, tmp_path):
"""Genere con `/` (tipico di musicbrainz) va sanitizzato in `_`."""
src = _touch(tmp_path / "src" / "song.mp3")
target = tmp_path / "cat"
entries = [{
"path": str(src), "matched": True, "year": 2020,
"genre": "electronic/house",
}]
res = catalog.move_files(entries, str(target))
assert res["moved"] == 1
# NON deve creare "electronic" e dentro "house": e' un solo nome.
assert not (target / "2020" / "electronic").is_dir()
# La cartella deve contenere entrambe le parti sanitizzate
year_dir = target / "2020"
subdirs = [p.name for p in year_dir.iterdir() if p.is_dir()]
assert len(subdirs) == 1
assert "/" not in subdirs[0]
assert "electronic" in subdirs[0] and "house" in subdirs[0]
def test_move_files_missing_source_reports_failed(self, tmp_path):
"""File che non esiste piu' → finisce in failed, non alza."""
target = tmp_path / "cat"
entries = [{
"path": str(tmp_path / "does_not_exist.mp3"),
"matched": True, "year": 2000, "genre": "X",
}]
res = catalog.move_files(entries, str(target))
assert res["moved"] == 0
assert len(res["failed"]) == 1
assert "non esiste" in res["failed"][0]["error"].lower()
def test_move_files_empty_list_returns_zero(self, tmp_path):
res = catalog.move_files([], str(tmp_path / "cat"))
assert res["moved"] == 0
assert res["failed"] == []
assert res["operations"] == []
# ------------------------------------------------------------------
# scan_folder — smoke integration test (mock fpcalc + lookup)
# ------------------------------------------------------------------
class TestScanFolder:
def test_scan_folder_empty_directory(self, tmp_path):
# Cartella senza file audio → lista vuota, no crash
with mock.patch.object(catalog, "find_fpcalc",
return_value="/fake/fpcalc"):
entries = catalog.scan_folder(str(tmp_path), recursive=False)
assert entries == []
def test_scan_folder_processes_files(self, tmp_path):
# Un file audio fake + mock di fpcalc + lookup
(tmp_path / "a.mp3").write_bytes(b"\x00" * 100)
(tmp_path / "readme.txt").write_text("hi")
with mock.patch.object(catalog, "find_fpcalc",
return_value="/fake/fpcalc"), \
mock.patch.object(catalog, "compute_fingerprint",
return_value={"fingerprint": "FP", "duration": 100}), \
mock.patch.object(catalog, "lookup",
return_value={
"matched": True, "year": 2018,
"genre": "House", "artist": "A",
"title": "T",
}):
entries = catalog.scan_folder(str(tmp_path), recursive=False)
assert len(entries) == 1
e = entries[0]
assert e["matched"] is True
assert e["year"] == 2018
assert e["genre"] == "House"
assert e["fingerprint"] == "FP"
def test_scan_folder_records_fpcalc_error(self, tmp_path):
"""fpcalc fallito → entry con matched=False + error."""
(tmp_path / "broken.mp3").write_bytes(b"\x00" * 10)
with mock.patch.object(catalog, "find_fpcalc",
return_value="/fake/fpcalc"), \
mock.patch.object(catalog, "compute_fingerprint",
return_value={"_error": "fingerprint vuoto"}):
entries = catalog.scan_folder(str(tmp_path), recursive=False)
assert len(entries) == 1
assert entries[0]["matched"] is False
assert "vuoto" in entries[0]["error"]