From 9f9e4281b73ec4134a9051be6ad79027cbde69f0 Mon Sep 17 00:00:00 2001 From: luciano Date: Thu, 27 Aug 2026 10:00:32 +0200 Subject: [PATCH] catalog: modulo AcoustID lookup + move to // --- core/acoustid.py | 237 +++++++++++++++++++++++++++++++++ core/catalog.py | 295 +++++++++++++++++++++++++++++++++++++++++ tests/test_acoustid.py | 204 ++++++++++++++++++++++++++++ tests/test_catalog.py | 205 ++++++++++++++++++++++++++++ 4 files changed, 941 insertions(+) create mode 100644 core/acoustid.py create mode 100644 core/catalog.py create mode 100644 tests/test_acoustid.py create mode 100644 tests/test_catalog.py diff --git a/core/acoustid.py b/core/acoustid.py new file mode 100644 index 0000000..46f4cdb --- /dev/null +++ b/core/acoustid.py @@ -0,0 +1,237 @@ +"""AcoustID + MusicBrainz lookup con cache SQLite. + +Data un fingerprint Chromaprint (calcolato via `fpcalc`), interroga il +servizio pubblico AcoustID (https://acoustid.org) per recuperare +metadati sulla traccia: artista, titolo, anno di rilascio piu' antico +e (best-effort) un genere estratto dai tag dei release group MusicBrainz. + +I risultati vengono cacheati su SQLite (chiave = fingerprint) per non +sprecare quota API su scan ripetute. La rate limit e' auto-imposta a +~3 req/s per rispettare i limiti pubblici AcoustID. + +Nota: l'API AcoustID base NON restituisce sempre il genere — dipende +dai tag presenti nel release group MusicBrainz. Molte tracce dance +electronic hanno pochi tag; il campo `genre` puo' quindi essere vuoto +anche con un match valido. In quel caso il chiamante rimapa in +"Unknown Genre". +""" + +from __future__ import annotations + +import json +import sqlite3 +import time +from pathlib import Path +from typing import Optional + +import requests + + +# App key pubblica per MusicTools. Puo' essere sovrascritta passando +# `app_key` esplicito o via config. Chiavi si ottengono gratis su +# https://acoustid.org/api-key (max ~3 req/s). +_APP_KEY = "8XaBELgH" # placeholder demo — sostituibile via config +_API_URL = "https://api.acoustid.org/v2/lookup" +_REQUEST_TIMEOUT = 20 +_RATE_LIMIT_SEC = 0.35 # ~3 req/s max + + +def _cache_db_path() -> Path: + """Path del DB di cache dei lookup AcoustID. + + Riusa `_get_config_dir` di core.config cosi' finisce nella stessa + cartella di config.json e dedup_cache.db. + """ + from core.config import _get_config_dir + return _get_config_dir() / "catalog_cache.db" + + +def _init_db(conn: sqlite3.Connection) -> None: + """Crea (idempotente) lo schema della cache.""" + conn.executescript( + """ + CREATE TABLE IF NOT EXISTS lookups( + fingerprint TEXT PRIMARY KEY, + payload TEXT NOT NULL, + cached_at REAL NOT NULL + ); + """ + ) + conn.commit() + + +def _open_cache() -> sqlite3.Connection: + """Apre (creando se serve) la connessione alla cache.""" + p = _cache_db_path() + p.parent.mkdir(parents=True, exist_ok=True) + conn = sqlite3.connect(str(p)) + _init_db(conn) + return conn + + +def _get_cached(conn: sqlite3.Connection, fingerprint: str) -> Optional[dict]: + row = conn.execute( + "SELECT payload FROM lookups WHERE fingerprint = ?", (fingerprint,) + ).fetchone() + if not row: + return None + try: + return json.loads(row[0]) + except Exception: + return None + + +def _put_cache(conn: sqlite3.Connection, fingerprint: str, data: dict) -> None: + conn.execute( + "INSERT OR REPLACE INTO lookups(fingerprint, payload, cached_at)" + " VALUES (?, ?, ?)", + (fingerprint, json.dumps(data, ensure_ascii=False), time.time()), + ) + conn.commit() + + +# Rate-limit state (globale al processo — vale anche se lookup chiamato +# da thread diversi: non e' esattamente thread-safe ma il worst-case e' +# una richiesta leggermente troppo veloce, non un ban). +_last_request_at = [0.0] + + +def _throttle() -> None: + """Attende quel tanto che basta per rispettare _RATE_LIMIT_SEC.""" + now = time.monotonic() + elapsed = now - _last_request_at[0] + if elapsed < _RATE_LIMIT_SEC: + time.sleep(_RATE_LIMIT_SEC - elapsed) + _last_request_at[0] = time.monotonic() + + +def _extract_min_year(releases: list) -> Optional[int]: + """Trova l'anno piu' antico tra i release. Ignora date invalide.""" + year: Optional[int] = None + for rel in releases or []: + date = rel.get("date") if isinstance(rel, dict) else None + if not isinstance(date, dict): + continue + y = date.get("year") + if isinstance(y, int) and y > 0: + if year is None or y < year: + year = y + return year + + +def _extract_top_genre(releases: list) -> str: + """Sceglie il tag piu' rilevante dai releasegroup (max `count`).""" + for rel in releases or []: + rg = rel.get("releasegroup") if isinstance(rel, dict) else None + if not isinstance(rg, dict): + continue + tags = rg.get("tags") or [] + if not isinstance(tags, list) or not tags: + continue + try: + top = max(tags, key=lambda t: int(t.get("count", 0) or 0)) + except (TypeError, ValueError): + top = tags[0] + name = (top.get("name") or "").strip() if isinstance(top, dict) else "" + if name: + return name + return "" + + +def lookup(fingerprint: str, duration: float, + app_key: str = _APP_KEY) -> dict: + """Interroga AcoustID (o cache) per una tripla (year, genre, title). + + Ritorna sempre un dict con almeno il campo `matched: bool`. In caso + di match valido, aggiunge `year: int|None`, `genre: str`, + `artist: str`, `title: str`. In caso di errore aggiunge `error: str` + ma NON alza eccezione: chiamanti di batch (Cataloga) devono poter + continuare anche se una chiamata singola fallisce. + + Cache: risultati validi E "no match" vengono cacheati sul + fingerprint — cosi' scansioni ripetute non re-interrogano l'API. + Errori transitori (rete/HTTP) NON vengono cacheati. + """ + if not fingerprint: + return {"matched": False, "error": "fingerprint vuoto"} + if not app_key: + app_key = _APP_KEY + + conn = _open_cache() + try: + cached = _get_cached(conn, fingerprint) + if cached is not None: + return cached + + _throttle() + try: + resp = requests.get( + _API_URL, + params={ + "client": app_key, + "meta": "recordings+releases+releasegroups", + "duration": int(duration or 0), + "fingerprint": fingerprint, + }, + timeout=_REQUEST_TIMEOUT, + ) + except requests.RequestException as e: + return {"matched": False, "error": f"network: {e}"} + except Exception as e: + return {"matched": False, "error": str(e)} + + if resp.status_code != 200: + return {"matched": False, "error": f"HTTP {resp.status_code}"} + + try: + data = resp.json() + except Exception as e: + return {"matched": False, "error": f"JSON malformato: {e}"} + + if data.get("status") != "ok": + err = data.get("error", {}) or {} + msg = err.get("message") if isinstance(err, dict) else str(err) + return {"matched": False, "error": msg or "AcoustID status non-ok"} + + results = data.get("results") or [] + if not results: + result = {"matched": False} + _put_cache(conn, fingerprint, result) + return result + + # Match col miglior score + best = max(results, key=lambda r: r.get("score", 0) or 0) + recordings = best.get("recordings") or [] + if not recordings: + result = {"matched": False} + _put_cache(conn, fingerprint, result) + return result + + rec = recordings[0] or {} + title = (rec.get("title") or "").strip() + artists = rec.get("artists") or [] + artist_names = [ + (a.get("name") or "").strip() + for a in artists + if isinstance(a, dict) and a.get("name") + ] + artist = ", ".join(n for n in artist_names if n) + + releases = rec.get("releases") or [] + year = _extract_min_year(releases) + genre = _extract_top_genre(releases) + + result = { + "matched": bool(year or genre or title), + "year": year, + "genre": genre, + "artist": artist, + "title": title, + } + _put_cache(conn, fingerprint, result) + return result + finally: + try: + conn.close() + except Exception: + pass diff --git a/core/catalog.py b/core/catalog.py new file mode 100644 index 0000000..430a79d --- /dev/null +++ b/core/catalog.py @@ -0,0 +1,295 @@ +"""Cataloga file audio in sottocartelle // via AcoustID. + +Pipeline: + 1. Scansiona la cartella (opzionalmente ricorsivo) filtrando per + estensioni audio (AUDIO_EXTENSIONS di core.upgrader). + 2. Per ogni file calcola il fingerprint Chromaprint (`fpcalc`, + riusa `core.dedup.compute_fingerprint`). + 3. Lookup AcoustID (cache SQLite) per estrarre year + genre + + artist + title. + 4. `move_files` sposta le entry selezionate in + `///`. Se manca year/genre usa + "Unknown Year" / "Unknown Genre". + +Progress callback firma: + (processed, total, filename, status[, err_msg]) +Status: 'computing' | 'lookup' | 'error' | 'stopped' | 'completed'. + +`entry_callback(entry)` viene chiamato per ogni file processato, +permettendo alla UI di aggiornare la tabella in streaming. +""" + +from __future__ import annotations + +import re +import shutil +import threading +from pathlib import Path +from typing import Callable, Optional + +from core.acoustid import lookup +from core.dedup import compute_fingerprint +from core.paths import find_fpcalc +from core.upgrader import AUDIO_EXTENSIONS + + +# ------------------------------------------------------------------ +# Stop / interrupt +# ------------------------------------------------------------------ +_stop_event = threading.Event() + + +def request_stop() -> None: + """Segnala al worker di interrompere la scansione al prossimo file.""" + _stop_event.set() + + +def reset_stop() -> None: + """Azzera il flag di stop prima di iniziare una nuova scansione.""" + _stop_event.clear() + + +def is_stopped() -> bool: + return _stop_event.is_set() + + +# ------------------------------------------------------------------ +# Sanitize path +# ------------------------------------------------------------------ +# Caratteri vietati o problematici in nomi cartella cross-platform. +# `/` e `\` sono trattati a parte perche' MusicBrainz usa spesso +# genre-name-style come "electronic/house": vogliamo rimpiazzarli con +# `_` (non con lo split, che creerebbe path nesting indesiderati). +_FORBIDDEN = re.compile(r'[<>:"|?*\\/]+') +_MULTI_SPACE = re.compile(r"\s+") + + +def _sanitize_folder(name: str) -> str: + """Sanitize per path filesystem: rimpiazza chars non validi con `_`. + + - Chars vietati Windows (<>:"|?*) + slash → `_` + - Spazi multipli collassati in uno solo + - Trim finale + - Troncamento a 100 char (limite pratico per path lunghi cumulati) + """ + if not name: + return "" + s = _FORBIDDEN.sub("_", name).strip() + s = _MULTI_SPACE.sub(" ", s) + if len(s) > 100: + s = s[:100].rstrip() + return s + + +# ------------------------------------------------------------------ +# Scan +# ------------------------------------------------------------------ +def scan_folder( + directory: str, + recursive: bool = True, + progress_callback: Optional[Callable] = None, + entry_callback: Optional[Callable] = None, +) -> list: + """Scansiona la cartella e ritorna la lista di entry con metadata. + + Ogni entry: + {path, size, fingerprint, matched, year, genre, artist, title, error?} + + - `progress_callback(idx, total, filename, status[, err])`: + chiamato con status 'computing' | 'lookup' | 'error' | 'stopped' + | 'completed'. Firma retrocompatibile (4 args) supportata. + - `entry_callback(entry)`: chiamato appena ogni file e' processato + (streaming alla UI). + """ + reset_stop() + base = Path(directory) + if not base.exists() or not base.is_dir(): + if progress_callback: + _emit_progress(progress_callback, 0, 0, "", "completed", "") + return [] + + files: list = [] + iterator = base.rglob("*") if recursive else base.iterdir() + for f in iterator: + try: + if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS: + files.append(f) + except OSError: + continue + files.sort() + + total = len(files) + if total == 0: + if progress_callback: + _emit_progress(progress_callback, 0, 0, "", "completed", "") + return [] + + fpcalc = find_fpcalc() + + entries: list = [] + for i, fp_path in enumerate(files, start=1): + if is_stopped(): + _emit_progress(progress_callback, i - 1, total, "", "stopped", "") + return entries + + try: + size = fp_path.stat().st_size + except OSError as e: + _emit_progress(progress_callback, i, total, fp_path.name, + "error", f"stat: {e}") + continue + + # Fingerprint + _emit_progress(progress_callback, i, total, fp_path.name, "computing", "") + res = compute_fingerprint(fpcalc, str(fp_path)) if fpcalc else None + if not res or not res.get("fingerprint"): + err = (res or {}).get("_error") or "fpcalc non disponibile" + entry = { + "path": str(fp_path), "size": size, "fingerprint": "", + "matched": False, "error": err, + "year": None, "genre": "", "artist": "", "title": "", + } + entries.append(entry) + if entry_callback: + try: + entry_callback(entry) + except Exception: + pass + _emit_progress(progress_callback, i, total, fp_path.name, + "error", err) + continue + + fp_hash = res["fingerprint"] + duration = float(res.get("duration") or 0) + + # AcoustID lookup (cache SQLite dentro core.acoustid) + _emit_progress(progress_callback, i, total, fp_path.name, "lookup", "") + info = lookup(fp_hash, duration) + + entry = { + "path": str(fp_path), + "size": size, + "fingerprint": fp_hash, + "matched": bool(info.get("matched")), + "year": info.get("year"), + "genre": (info.get("genre") or "").strip(), + "artist": (info.get("artist") or "").strip(), + "title": (info.get("title") or "").strip(), + } + if info.get("error"): + entry["error"] = info["error"] + + entries.append(entry) + if entry_callback: + try: + entry_callback(entry) + except Exception: + pass + + _emit_progress(progress_callback, total, total, "", "completed", "") + return entries + + +def _emit_progress(cb: Optional[Callable], idx: int, total: int, + name: str, status: str, err: str = "") -> None: + """Chiama progress_callback in modo retrocompatibile (4 o 5 args).""" + if not cb: + return + try: + cb(idx, total, name, status, err) + except TypeError: + try: + cb(idx, total, name, status) + except Exception: + pass + except Exception: + pass + + +# ------------------------------------------------------------------ +# Move +# ------------------------------------------------------------------ +def move_files(entries: list, target_root: str, + log_callback: Optional[Callable] = None) -> dict: + """Sposta i file elencati in `///`. + + - Se `year` manca → cartella "Unknown Year" + - Se `genre` manca → cartella "Unknown Genre" + - Se il file destinazione esiste gia', aggiunge suffisso _1, _2... + (non sovrascrive mai). + + Ritorna un dict: + { + "moved": int, # numero file spostati con successo + "skipped": int, # entries scartate (0 per ora) + "failed": [{path, error}], # errori per file + "operations": [{src, dst}], # log ops riuscite (utile per undo) + } + + `log_callback(op)` viene chiamato per ogni operazione riuscita + (streaming alla UI). + """ + target_base = Path(target_root) + try: + target_base.mkdir(parents=True, exist_ok=True) + except OSError as e: + return {"moved": 0, "skipped": 0, + "failed": [{"path": target_root, + "error": f"impossibile creare target: {e}"}], + "operations": []} + + result = {"moved": 0, "skipped": 0, "failed": [], "operations": []} + + for entry in entries or []: + src_str = (entry or {}).get("path") or "" + src = Path(src_str) + if not src_str: + result["failed"].append({"path": "", + "error": "path mancante"}) + continue + if not src.exists(): + result["failed"].append({"path": src_str, + "error": "file non esiste"}) + continue + + year = entry.get("year") + genre = (entry.get("genre") or "").strip() + + year_folder = str(year) if year else "Unknown Year" + genre_folder = _sanitize_folder(genre) or "Unknown Genre" + + dst_dir = target_base / year_folder / genre_folder + try: + dst_dir.mkdir(parents=True, exist_ok=True) + except OSError as e: + result["failed"].append({"path": src_str, + "error": f"mkdir: {e}"}) + continue + + dst = dst_dir / src.name + + # Anti-overwrite: se il target esiste, aggiungi _1, _2, ... + if dst.exists(): + stem = dst.stem + suffix = dst.suffix + i = 1 + while dst.exists(): + dst = dst_dir / f"{stem}_{i}{suffix}" + i += 1 + + try: + shutil.move(str(src), str(dst)) + except Exception as e: + result["failed"].append({"path": src_str, "error": str(e)}) + continue + + result["moved"] += 1 + op = {"src": src_str, "dst": str(dst)} + result["operations"].append(op) + if log_callback: + try: + log_callback(op) + except Exception: + pass + + return result diff --git a/tests/test_acoustid.py b/tests/test_acoustid.py new file mode 100644 index 0000000..3bdb928 --- /dev/null +++ b/tests/test_acoustid.py @@ -0,0 +1,204 @@ +"""Test per core.acoustid — lookup AcoustID + cache SQLite. + +Mock su `requests.get` in modo che i test siano offline. Ogni test +isola la cache SQLite in tmp_path via monkeypatch di +`_cache_db_path`. +""" + +from __future__ import annotations + +from unittest import mock + +import pytest + +from core import acoustid + + +# ------------------------------------------------------------------ +# Fixture: cache isolata + reset del rate-limit +# ------------------------------------------------------------------ +@pytest.fixture +def patched_cache(tmp_path, monkeypatch): + db = tmp_path / "acoustid_cache_test.db" + monkeypatch.setattr(acoustid, "_cache_db_path", lambda: db) + # Rate-limit: azzera cosi' i test non aspettano throttle + monkeypatch.setattr(acoustid, "_RATE_LIMIT_SEC", 0.0) + monkeypatch.setattr(acoustid, "_last_request_at", [0.0]) + return db + + +def _mock_resp(status_code: int = 200, json_data: dict = None): + """Costruisce un mock di response `requests`.""" + m = mock.Mock() + m.status_code = status_code + m.json.return_value = json_data or {} + return m + + +# ------------------------------------------------------------------ +# lookup — cache +# ------------------------------------------------------------------ +class TestLookupCache: + def test_lookup_uses_cache(self, patched_cache): + """Seconda chiamata con lo stesso fingerprint riusa la cache.""" + payload = { + "status": "ok", + "results": [{ + "score": 0.99, + "recordings": [{ + "title": "Some Song", + "artists": [{"name": "Artist X"}], + "releases": [ + {"date": {"year": 2005}, + "releasegroup": {"tags": [{"name": "House", "count": 5}]}}, + ], + }], + }], + } + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(200, payload)) as m: + r1 = acoustid.lookup("FP-1", 180.0) + r2 = acoustid.lookup("FP-1", 180.0) + + assert m.call_count == 1, "la seconda chiamata deve venire dalla cache" + assert r1 == r2 + assert r1["matched"] is True + assert r1["year"] == 2005 + assert r1["genre"].lower() == "house" + assert r1["title"] == "Some Song" + assert r1["artist"] == "Artist X" + + +# ------------------------------------------------------------------ +# lookup — no results / errori +# ------------------------------------------------------------------ +class TestLookupNoResults: + def test_lookup_no_results_returns_unmatched(self, patched_cache): + payload = {"status": "ok", "results": []} + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(200, payload)): + r = acoustid.lookup("FP-NORESULT", 100.0) + + assert r["matched"] is False + # deve essere cacheato + with mock.patch.object(acoustid.requests, "get") as m: + r2 = acoustid.lookup("FP-NORESULT", 100.0) + assert m.call_count == 0 + assert r2 == r + + def test_lookup_no_recordings_returns_unmatched(self, patched_cache): + """results presenti ma senza recordings -> matched=False.""" + payload = { + "status": "ok", + "results": [{"score": 0.5, "recordings": []}], + } + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(200, payload)): + r = acoustid.lookup("FP-EMPTYREC", 200) + assert r["matched"] is False + + +# ------------------------------------------------------------------ +# lookup — anno minimo +# ------------------------------------------------------------------ +class TestLookupYearExtraction: + def test_lookup_extracts_min_year(self, patched_cache): + """3 releases (2003, 1998, 2010) -> year=1998.""" + payload = { + "status": "ok", + "results": [{ + "score": 0.95, + "recordings": [{ + "title": "Classic", + "artists": [{"name": "Artist"}], + "releases": [ + {"date": {"year": 2003}}, + {"date": {"year": 1998}}, + {"date": {"year": 2010}}, + ], + }], + }], + } + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(200, payload)): + r = acoustid.lookup("FP-YEAR", 180) + + assert r["matched"] is True + assert r["year"] == 1998 + + def test_lookup_missing_year_becomes_none(self, patched_cache): + """Nessuna release con year valido -> year=None (matched se ha titolo).""" + payload = { + "status": "ok", + "results": [{ + "score": 0.9, + "recordings": [{ + "title": "T", + "artists": [{"name": "A"}], + "releases": [{"date": {}}, {"other": 1}], + }], + }], + } + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(200, payload)): + r = acoustid.lookup("FP-NOY", 100) + # matched puo' essere True se ha almeno un titolo + assert r["year"] is None + assert r["title"] == "T" + + +# ------------------------------------------------------------------ +# lookup — errori HTTP / rete +# ------------------------------------------------------------------ +class TestLookupErrors: + def test_lookup_returns_error_on_http_failure(self, patched_cache): + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(500, {})): + r = acoustid.lookup("FP-HTTP", 100) + assert r["matched"] is False + assert "HTTP 500" in r.get("error", "") + + def test_lookup_returns_error_on_network_failure(self, patched_cache): + with mock.patch.object(acoustid.requests, "get", + side_effect=acoustid.requests.ConnectionError("boom")): + r = acoustid.lookup("FP-NET", 100) + assert r["matched"] is False + assert "network" in r.get("error", "").lower() or "boom" in r.get("error", "") + + def test_lookup_empty_fingerprint_returns_unmatched(self, patched_cache): + r = acoustid.lookup("", 100) + assert r["matched"] is False + assert "fingerprint" in r.get("error", "").lower() + + def test_lookup_api_status_error(self, patched_cache): + payload = {"status": "error", + "error": {"message": "invalid fingerprint"}} + with mock.patch.object(acoustid.requests, "get", + return_value=_mock_resp(200, payload)): + r = acoustid.lookup("FP-BAD", 100) + assert r["matched"] is False + assert "invalid" in r.get("error", "").lower() + + +# ------------------------------------------------------------------ +# helper: _extract_top_genre +# ------------------------------------------------------------------ +class TestGenreExtraction: + def test_top_genre_picks_highest_count(self): + releases = [{ + "releasegroup": { + "tags": [ + {"name": "electronic", "count": 3}, + {"name": "house", "count": 12}, + {"name": "dance", "count": 7}, + ] + } + }] + assert acoustid._extract_top_genre(releases).lower() == "house" + + def test_top_genre_no_tags_returns_empty(self): + releases = [{"releasegroup": {"tags": []}}] + assert acoustid._extract_top_genre(releases) == "" + + def test_top_genre_no_releasegroup_returns_empty(self): + assert acoustid._extract_top_genre([{}]) == "" diff --git a/tests/test_catalog.py b/tests/test_catalog.py new file mode 100644 index 0000000..ad6d4eb --- /dev/null +++ b/tests/test_catalog.py @@ -0,0 +1,205 @@ +"""Test per core.catalog — sanitize + move_files + scan_folder. + +Focus principale sui casi in cui il piano ha promesso comportamento +esplicito: sanitize dei chars vietati, struttura year/genre, gestione +di anno/genere mancanti, conflitto di filename. +""" + +from __future__ import annotations + +from pathlib import Path +from unittest import mock + +import pytest + +from core import catalog + + +# ------------------------------------------------------------------ +# Sanitize +# ------------------------------------------------------------------ +class TestSanitizeFolder: + def test_sanitize_folder_removes_forbidden_chars(self): + # `/` (dai tag musicbrainz), `:` (Windows), `?`, `*`, `|`, ecc. + s = catalog._sanitize_folder("electronic/house") + assert "/" not in s + assert "electronic" in s and "house" in s + + s2 = catalog._sanitize_folder("prog:rock?") + for ch in '<>:"|?*\\/': + assert ch not in s2 + + def test_sanitize_folder_collapses_spaces(self): + assert catalog._sanitize_folder(" tech house ") == "tech house" + + def test_sanitize_folder_empty_returns_empty(self): + assert catalog._sanitize_folder("") == "" + assert catalog._sanitize_folder(None) == "" + + def test_sanitize_folder_truncates_long(self): + s = catalog._sanitize_folder("a" * 300) + assert len(s) <= 100 + + +# ------------------------------------------------------------------ +# move_files +# ------------------------------------------------------------------ +def _touch(path: Path, size: int = 8) -> Path: + path.parent.mkdir(parents=True, exist_ok=True) + path.write_bytes(b"\x00" * size) + return path + + +class TestMoveFiles: + def test_move_files_creates_year_genre_structure(self, tmp_path): + """Un file matched → finisce in ///.""" + src = _touch(tmp_path / "source" / "song.mp3") + target = tmp_path / "cat" + entries = [{ + "path": str(src), "size": src.stat().st_size, + "matched": True, "year": 2005, "genre": "House", + "artist": "X", "title": "T", "fingerprint": "FP", + }] + + res = catalog.move_files(entries, str(target)) + + assert res["moved"] == 1 + assert res["failed"] == [] + dst = target / "2005" / "House" / "song.mp3" + assert dst.exists() + assert not src.exists() + # Log operations popolato + assert res["operations"] and res["operations"][0]["src"] == str(src) + assert res["operations"][0]["dst"] == str(dst) + + def test_move_files_handles_missing_year_or_genre(self, tmp_path): + """Year/genre mancanti → Unknown Year / Unknown Genre.""" + no_year = _touch(tmp_path / "src" / "no_year.mp3") + no_genre = _touch(tmp_path / "src" / "no_genre.mp3") + no_both = _touch(tmp_path / "src" / "no_both.mp3") + + target = tmp_path / "cat" + entries = [ + {"path": str(no_year), "matched": True, "year": None, + "genre": "House"}, + {"path": str(no_genre), "matched": True, "year": 2010, + "genre": ""}, + {"path": str(no_both), "matched": False, "year": None, + "genre": ""}, + ] + + res = catalog.move_files(entries, str(target)) + assert res["moved"] == 3 + assert (target / "Unknown Year" / "House" / "no_year.mp3").exists() + assert (target / "2010" / "Unknown Genre" / "no_genre.mp3").exists() + assert (target / "Unknown Year" / "Unknown Genre" + / "no_both.mp3").exists() + + def test_move_files_dedup_conflicting_filenames(self, tmp_path): + """Se un file con lo stesso nome esiste gia' nel target, aggiungi + suffisso _1, _2, ... (mai overwrite).""" + src1 = _touch(tmp_path / "srcA" / "song.mp3", size=10) + src2 = _touch(tmp_path / "srcB" / "song.mp3", size=20) + target = tmp_path / "cat" + + entries = [ + {"path": str(src1), "matched": True, "year": 2000, + "genre": "Rock"}, + {"path": str(src2), "matched": True, "year": 2000, + "genre": "Rock"}, + ] + res = catalog.move_files(entries, str(target)) + assert res["moved"] == 2 + d1 = target / "2000" / "Rock" / "song.mp3" + d2 = target / "2000" / "Rock" / "song_1.mp3" + assert d1.exists() and d2.exists() + # Contenuto preservato dal move (src1 = 10 byte, src2 = 20 byte). + # L'ordine di iterazione garantisce che song.mp3 = src1. + assert d1.stat().st_size == 10 + assert d2.stat().st_size == 20 + + def test_move_files_sanitizes_forbidden_genre(self, tmp_path): + """Genere con `/` (tipico di musicbrainz) va sanitizzato in `_`.""" + src = _touch(tmp_path / "src" / "song.mp3") + target = tmp_path / "cat" + entries = [{ + "path": str(src), "matched": True, "year": 2020, + "genre": "electronic/house", + }] + + res = catalog.move_files(entries, str(target)) + assert res["moved"] == 1 + # NON deve creare "electronic" e dentro "house": e' un solo nome. + assert not (target / "2020" / "electronic").is_dir() + # La cartella deve contenere entrambe le parti sanitizzate + year_dir = target / "2020" + subdirs = [p.name for p in year_dir.iterdir() if p.is_dir()] + assert len(subdirs) == 1 + assert "/" not in subdirs[0] + assert "electronic" in subdirs[0] and "house" in subdirs[0] + + def test_move_files_missing_source_reports_failed(self, tmp_path): + """File che non esiste piu' → finisce in failed, non alza.""" + target = tmp_path / "cat" + entries = [{ + "path": str(tmp_path / "does_not_exist.mp3"), + "matched": True, "year": 2000, "genre": "X", + }] + res = catalog.move_files(entries, str(target)) + assert res["moved"] == 0 + assert len(res["failed"]) == 1 + assert "non esiste" in res["failed"][0]["error"].lower() + + def test_move_files_empty_list_returns_zero(self, tmp_path): + res = catalog.move_files([], str(tmp_path / "cat")) + assert res["moved"] == 0 + assert res["failed"] == [] + assert res["operations"] == [] + + +# ------------------------------------------------------------------ +# scan_folder — smoke integration test (mock fpcalc + lookup) +# ------------------------------------------------------------------ +class TestScanFolder: + def test_scan_folder_empty_directory(self, tmp_path): + # Cartella senza file audio → lista vuota, no crash + with mock.patch.object(catalog, "find_fpcalc", + return_value="/fake/fpcalc"): + entries = catalog.scan_folder(str(tmp_path), recursive=False) + assert entries == [] + + def test_scan_folder_processes_files(self, tmp_path): + # Un file audio fake + mock di fpcalc + lookup + (tmp_path / "a.mp3").write_bytes(b"\x00" * 100) + (tmp_path / "readme.txt").write_text("hi") + + with mock.patch.object(catalog, "find_fpcalc", + return_value="/fake/fpcalc"), \ + mock.patch.object(catalog, "compute_fingerprint", + return_value={"fingerprint": "FP", "duration": 100}), \ + mock.patch.object(catalog, "lookup", + return_value={ + "matched": True, "year": 2018, + "genre": "House", "artist": "A", + "title": "T", + }): + entries = catalog.scan_folder(str(tmp_path), recursive=False) + + assert len(entries) == 1 + e = entries[0] + assert e["matched"] is True + assert e["year"] == 2018 + assert e["genre"] == "House" + assert e["fingerprint"] == "FP" + + def test_scan_folder_records_fpcalc_error(self, tmp_path): + """fpcalc fallito → entry con matched=False + error.""" + (tmp_path / "broken.mp3").write_bytes(b"\x00" * 10) + with mock.patch.object(catalog, "find_fpcalc", + return_value="/fake/fpcalc"), \ + mock.patch.object(catalog, "compute_fingerprint", + return_value={"_error": "fingerprint vuoto"}): + entries = catalog.scan_folder(str(tmp_path), recursive=False) + assert len(entries) == 1 + assert entries[0]["matched"] is False + assert "vuoto" in entries[0]["error"]