dedup: modulo audio fingerprinting via fpcalc + cache SQLite
Nuovo modulo core.dedup per rilevare brani audio duplicati usando Chromaprint (fpcalc): scan cartella (opzionalmente ricorsivo), fingerprint acustico via fpcalc -json, cache SQLite in _get_config_dir()/dedup_cache.db per non ricalcolare al re-scan, raggruppamento per fingerprint identico + ordering per bitrate DESC. move_to_trash() usa send2trash (reversibile via Finder/Explorer). - core/paths.py: find_fpcalc() con fallback dev su project bundle_bin - core/dedup.py: scan_folder, compute_fingerprint, move_to_trash - tests/test_dedup.py: 9 unit test (mock fpcalc + send2trash) - requirements.txt: send2trash>=1.8.0 - build_macos.py: download fpcalc universal binary da GitHub releases - build_windows.py: download fpcalc.exe da GitHub releases
This commit is contained in:
1 parent
e09ee1382b
commit
fb13b0f70e
6 files changed
+635
No files matched your search
@@ -64,6 +64,48 @@ def download_ytdlp():
|
|||||||
return dest
|
return dest
|
||||||
|
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# 1b. Scarica fpcalc (Chromaprint) — usato dal tab Dedup
|
||||||
|
# =========================================================================
|
||||||
|
def download_fpcalc():
|
||||||
|
"""Scarica il binario universal Chromaprint fpcalc per macOS."""
|
||||||
|
dest = BUNDLE_DIR / "fpcalc"
|
||||||
|
if dest.exists():
|
||||||
|
log(f"fpcalc gia presente: {dest}")
|
||||||
|
return dest
|
||||||
|
|
||||||
|
url = ("https://github.com/acoustid/chromaprint/releases/download/"
|
||||||
|
"v1.5.1/chromaprint-fpcalc-1.5.1-macos-universal.tar.gz")
|
||||||
|
log(f"Scarico fpcalc da {url} ...")
|
||||||
|
BUNDLE_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
tmp_tar = BUNDLE_DIR / "_fpcalc.tar.gz"
|
||||||
|
subprocess.run(["curl", "-L", "-o", str(tmp_tar), url], check=True)
|
||||||
|
# Estrae ovunque nella cartella, poi trova fpcalc e lo sposta al posto
|
||||||
|
tmp_extract = BUNDLE_DIR / "_fpcalc_extract"
|
||||||
|
if tmp_extract.exists():
|
||||||
|
shutil.rmtree(tmp_extract)
|
||||||
|
tmp_extract.mkdir()
|
||||||
|
subprocess.run(["tar", "-xzf", str(tmp_tar), "-C", str(tmp_extract)],
|
||||||
|
check=True)
|
||||||
|
# Trova fpcalc nel folder estratto
|
||||||
|
found = None
|
||||||
|
for p in tmp_extract.rglob("fpcalc"):
|
||||||
|
if p.is_file():
|
||||||
|
found = p
|
||||||
|
break
|
||||||
|
if not found:
|
||||||
|
shutil.rmtree(tmp_extract, ignore_errors=True)
|
||||||
|
tmp_tar.unlink(missing_ok=True)
|
||||||
|
raise RuntimeError("fpcalc non trovato nell'archivio")
|
||||||
|
shutil.copy2(found, dest)
|
||||||
|
dest.chmod(dest.stat().st_mode | stat.S_IEXEC)
|
||||||
|
shutil.rmtree(tmp_extract, ignore_errors=True)
|
||||||
|
tmp_tar.unlink(missing_ok=True)
|
||||||
|
log(f"fpcalc scaricato: {dest}")
|
||||||
|
return dest
|
||||||
|
|
||||||
|
|
||||||
# =========================================================================
|
# =========================================================================
|
||||||
# 2. Raccogli ffmpeg/ffprobe + dylib
|
# 2. Raccogli ffmpeg/ffprobe + dylib
|
||||||
# =========================================================================
|
# =========================================================================
|
||||||
@@ -511,6 +553,7 @@ def main():
|
|||||||
shutil.rmtree(BUNDLE_DIR)
|
shutil.rmtree(BUNDLE_DIR)
|
||||||
|
|
||||||
download_ytdlp()
|
download_ytdlp()
|
||||||
|
download_fpcalc()
|
||||||
bundle_ffmpeg()
|
bundle_ffmpeg()
|
||||||
run_pyinstaller()
|
run_pyinstaller()
|
||||||
|
|
||||||
|
|||||||
@@ -31,6 +31,8 @@ BUNDLE_DIR = ROOT / "bundle_bin"
|
|||||||
|
|
||||||
YTDLP_URL = "https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.exe"
|
YTDLP_URL = "https://github.com/yt-dlp/yt-dlp/releases/latest/download/yt-dlp.exe"
|
||||||
FFMPEG_URL = "https://github.com/BtbN/FFmpeg-Builds/releases/download/latest/ffmpeg-master-latest-win64-gpl.zip"
|
FFMPEG_URL = "https://github.com/BtbN/FFmpeg-Builds/releases/download/latest/ffmpeg-master-latest-win64-gpl.zip"
|
||||||
|
FPCALC_URL = ("https://github.com/acoustid/chromaprint/releases/download/"
|
||||||
|
"v1.5.1/chromaprint-fpcalc-1.5.1-windows-x86_64.zip")
|
||||||
|
|
||||||
|
|
||||||
def log(msg):
|
def log(msg):
|
||||||
@@ -87,6 +89,33 @@ def download_ffmpeg():
|
|||||||
sys.exit(1)
|
sys.exit(1)
|
||||||
|
|
||||||
|
|
||||||
|
# =========================================================================
|
||||||
|
# 2b. Scarica fpcalc.exe (Chromaprint) — usato dal tab Dedup
|
||||||
|
# =========================================================================
|
||||||
|
def download_fpcalc():
|
||||||
|
dest = BUNDLE_DIR / "fpcalc.exe"
|
||||||
|
if dest.exists():
|
||||||
|
log("fpcalc.exe gia presente")
|
||||||
|
return
|
||||||
|
|
||||||
|
log(f"Scarico fpcalc.exe ...")
|
||||||
|
BUNDLE_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
|
||||||
|
response = urllib.request.urlopen(FPCALC_URL)
|
||||||
|
zip_data = io.BytesIO(response.read())
|
||||||
|
with zipfile.ZipFile(zip_data) as zf:
|
||||||
|
for member in zf.namelist():
|
||||||
|
basename = Path(member).name
|
||||||
|
if basename == "fpcalc.exe":
|
||||||
|
data = zf.read(member)
|
||||||
|
dest.write_bytes(data)
|
||||||
|
log(f" Estratto: fpcalc.exe ({len(data) // 1024} KB)")
|
||||||
|
|
||||||
|
if not dest.exists():
|
||||||
|
print("ERRORE: fpcalc.exe non trovato nello zip!")
|
||||||
|
sys.exit(1)
|
||||||
|
|
||||||
|
|
||||||
# =========================================================================
|
# =========================================================================
|
||||||
# 3. PyInstaller
|
# 3. PyInstaller
|
||||||
# =========================================================================
|
# =========================================================================
|
||||||
@@ -187,6 +216,7 @@ def main():
|
|||||||
|
|
||||||
download_ytdlp()
|
download_ytdlp()
|
||||||
download_ffmpeg()
|
download_ffmpeg()
|
||||||
|
download_fpcalc()
|
||||||
run_pyinstaller()
|
run_pyinstaller()
|
||||||
|
|
||||||
print("\n" + "=" * 50)
|
print("\n" + "=" * 50)
|
||||||
|
|||||||
+346
@@ -0,0 +1,346 @@
|
|||||||
|
"""Deduplicator audio via Chromaprint fingerprinting + SQLite cache.
|
||||||
|
|
||||||
|
Pipeline:
|
||||||
|
1. Scansiona la cartella (opzionalmente ricorsivo) filtrando per
|
||||||
|
estensioni audio (AUDIO_EXTENSIONS di core.upgrader).
|
||||||
|
2. Per ogni file calcola il fingerprint Chromaprint (`fpcalc -json`).
|
||||||
|
Il valore viene messo in cache SQLite: al re-scan, se
|
||||||
|
(size, mtime) coincide col record, riusiamo il fingerprint senza
|
||||||
|
rilanciare fpcalc.
|
||||||
|
3. Raggruppa i file per fingerprint identico (>= 2 file). Per ogni
|
||||||
|
gruppo, i file vengono ordinati per bitrate DESC (tie-break: size
|
||||||
|
DESC): il primo e' quello "da tenere", gli altri i duplicati.
|
||||||
|
4. `move_to_trash` invia i path selezionati al cestino di sistema
|
||||||
|
tramite send2trash (reversibile via Finder/Explorer).
|
||||||
|
|
||||||
|
Progress callback firma:
|
||||||
|
(processed: int, total: int, filename: str, status: str)
|
||||||
|
Status validi: 'scanning' | 'computing' | 'cached' | 'error' | 'stopped'
|
||||||
|
| 'completed'.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import sqlite3
|
||||||
|
import subprocess
|
||||||
|
import threading
|
||||||
|
from pathlib import Path
|
||||||
|
from typing import Callable, Optional
|
||||||
|
|
||||||
|
from core.paths import find_fpcalc, subprocess_flags
|
||||||
|
from core.upgrader import AUDIO_EXTENSIONS, get_bitrate
|
||||||
|
|
||||||
|
|
||||||
|
# Timeout massimo per una singola invocazione fpcalc.
|
||||||
|
_FPCALC_TIMEOUT_SEC = 30
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Stop / interrupt
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
_stop_event = threading.Event()
|
||||||
|
|
||||||
|
|
||||||
|
def request_stop() -> None:
|
||||||
|
"""Segnala al worker di interrompere la scansione al prossimo file."""
|
||||||
|
_stop_event.set()
|
||||||
|
|
||||||
|
|
||||||
|
def reset_stop() -> None:
|
||||||
|
"""Azzera il flag di stop prima di iniziare una nuova scansione."""
|
||||||
|
_stop_event.clear()
|
||||||
|
|
||||||
|
|
||||||
|
def is_stopped() -> bool:
|
||||||
|
return _stop_event.is_set()
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Cache SQLite
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
def _cache_db_path() -> Path:
|
||||||
|
"""Path del DB di cache dei fingerprint.
|
||||||
|
|
||||||
|
Riusa `_get_config_dir` di core.config cosi' finisce nella stessa
|
||||||
|
cartella di config.json (~/Library/Application Support/MusicTools/
|
||||||
|
su macOS, %APPDATA%/MusicTools/ su Windows, project root in dev).
|
||||||
|
"""
|
||||||
|
from core.config import _get_config_dir
|
||||||
|
return _get_config_dir() / "dedup_cache.db"
|
||||||
|
|
||||||
|
|
||||||
|
def _init_db(conn: sqlite3.Connection) -> None:
|
||||||
|
"""Crea (idempotente) lo schema della cache."""
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
CREATE TABLE IF NOT EXISTS files (
|
||||||
|
path TEXT PRIMARY KEY,
|
||||||
|
size INTEGER NOT NULL,
|
||||||
|
mtime REAL NOT NULL,
|
||||||
|
duration REAL,
|
||||||
|
fingerprint TEXT,
|
||||||
|
bitrate INTEGER
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def _open_cache(db_path: Optional[Path] = None) -> sqlite3.Connection:
|
||||||
|
"""Apre (creando se serve) la connessione alla cache."""
|
||||||
|
p = db_path or _cache_db_path()
|
||||||
|
p.parent.mkdir(parents=True, exist_ok=True)
|
||||||
|
conn = sqlite3.connect(str(p))
|
||||||
|
_init_db(conn)
|
||||||
|
return conn
|
||||||
|
|
||||||
|
|
||||||
|
def _cache_get(conn: sqlite3.Connection, path: str,
|
||||||
|
size: int, mtime: float) -> Optional[dict]:
|
||||||
|
"""Ritorna il record se (size, mtime) invariato, altrimenti None."""
|
||||||
|
cur = conn.execute(
|
||||||
|
"SELECT size, mtime, duration, fingerprint, bitrate FROM files WHERE path = ?",
|
||||||
|
(path,),
|
||||||
|
)
|
||||||
|
row = cur.fetchone()
|
||||||
|
if not row:
|
||||||
|
return None
|
||||||
|
csize, cmtime, dur, fp, br = row
|
||||||
|
# Tolleranza minima sul mtime (float precision su alcuni FS)
|
||||||
|
if csize != size or abs(float(cmtime) - float(mtime)) > 0.001:
|
||||||
|
return None
|
||||||
|
if not fp:
|
||||||
|
return None
|
||||||
|
return {
|
||||||
|
"size": int(csize),
|
||||||
|
"mtime": float(cmtime),
|
||||||
|
"duration": float(dur) if dur is not None else 0.0,
|
||||||
|
"fingerprint": str(fp),
|
||||||
|
"bitrate": int(br) if br is not None else 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _cache_put(conn: sqlite3.Connection, path: str, size: int, mtime: float,
|
||||||
|
duration: float, fingerprint: str, bitrate: int) -> None:
|
||||||
|
"""Upsert (SQLite ha ON CONFLICT REPLACE via INSERT OR REPLACE)."""
|
||||||
|
conn.execute(
|
||||||
|
"INSERT OR REPLACE INTO files (path, size, mtime, duration, fingerprint, bitrate)"
|
||||||
|
" VALUES (?, ?, ?, ?, ?, ?)",
|
||||||
|
(path, int(size), float(mtime), float(duration or 0),
|
||||||
|
str(fingerprint or ""), int(bitrate or 0)),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# fpcalc
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
def compute_fingerprint(fpcalc: str, path: str) -> Optional[dict]:
|
||||||
|
"""Chiama `fpcalc -json <file>` e ritorna {duration, fingerprint}.
|
||||||
|
|
||||||
|
Ritorna None su qualsiasi errore (fpcalc mancante, file corrotto,
|
||||||
|
timeout, JSON malformato).
|
||||||
|
"""
|
||||||
|
if not fpcalc:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
proc = subprocess.run(
|
||||||
|
[fpcalc, "-json", str(path)],
|
||||||
|
capture_output=True,
|
||||||
|
text=True,
|
||||||
|
timeout=_FPCALC_TIMEOUT_SEC,
|
||||||
|
**subprocess_flags(),
|
||||||
|
)
|
||||||
|
except subprocess.TimeoutExpired:
|
||||||
|
return None
|
||||||
|
except (OSError, ValueError):
|
||||||
|
return None
|
||||||
|
if proc.returncode != 0:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
data = json.loads(proc.stdout or "{}")
|
||||||
|
except (json.JSONDecodeError, ValueError):
|
||||||
|
return None
|
||||||
|
fp = data.get("fingerprint")
|
||||||
|
if not fp:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
dur = float(data.get("duration") or 0)
|
||||||
|
except (TypeError, ValueError):
|
||||||
|
dur = 0.0
|
||||||
|
return {"duration": dur, "fingerprint": str(fp)}
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Scan
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
def _iter_audio_files(directory: str, recursive: bool) -> list[Path]:
|
||||||
|
"""Elenca tutti i file audio (estensione case-insensitive)."""
|
||||||
|
base = Path(directory)
|
||||||
|
if not base.exists() or not base.is_dir():
|
||||||
|
return []
|
||||||
|
files: list[Path] = []
|
||||||
|
if recursive:
|
||||||
|
for f in base.rglob("*"):
|
||||||
|
if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS:
|
||||||
|
files.append(f)
|
||||||
|
else:
|
||||||
|
for f in base.iterdir():
|
||||||
|
if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS:
|
||||||
|
files.append(f)
|
||||||
|
files.sort()
|
||||||
|
return files
|
||||||
|
|
||||||
|
|
||||||
|
def scan_folder(
|
||||||
|
directory: str,
|
||||||
|
recursive: bool = True,
|
||||||
|
progress_callback: Optional[Callable] = None,
|
||||||
|
) -> list[list[dict]]:
|
||||||
|
"""Ritorna la lista di gruppi di file duplicati (>= 2 file).
|
||||||
|
|
||||||
|
Ogni file nel gruppo e' un dict:
|
||||||
|
{path, size, bitrate, duration, fingerprint}
|
||||||
|
|
||||||
|
Gruppi ordinati per size del file piu' grande DESC (i gruppi che
|
||||||
|
occupano piu' spazio vengono prima). All'interno di ogni gruppo:
|
||||||
|
bitrate DESC, poi size DESC (il primo e' quello "da tenere").
|
||||||
|
|
||||||
|
Il fingerprint viene calcolato via `fpcalc -json` e messo in cache
|
||||||
|
SQLite. Al re-scan, se (size, mtime) invariati, non si rilancia fpcalc.
|
||||||
|
"""
|
||||||
|
reset_stop()
|
||||||
|
files = _iter_audio_files(directory, recursive)
|
||||||
|
total = len(files)
|
||||||
|
if total == 0:
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(0, 0, "", "completed")
|
||||||
|
return []
|
||||||
|
|
||||||
|
fpcalc = find_fpcalc()
|
||||||
|
if not fpcalc:
|
||||||
|
# Senza fpcalc non possiamo fare nulla. Segnaliamo errore su ogni
|
||||||
|
# file e ritorniamo lista vuota.
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(0, total, "", "error")
|
||||||
|
return []
|
||||||
|
|
||||||
|
conn = _open_cache()
|
||||||
|
try:
|
||||||
|
# {fingerprint: [entry, ...]}
|
||||||
|
by_fp: dict[str, list[dict]] = {}
|
||||||
|
|
||||||
|
for i, fp_path in enumerate(files, start=1):
|
||||||
|
if is_stopped():
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(i - 1, total, "", "stopped")
|
||||||
|
return []
|
||||||
|
try:
|
||||||
|
st = fp_path.stat()
|
||||||
|
size = st.st_size
|
||||||
|
mtime = st.st_mtime
|
||||||
|
except OSError:
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(i, total, fp_path.name, "error")
|
||||||
|
continue
|
||||||
|
|
||||||
|
path_str = str(fp_path)
|
||||||
|
cached = _cache_get(conn, path_str, size, mtime)
|
||||||
|
if cached:
|
||||||
|
fp_hash = cached["fingerprint"]
|
||||||
|
duration = cached["duration"]
|
||||||
|
bitrate = cached["bitrate"] or get_bitrate(fp_path)
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(i, total, fp_path.name, "cached")
|
||||||
|
else:
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(i, total, fp_path.name, "computing")
|
||||||
|
res = compute_fingerprint(fpcalc, path_str)
|
||||||
|
if not res:
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(i, total, fp_path.name, "error")
|
||||||
|
continue
|
||||||
|
fp_hash = res["fingerprint"]
|
||||||
|
duration = res["duration"]
|
||||||
|
try:
|
||||||
|
bitrate = get_bitrate(fp_path)
|
||||||
|
except Exception:
|
||||||
|
bitrate = 0
|
||||||
|
_cache_put(conn, path_str, size, mtime, duration, fp_hash, bitrate)
|
||||||
|
|
||||||
|
entry = {
|
||||||
|
"path": path_str,
|
||||||
|
"size": int(size),
|
||||||
|
"bitrate": int(bitrate or 0),
|
||||||
|
"duration": float(duration or 0),
|
||||||
|
"fingerprint": fp_hash,
|
||||||
|
}
|
||||||
|
by_fp.setdefault(fp_hash, []).append(entry)
|
||||||
|
finally:
|
||||||
|
try:
|
||||||
|
conn.close()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
# Filtra: solo gruppi con >= 2 file
|
||||||
|
groups = [g for g in by_fp.values() if len(g) >= 2]
|
||||||
|
|
||||||
|
# Sort dei file dentro il gruppo: bitrate DESC, size DESC.
|
||||||
|
# Sort dei gruppi: size del file piu' grande DESC (usa max del gruppo).
|
||||||
|
for g in groups:
|
||||||
|
g.sort(key=lambda e: (-int(e.get("bitrate") or 0),
|
||||||
|
-int(e.get("size") or 0)))
|
||||||
|
groups.sort(key=lambda g: -max(int(e.get("size") or 0) for e in g))
|
||||||
|
|
||||||
|
if progress_callback:
|
||||||
|
progress_callback(total, total, "", "completed")
|
||||||
|
|
||||||
|
return groups
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Trash
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
def move_to_trash(paths: list[str]) -> dict:
|
||||||
|
"""Sposta i file in cestino tramite send2trash.
|
||||||
|
|
||||||
|
Ritorna {moved: [...], failed: [{path, error}, ...]}. Non solleva
|
||||||
|
mai eccezioni: gli errori per file singolo finiscono in `failed`.
|
||||||
|
Aggiorna la cache SQLite rimuovendo i record dei file spostati (per
|
||||||
|
quelli riusciti), cosi' un re-scan non li propone piu'.
|
||||||
|
"""
|
||||||
|
# Import interno per rendere il modulo importabile anche se
|
||||||
|
# send2trash non e' installato (i test possono mockarlo).
|
||||||
|
try:
|
||||||
|
from send2trash import send2trash
|
||||||
|
except Exception as e: # pragma: no cover — solo se pacchetto mancante
|
||||||
|
return {
|
||||||
|
"moved": [],
|
||||||
|
"failed": [{"path": p, "error": f"send2trash non disponibile: {e}"}
|
||||||
|
for p in (paths or [])],
|
||||||
|
}
|
||||||
|
|
||||||
|
moved: list[str] = []
|
||||||
|
failed: list[dict] = []
|
||||||
|
for p in (paths or []):
|
||||||
|
try:
|
||||||
|
send2trash(p)
|
||||||
|
moved.append(p)
|
||||||
|
except Exception as e:
|
||||||
|
failed.append({"path": p, "error": str(e)})
|
||||||
|
|
||||||
|
# Cache cleanup best-effort (non fatale se fallisce)
|
||||||
|
if moved:
|
||||||
|
try:
|
||||||
|
conn = _open_cache()
|
||||||
|
try:
|
||||||
|
for p in moved:
|
||||||
|
conn.execute("DELETE FROM files WHERE path = ?", (p,))
|
||||||
|
conn.commit()
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return {"moved": moved, "failed": failed}
|
||||||
@@ -145,3 +145,28 @@ def find_ffmpeg() -> Optional[str]:
|
|||||||
def find_ffprobe() -> Optional[str]:
|
def find_ffprobe() -> Optional[str]:
|
||||||
"""Ritorna il path completo di ffprobe."""
|
"""Ritorna il path completo di ffprobe."""
|
||||||
return _find_binary("ffprobe")
|
return _find_binary("ffprobe")
|
||||||
|
|
||||||
|
|
||||||
|
def find_fpcalc() -> Optional[str]:
|
||||||
|
"""Ritorna il path completo di fpcalc (Chromaprint), o None.
|
||||||
|
|
||||||
|
Cerca nelle stesse directory di find_ffmpeg / find_ytdlp: bundle
|
||||||
|
PyInstaller prima, poi percorsi noti (Homebrew su macOS, LOCALAPPDATA
|
||||||
|
su Windows), infine PATH generico. In dev mode aggiunge anche
|
||||||
|
`<project_root>/bundle_bin/` così l'app funziona con `python main.py`
|
||||||
|
dopo aver scaricato fpcalc via build script.
|
||||||
|
|
||||||
|
`fpcalc` viene usato dal modulo core.dedup per calcolare fingerprint
|
||||||
|
audio (identificazione di duplicati).
|
||||||
|
"""
|
||||||
|
found = _find_binary("fpcalc")
|
||||||
|
if found:
|
||||||
|
return found
|
||||||
|
# Fallback dev: bundle_bin del progetto
|
||||||
|
if not _is_frozen():
|
||||||
|
exe_name = _exe("fpcalc")
|
||||||
|
project_bundle = Path(__file__).resolve().parent.parent / "bundle_bin"
|
||||||
|
cand = project_bundle / exe_name
|
||||||
|
if cand.exists():
|
||||||
|
return str(cand)
|
||||||
|
return None
|
||||||
@@ -11,6 +11,7 @@ pythonnet==3.0.5 ; sys_platform == "win32"
|
|||||||
clr-loader==0.2.7.post0 ; sys_platform == "win32"
|
clr-loader==0.2.7.post0 ; sys_platform == "win32"
|
||||||
curl_cffi>=0.9.0
|
curl_cffi>=0.9.0
|
||||||
beautifulsoup4>=4.12.0
|
beautifulsoup4>=4.12.0
|
||||||
|
send2trash>=1.8.0
|
||||||
|
|
||||||
# --- dev only ---
|
# --- dev only ---
|
||||||
pytest>=8.0.0
|
pytest>=8.0.0
|
||||||
|
|||||||
@@ -0,0 +1,190 @@
|
|||||||
|
"""Test per core.dedup — audio fingerprinting via fpcalc + cache SQLite.
|
||||||
|
|
||||||
|
Tutti gli unit test usano mock per fpcalc / send2trash: nessuna
|
||||||
|
integrazione reale, nessun file audio necessario.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import subprocess
|
||||||
|
from pathlib import Path
|
||||||
|
from unittest import mock
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from core import dedup
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Fixture: dedup con cache DB isolato in tmp_path
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
@pytest.fixture
|
||||||
|
def patched_cache(tmp_path, monkeypatch):
|
||||||
|
"""Isola la cache SQLite in tmp_path per non toccare il config dir."""
|
||||||
|
db = tmp_path / "dedup_cache_test.db"
|
||||||
|
monkeypatch.setattr(dedup, "_cache_db_path", lambda: db)
|
||||||
|
return db
|
||||||
|
|
||||||
|
|
||||||
|
def _make_fake_audio(tmp_path: Path, name: str, size: int = 1024) -> Path:
|
||||||
|
"""Crea un file 'audio' fake (byte casuali con estensione .mp3)."""
|
||||||
|
f = tmp_path / name
|
||||||
|
f.write_bytes(b"\x00" * size)
|
||||||
|
return f
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# scan_folder
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
class TestScanFolder:
|
||||||
|
def test_no_audio_files(self, tmp_path, patched_cache):
|
||||||
|
"""Cartella senza audio -> gruppi vuoti, nessuna eccezione."""
|
||||||
|
# Solo un file .txt (non audio)
|
||||||
|
(tmp_path / "readme.txt").write_text("hello")
|
||||||
|
|
||||||
|
# Anche senza fpcalc disponibile, con 0 audio file ritorna [].
|
||||||
|
with mock.patch.object(dedup, "find_fpcalc", return_value="/fake/fpcalc"):
|
||||||
|
groups = dedup.scan_folder(str(tmp_path), recursive=False)
|
||||||
|
|
||||||
|
assert groups == []
|
||||||
|
|
||||||
|
def test_uses_cache_on_second_scan(self, tmp_path, patched_cache):
|
||||||
|
"""Prima scan chiama fpcalc; seconda scan riusa la cache."""
|
||||||
|
_make_fake_audio(tmp_path, "song.mp3", size=2048)
|
||||||
|
|
||||||
|
calls: list[str] = []
|
||||||
|
|
||||||
|
def fake_compute(fpcalc, path):
|
||||||
|
calls.append(path)
|
||||||
|
return {"duration": 180.5, "fingerprint": "FP-A"}
|
||||||
|
|
||||||
|
with mock.patch.object(dedup, "find_fpcalc", return_value="/fake/fpcalc"), \
|
||||||
|
mock.patch.object(dedup, "compute_fingerprint", side_effect=fake_compute), \
|
||||||
|
mock.patch.object(dedup, "get_bitrate", return_value=320):
|
||||||
|
# Prima invocazione: fpcalc DEVE essere chiamato
|
||||||
|
groups1 = dedup.scan_folder(str(tmp_path), recursive=False)
|
||||||
|
first_calls = len(calls)
|
||||||
|
|
||||||
|
# Seconda invocazione (stesso file, stesso mtime/size):
|
||||||
|
# cache HIT, fpcalc NON viene richiamato
|
||||||
|
groups2 = dedup.scan_folder(str(tmp_path), recursive=False)
|
||||||
|
second_calls = len(calls)
|
||||||
|
|
||||||
|
assert first_calls == 1, "prima scan deve chiamare fpcalc una volta"
|
||||||
|
assert second_calls == 1, "seconda scan deve riusare la cache"
|
||||||
|
# Con un solo file, nessun gruppo di duplicati
|
||||||
|
assert groups1 == []
|
||||||
|
assert groups2 == []
|
||||||
|
|
||||||
|
def test_groups_duplicates(self, tmp_path, patched_cache):
|
||||||
|
"""3 file con lo stesso fingerprint -> 1 gruppo di 3, ordinato per bitrate DESC."""
|
||||||
|
_make_fake_audio(tmp_path, "a.mp3", size=1000)
|
||||||
|
_make_fake_audio(tmp_path, "b.mp3", size=3000) # size maggiore
|
||||||
|
_make_fake_audio(tmp_path, "c.mp3", size=2000)
|
||||||
|
|
||||||
|
# Tutti stesso fingerprint. Bitrate differente per verificare
|
||||||
|
# l'ordinamento: b=320 (top), a=192, c=128.
|
||||||
|
bitrate_by_name = {"a.mp3": 192, "b.mp3": 320, "c.mp3": 128}
|
||||||
|
|
||||||
|
with mock.patch.object(dedup, "find_fpcalc", return_value="/fake/fpcalc"), \
|
||||||
|
mock.patch.object(dedup, "compute_fingerprint",
|
||||||
|
return_value={"duration": 200, "fingerprint": "SAME-FP"}), \
|
||||||
|
mock.patch.object(dedup, "get_bitrate",
|
||||||
|
side_effect=lambda p: bitrate_by_name[Path(p).name]):
|
||||||
|
groups = dedup.scan_folder(str(tmp_path), recursive=False)
|
||||||
|
|
||||||
|
assert len(groups) == 1, "esattamente un gruppo di duplicati"
|
||||||
|
g = groups[0]
|
||||||
|
assert len(g) == 3, "tre file nel gruppo"
|
||||||
|
# Ordine: bitrate DESC -> b (320), a (192), c (128)
|
||||||
|
assert [Path(e["path"]).name for e in g] == ["b.mp3", "a.mp3", "c.mp3"]
|
||||||
|
# Ogni entry ha i campi attesi
|
||||||
|
for e in g:
|
||||||
|
assert set(e.keys()) >= {"path", "size", "bitrate", "duration", "fingerprint"}
|
||||||
|
assert e["fingerprint"] == "SAME-FP"
|
||||||
|
|
||||||
|
def test_ignores_singletons(self, tmp_path, patched_cache):
|
||||||
|
"""File con fingerprint unico non compaiono nei gruppi."""
|
||||||
|
_make_fake_audio(tmp_path, "dup1.mp3")
|
||||||
|
_make_fake_audio(tmp_path, "dup2.mp3")
|
||||||
|
_make_fake_audio(tmp_path, "unique.mp3")
|
||||||
|
|
||||||
|
fp_by_name = {"dup1.mp3": "FP-X", "dup2.mp3": "FP-X", "unique.mp3": "FP-Y"}
|
||||||
|
|
||||||
|
def fake_compute(fpcalc, path):
|
||||||
|
return {"duration": 100, "fingerprint": fp_by_name[Path(path).name]}
|
||||||
|
|
||||||
|
with mock.patch.object(dedup, "find_fpcalc", return_value="/fake/fpcalc"), \
|
||||||
|
mock.patch.object(dedup, "compute_fingerprint", side_effect=fake_compute), \
|
||||||
|
mock.patch.object(dedup, "get_bitrate", return_value=256):
|
||||||
|
groups = dedup.scan_folder(str(tmp_path), recursive=False)
|
||||||
|
|
||||||
|
# Solo il gruppo con dup1/dup2
|
||||||
|
assert len(groups) == 1
|
||||||
|
names = {Path(e["path"]).name for e in groups[0]}
|
||||||
|
assert names == {"dup1.mp3", "dup2.mp3"}
|
||||||
|
# unique.mp3 non appare in nessun gruppo
|
||||||
|
for g in groups:
|
||||||
|
for e in g:
|
||||||
|
assert Path(e["path"]).name != "unique.mp3"
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# move_to_trash
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
class TestMoveToTrash:
|
||||||
|
def test_returns_moved_and_failed_summary(self, tmp_path, patched_cache):
|
||||||
|
"""Verifica che il summary contenga moved/failed correttamente."""
|
||||||
|
# 2 path OK + 1 path che alza eccezione
|
||||||
|
ok1 = str(tmp_path / "ok1.mp3")
|
||||||
|
ok2 = str(tmp_path / "ok2.mp3")
|
||||||
|
bad = str(tmp_path / "bad.mp3")
|
||||||
|
|
||||||
|
def fake_send(p):
|
||||||
|
if p == bad:
|
||||||
|
raise OSError("simulated failure")
|
||||||
|
# ok: no-op
|
||||||
|
|
||||||
|
# Il modulo importa send2trash *dentro* la funzione, quindi
|
||||||
|
# dobbiamo patchare il modulo importato.
|
||||||
|
with mock.patch("send2trash.send2trash", side_effect=fake_send):
|
||||||
|
res = dedup.move_to_trash([ok1, bad, ok2])
|
||||||
|
|
||||||
|
assert set(res["moved"]) == {ok1, ok2}
|
||||||
|
assert len(res["failed"]) == 1
|
||||||
|
assert res["failed"][0]["path"] == bad
|
||||||
|
assert "simulated failure" in res["failed"][0]["error"]
|
||||||
|
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# compute_fingerprint
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
class TestComputeFingerprint:
|
||||||
|
def test_timeout_returns_none(self):
|
||||||
|
"""Timeout di fpcalc -> None (non alza eccezione)."""
|
||||||
|
with mock.patch("core.dedup.subprocess.run",
|
||||||
|
side_effect=subprocess.TimeoutExpired(cmd="fpcalc", timeout=30)):
|
||||||
|
result = dedup.compute_fingerprint("/fake/fpcalc", "/some/file.mp3")
|
||||||
|
assert result is None
|
||||||
|
|
||||||
|
def test_success_returns_dict(self):
|
||||||
|
"""Output JSON valido -> {duration, fingerprint}."""
|
||||||
|
fake_proc = mock.Mock()
|
||||||
|
fake_proc.returncode = 0
|
||||||
|
fake_proc.stdout = '{"duration": 123.4, "fingerprint": "ABCDEF"}'
|
||||||
|
with mock.patch("core.dedup.subprocess.run", return_value=fake_proc):
|
||||||
|
result = dedup.compute_fingerprint("/fake/fpcalc", "/some/file.mp3")
|
||||||
|
assert result == {"duration": 123.4, "fingerprint": "ABCDEF"}
|
||||||
|
|
||||||
|
def test_bad_json_returns_none(self):
|
||||||
|
fake_proc = mock.Mock()
|
||||||
|
fake_proc.returncode = 0
|
||||||
|
fake_proc.stdout = "not json at all"
|
||||||
|
with mock.patch("core.dedup.subprocess.run", return_value=fake_proc):
|
||||||
|
result = dedup.compute_fingerprint("/fake/fpcalc", "/some/file.mp3")
|
||||||
|
assert result is None
|
||||||
|
|
||||||
|
def test_missing_fpcalc_returns_none(self):
|
||||||
|
# Nessuna chiamata subprocess se fpcalc e' vuoto
|
||||||
|
result = dedup.compute_fingerprint("", "/some/file.mp3")
|
||||||
|
assert result is None
|
||||||
Reference in new issue
Block a user