diff --git a/.gitignore b/.gitignore index fadc4fb..afa4616 100644 --- a/.gitignore +++ b/.gitignore @@ -49,3 +49,7 @@ server/node_modules/ server/.wrangler/ server/.dev.vars server/dist/ + +# Dedup local cache (SQLite fingerprint cache, per macchina dell'utente) +dedup_cache.db +dedup_cache.db-journal diff --git a/api/bridge.py b/api/bridge.py index 20e1660..56ef704 100644 --- a/api/bridge.py +++ b/api/bridge.py @@ -1841,24 +1841,28 @@ class Api: directory = (payload.get("directory") or "").strip() recursive = bool(payload.get("recursive", True)) + method = (payload.get("method") or "fingerprint").strip() + if method not in ("fingerprint", "filename"): + method = "fingerprint" if not directory: return {"ok": False, "error": "Cartella non impostata"} if not os.path.isdir(directory): return {"ok": False, "error": "Cartella non trovata"} - # Persist last folder/recursive + # Persist last folder/recursive/method try: cfg = load_config() cfg["dedup_last_folder"] = directory cfg["dedup_recursive"] = recursive + cfg["dedup_method"] = method save_config(cfg) except Exception: pass self._dedup_thread = threading.Thread( target=self._dedup_worker, - args=(directory, recursive), + args=(directory, recursive, method), daemon=True, ) self._dedup_thread.start() @@ -1890,16 +1894,18 @@ class Api: "failed": result.get("failed", []), "moved_count": moved_n, "failed_count": failed_n} - def _dedup_worker(self, directory: str, recursive: bool) -> None: + def _dedup_worker(self, directory: str, recursive: bool, + method: str = "fingerprint") -> None: """Esegue la scansione in background e emette progress/done.""" view = "dedup" dedup.reset_stop() - self._log(view, f"[INFO] Scansione: {directory} (recursive={recursive})") + method_label = "audio fingerprint" if method == "fingerprint" else "nome file" + self._log(view, f"[INFO] Scansione: {directory} (metodo: {method_label}, recursive={recursive})") _last = [0.0] _THROTTLE = 0.05 - def progress_cb(idx: int, total: int, filename: str, status: str) -> None: + def progress_cb(idx: int, total: int, filename: str, status: str, err_msg: str = "") -> None: # Throttle solo eventi 'computing'/'cached' (che possono essere migliaia) if status in ("computing", "cached"): now = time.monotonic() @@ -1910,6 +1916,8 @@ class Api: "idx": idx, "total": total, "filename": filename, "status": status, } + if err_msg: + payload_evt["error_msg"] = err_msg if total > 0: payload_evt["overall"] = min(idx / total, 1.0) if status == "completed": @@ -1918,12 +1926,22 @@ class Api: elif status == "stopped": self._log(view, "[INFO] Scansione interrotta.") elif status == "error" and filename: - self._log(view, f"[ERRORE] {filename}: errore lettura") + detail = f": {err_msg}" if err_msg else "" + self._log(view, f"[ERRORE] {filename}{detail}") self._emit("dedup:progress", payload_evt) + def group_cb(group: dict) -> None: + """Emit streaming: appena un gruppo raggiunge/aggiorna >=2 file.""" + try: + self._emit("dedup:group", group) + except Exception: + pass + try: groups = dedup.scan_folder(directory, recursive=recursive, - progress_callback=progress_cb) + progress_callback=progress_cb, + method=method, + group_callback=group_cb) except Exception as e: self._log(view, f"[ERRORE] {e}") self._emit("dedup:done", {"ok": False, "error": str(e), diff --git a/core/config.py b/core/config.py index 35f4af7..e74b559 100644 --- a/core/config.py +++ b/core/config.py @@ -81,6 +81,7 @@ DEFAULTS = { # ---- Dedup (audio duplicati via Chromaprint) ---- "dedup_last_folder": "", "dedup_recursive": True, + "dedup_method": "fingerprint", # "fingerprint" | "filename" # ---- Licenza ---- "license_key": "", # chiave fornita all'utente via email "license_email": "", # email associata all'acquisto diff --git a/core/dedup.py b/core/dedup.py index fbf77c5..93b363e 100644 --- a/core/dedup.py +++ b/core/dedup.py @@ -136,35 +136,36 @@ def _cache_put(conn: sqlite3.Connection, path: str, size: int, mtime: float, # ------------------------------------------------------------------ # fpcalc # ------------------------------------------------------------------ -def compute_fingerprint(fpcalc: str, path: str) -> Optional[dict]: - """Chiama `fpcalc -json ` e ritorna {duration, fingerprint}. - - Ritorna None su qualsiasi errore (fpcalc mancante, file corrotto, - timeout, JSON malformato). - """ - if not fpcalc: - return None +def _run_fpcalc(fpcalc: str, path: str, length: Optional[int] = None) -> dict: + """Esegue fpcalc una volta. Ritorna {duration, fingerprint} su successo + o {_error: str} su fallimento.""" + cmd = [fpcalc, "-json"] + if length is not None: + cmd += ["-length", str(length)] + cmd.append(str(path)) try: proc = subprocess.run( - [fpcalc, "-json", str(path)], + cmd, capture_output=True, text=True, timeout=_FPCALC_TIMEOUT_SEC, **subprocess_flags(), ) except subprocess.TimeoutExpired: - return None - except (OSError, ValueError): - return None + return {"_error": f"timeout {_FPCALC_TIMEOUT_SEC}s"} + except (OSError, ValueError) as e: + return {"_error": f"subprocess: {e}"} if proc.returncode != 0: - return None + err = (proc.stderr or proc.stdout or "").strip().splitlines() + msg = err[-1] if err else f"exit {proc.returncode}" + return {"_error": msg[:200]} try: data = json.loads(proc.stdout or "{}") - except (json.JSONDecodeError, ValueError): - return None + except (json.JSONDecodeError, ValueError) as e: + return {"_error": f"JSON malformato: {e}"} fp = data.get("fingerprint") if not fp: - return None + return {"_error": "fingerprint vuoto (audio troppo corto?)"} try: dur = float(data.get("duration") or 0) except (TypeError, ValueError): @@ -172,6 +173,64 @@ def compute_fingerprint(fpcalc: str, path: str) -> Optional[dict]: return {"duration": dur, "fingerprint": str(fp)} +def compute_fingerprint(fpcalc: str, path: str) -> Optional[dict]: + """Chiama fpcalc e ritorna {duration, fingerprint} o {_error}. + + Se il primo tentativo (full length) fallisce con "Invalid data" o simili + (frame audio corrotti che libav rifiuta), riprova con `-length 30`. + Molti file danneggiati hanno i frame corrotti nella parte finale e + limitando la scansione ai primi 30s si riesce a estrarre comunque + un fingerprint affidabile (30s bastano per l'unicità Chromaprint). + """ + if not fpcalc: + return {"_error": "fpcalc non trovato nel bundle"} + + res = _run_fpcalc(fpcalc, path) + if "fingerprint" in res: + return res + + err_msg = res.get("_error", "").lower() + + # Retry 1: frame audio corrotti → riduci finestra a 30s + corrupt_signals = ("invalid data", "decoding audio frame", + "error while decoding", "invalid frame") + if any(sig in err_msg for sig in corrupt_signals): + res2 = _run_fpcalc(fpcalc, path, length=30) + if "fingerprint" in res2: + res2["_partial"] = True # 30s soltanto + return res2 + + # Retry 2: fingerprint vuoto → prova con finestra piu' lunga (60s) + # nel caso l'intro sia silenzio/muto (chromaprint richiede audio "reale") + if "vuoto" in err_msg or "empty" in err_msg: + res2 = _run_fpcalc(fpcalc, path, length=60) + if "fingerprint" in res2: + res2["_partial"] = True + return res2 + # Ancora vuoto → prova algoritmo differente (chromaprint algo 1) + # tramite subprocess diretto perche' _run_fpcalc non lo supporta + try: + proc = subprocess.run( + [fpcalc, "-json", "-length", "60", "-algorithm", "1", str(path)], + capture_output=True, text=True, + timeout=_FPCALC_TIMEOUT_SEC, + **subprocess_flags(), + ) + if proc.returncode == 0: + data = json.loads(proc.stdout or "{}") + fp = data.get("fingerprint") + if fp: + return { + "duration": float(data.get("duration") or 0), + "fingerprint": str(fp), + "_partial": True, + } + except Exception: + pass + + return res + + # ------------------------------------------------------------------ # Scan # ------------------------------------------------------------------ @@ -193,10 +252,97 @@ def _iter_audio_files(directory: str, recursive: bool) -> list[Path]: return files +def _scan_by_filename(files: list, progress_callback: Optional[Callable], + group_callback: Optional[Callable] = None, + similarity_threshold: float = 0.8) -> list[list[dict]]: + """Raggruppa file per similarità nome (Jaccard sui token normalizzati), + algoritmo INCREMENTALE: per ogni nuovo file cerca match tra i gruppi già + formati (lookup O(K) dove K = numero gruppi). Emette streaming via + `group_callback` appena un gruppo raggiunge ≥ 2 file. + """ + from core.upgrader import _normalize_stem # riuso + + def _pc(idx, total_n, name, status, err=""): + if not progress_callback: + return + try: + progress_callback(idx, total_n, name, status, err) + except TypeError: + progress_callback(idx, total_n, name, status) + + def _gc(group_id: str, entries: list) -> None: + if group_callback: + try: + group_callback({"id": group_id, "entries": list(entries)}) + except Exception: + pass + + total = len(files) + # Ogni voce: {"id": str, "key_tokens": frozenset, "entries": [dict]} + groups: list = [] + + for i, fp_path in enumerate(files, start=1): + if is_stopped(): + _pc(i - 1, total, "", "stopped") + break + try: + size = fp_path.stat().st_size + except OSError as e: + _pc(i, total, fp_path.name, "error", f"stat: {e}") + continue + tokens = frozenset(_normalize_stem(fp_path.stem)) + if not tokens: + _pc(i, total, fp_path.name, "error", "nome senza token utili") + continue + try: + bitrate = get_bitrate(fp_path) + except Exception: + bitrate = 0 + entry = { + "path": str(fp_path), "size": size, "bitrate": bitrate, + "duration": 0, "fingerprint": "", + } + + # Cerca match nei gruppi già formati (lineare sui gruppi, non sui file) + matched = None + for g in groups: + common = len(tokens & g["key_tokens"]) + if common == 0: + continue + union = len(tokens | g["key_tokens"]) + if union > 0 and (common / union) >= similarity_threshold: + matched = g + break + + if matched is not None: + was_solo = len(matched["entries"]) == 1 + matched["entries"].append(entry) + matched["entries"].sort(key=lambda e: (-e["bitrate"], -e["size"])) + # Streaming: emit ogni volta che il gruppo diventa/rimane ≥ 2 file + _gc(matched["id"], matched["entries"]) + else: + gid = f"fn_{len(groups)}_{fp_path.stem[:20]}" + groups.append({ + "id": gid, + "key_tokens": set(tokens), + "entries": [entry], + }) + _pc(i, total, fp_path.name, "cached") + + # Ritorna solo i gruppi con >= 2 file + result = [sorted(g["entries"], key=lambda e: (-e["bitrate"], -e["size"])) + for g in groups if len(g["entries"]) >= 2] + result.sort(key=lambda g: -max(e["size"] for e in g)) + _pc(total, total, "", "completed") + return result + + def scan_folder( directory: str, recursive: bool = True, progress_callback: Optional[Callable] = None, + method: str = "fingerprint", + group_callback: Optional[Callable] = None, ) -> list[list[dict]]: """Ritorna la lista di gruppi di file duplicati (>= 2 file). @@ -207,8 +353,11 @@ def scan_folder( occupano piu' spazio vengono prima). All'interno di ogni gruppo: bitrate DESC, poi size DESC (il primo e' quello "da tenere"). - Il fingerprint viene calcolato via `fpcalc -json` e messo in cache - SQLite. Al re-scan, se (size, mtime) invariati, non si rilancia fpcalc. + `method`: + - "fingerprint" (default): Chromaprint via fpcalc, preciso ma lento. + Cache SQLite persistente. Raggruppa per fingerprint identico. + - "filename": similarità Jaccard sui nomi file. Veloce ma euristico. + Non richiede fpcalc. """ reset_stop() files = _iter_audio_files(directory, recursive) @@ -218,6 +367,9 @@ def scan_folder( progress_callback(0, 0, "", "completed") return [] + if method == "filename": + return _scan_by_filename(files, progress_callback, group_callback) + fpcalc = find_fpcalc() if not fpcalc: # Senza fpcalc non possiamo fare nulla. Segnaliamo errore su ogni @@ -231,18 +383,26 @@ def scan_folder( # {fingerprint: [entry, ...]} by_fp: dict[str, list[dict]] = {} + def _pc(idx, total_n, name, status, err=""): + """Chiama progress_callback in modo retrocompatibile: la firma + legacy è a 4 args, quella nuova a 5 con `error_msg` opzionale.""" + if not progress_callback: + return + try: + progress_callback(idx, total_n, name, status, err) + except TypeError: + progress_callback(idx, total_n, name, status) + for i, fp_path in enumerate(files, start=1): if is_stopped(): - if progress_callback: - progress_callback(i - 1, total, "", "stopped") + _pc(i - 1, total, "", "stopped") return [] try: st = fp_path.stat() size = st.st_size mtime = st.st_mtime - except OSError: - if progress_callback: - progress_callback(i, total, fp_path.name, "error") + except OSError as e: + _pc(i, total, fp_path.name, "error", f"stat: {e}") continue path_str = str(fp_path) @@ -251,15 +411,13 @@ def scan_folder( fp_hash = cached["fingerprint"] duration = cached["duration"] bitrate = cached["bitrate"] or get_bitrate(fp_path) - if progress_callback: - progress_callback(i, total, fp_path.name, "cached") + _pc(i, total, fp_path.name, "cached") else: - if progress_callback: - progress_callback(i, total, fp_path.name, "computing") + _pc(i, total, fp_path.name, "computing") res = compute_fingerprint(fpcalc, path_str) - if not res: - if progress_callback: - progress_callback(i, total, fp_path.name, "error") + if not res or not res.get("fingerprint"): + err_msg = (res or {}).get("_error", "errore sconosciuto") + _pc(i, total, fp_path.name, "error", err_msg) continue fp_hash = res["fingerprint"] duration = res["duration"] @@ -276,7 +434,21 @@ def scan_folder( "duration": float(duration or 0), "fingerprint": fp_hash, } - by_fp.setdefault(fp_hash, []).append(entry) + grp = by_fp.setdefault(fp_hash, []) + grp.append(entry) + # Streaming: appena il gruppo raggiunge (o supera) 2 elementi, + # emetti update (JS accumula/aggiorna in tempo reale) + if group_callback and len(grp) >= 2: + # Ordinamento intra-gruppo prima di emit (best-to-keep primo) + grp.sort(key=lambda e: (-int(e.get("bitrate") or 0), + -int(e.get("size") or 0))) + try: + group_callback({ + "id": f"fp_{fp_hash[:24]}", + "entries": list(grp), + }) + except Exception: + pass finally: try: conn.close() diff --git a/tests/test_dedup.py b/tests/test_dedup.py index 98a0e9f..3e55f15 100644 --- a/tests/test_dedup.py +++ b/tests/test_dedup.py @@ -160,12 +160,13 @@ class TestMoveToTrash: # compute_fingerprint # ------------------------------------------------------------------ class TestComputeFingerprint: - def test_timeout_returns_none(self): - """Timeout di fpcalc -> None (non alza eccezione).""" + def test_timeout_returns_error(self): + """Timeout di fpcalc -> dict con _error, no fingerprint.""" with mock.patch("core.dedup.subprocess.run", side_effect=subprocess.TimeoutExpired(cmd="fpcalc", timeout=30)): result = dedup.compute_fingerprint("/fake/fpcalc", "/some/file.mp3") - assert result is None + assert result and "_error" in result and "timeout" in result["_error"] + assert "fingerprint" not in result def test_success_returns_dict(self): """Output JSON valido -> {duration, fingerprint}.""" @@ -176,15 +177,17 @@ class TestComputeFingerprint: result = dedup.compute_fingerprint("/fake/fpcalc", "/some/file.mp3") assert result == {"duration": 123.4, "fingerprint": "ABCDEF"} - def test_bad_json_returns_none(self): + def test_bad_json_returns_error(self): fake_proc = mock.Mock() fake_proc.returncode = 0 fake_proc.stdout = "not json at all" with mock.patch("core.dedup.subprocess.run", return_value=fake_proc): result = dedup.compute_fingerprint("/fake/fpcalc", "/some/file.mp3") - assert result is None + assert result and "_error" in result and "JSON" in result["_error"] + assert "fingerprint" not in result - def test_missing_fpcalc_returns_none(self): + def test_missing_fpcalc_returns_error(self): # Nessuna chiamata subprocess se fpcalc e' vuoto result = dedup.compute_fingerprint("", "/some/file.mp3") - assert result is None + assert result and "_error" in result and "fpcalc" in result["_error"] + assert "fingerprint" not in result diff --git a/webui/css/style.css b/webui/css/style.css index ede070a..86f0ebe 100644 --- a/webui/css/style.css +++ b/webui/css/style.css @@ -1962,3 +1962,11 @@ input[type="number"]::-webkit-inner-spin-button { border: 1px solid var(--border); } +/* Lista risultati Dedup con scroll interno: l'header (hero sticky) e il + footer restano visibili anche se ci sono centinaia di gruppi. */ +.dedup-groups-scroll { + max-height: 55vh; + overflow-y: auto; + padding-right: 6px; /* spazio per la scrollbar sul bordo */ +} + diff --git a/webui/index.html b/webui/index.html index 6386d32..914614d 100644 --- a/webui/index.html +++ b/webui/index.html @@ -1241,6 +1241,15 @@ Ricerca ricorsiva (include sottocartelle) +
+ +
@@ -1259,10 +1268,13 @@ -
+