diff --git a/core/royalty_check.py b/core/royalty_check.py new file mode 100644 index 0000000..ab50380 --- /dev/null +++ b/core/royalty_check.py @@ -0,0 +1,398 @@ +"""Verifica royalty-free status di brani locali contro DB CC/open. + +Approccio onesto: whitelist royalty-free. L'app segnala un brano come +"royalty_free" SOLO se trova un match affidabile in uno dei DB CC/open. +"Non trovato" (status 'unknown') NON implica automaticamente SIAE — +significa semplicemente "da verificare manualmente". + +Pipeline: + 1. Scansiona la cartella (filtra per `AUDIO_EXTENSIONS` di + `core.upgrader`). + 2. Per ogni file legge i tag ID3 (artist + title) via + `core.metadata.read_metadata`; fallback a + `core.catalog._parse_filename` se i tag sono mancanti. + 3. Query sorgenti royalty-free: + - ccMixter (nessuna auth, sempre attivo) + - Jamendo (richiede `jamendo_client_id` in config; se vuoto, skip) + 4. Levenshtein combinato artist+title >= `_SIMILARITY_THRESHOLD` → + status 'royalty_free'. Altrimenti 'unknown'. + +Progress callback firma: + (processed, total, filename, status[, err_msg]) +Status: 'checking' | 'stopped' | 'completed' | 'error'. + +`entry_callback(entry)` viene chiamato per ogni file processato +(streaming alla UI). +""" + +from __future__ import annotations + +import json +import sqlite3 +import threading +import time +from difflib import SequenceMatcher +from pathlib import Path +from typing import Callable, Optional + +import requests + +from core.upgrader import AUDIO_EXTENSIONS + + +# ------------------------------------------------------------------ +# Costanti +# ------------------------------------------------------------------ +_CCMIXTER_URL = "https://ccmixter.org/api/query" +_JAMENDO_URL = "https://api.jamendo.com/v3.0/tracks/" +_HTTP_TIMEOUT = 15 +_RATE_LIMIT_SEC = 0.3 +_SIMILARITY_THRESHOLD = 0.80 + + +# ------------------------------------------------------------------ +# Stop / interrupt +# ------------------------------------------------------------------ +_stop_event = threading.Event() + + +def request_stop() -> None: + """Segnala al worker di interrompere la scansione al prossimo file.""" + _stop_event.set() + + +def reset_stop() -> None: + """Azzera il flag di stop prima di iniziare una nuova scansione.""" + _stop_event.clear() + + +def is_stopped() -> bool: + return _stop_event.is_set() + + +# ------------------------------------------------------------------ +# Cache SQLite +# ------------------------------------------------------------------ +def _cache_db_path() -> Path: + """Path del file di cache SQLite (patchable nei test).""" + from core.config import _get_config_dir + return _get_config_dir() / "royalty_cache.db" + + +def _init_db(conn: sqlite3.Connection) -> None: + conn.executescript( + """ + CREATE TABLE IF NOT EXISTS lookups( + key TEXT PRIMARY KEY, + payload TEXT, + cached_at REAL + ); + """ + ) + + +# ------------------------------------------------------------------ +# Similarity +# ------------------------------------------------------------------ +def _similarity(a: str, b: str) -> float: + """Ritorna SequenceMatcher ratio (0.0 - 1.0) case-insensitive.""" + if not a or not b: + return 0.0 + return SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio() + + +# ------------------------------------------------------------------ +# Rate limit (globale) +# ------------------------------------------------------------------ +_last_req = [0.0] + + +def _throttle() -> None: + now = time.monotonic() + elapsed = now - _last_req[0] + if elapsed < _RATE_LIMIT_SEC: + time.sleep(_RATE_LIMIT_SEC - elapsed) + _last_req[0] = time.monotonic() + + +# ------------------------------------------------------------------ +# Query sorgenti +# ------------------------------------------------------------------ +def _query_ccmixter(query: str) -> list: + """Query ccMixter API. Ritorna lista di candidati normalizzati.""" + if not query: + return [] + _throttle() + try: + resp = requests.get( + _CCMIXTER_URL, + params={"f": "json", "search": query, "limit": 10}, + timeout=_HTTP_TIMEOUT, + ) + if resp.status_code != 200: + return [] + data = resp.json() + except Exception: + return [] + if not isinstance(data, list): + return [] + out = [] + for item in data: + if not isinstance(item, dict): + continue + out.append({ + "source": "ccmixter", + "title": item.get("upload_name", "") or "", + "artist": (item.get("user_real_name") + or item.get("user_name", "") + or ""), + "license": item.get("license_name", "") or "", + "license_url": item.get("license_url", "") or "", + "url": item.get("file_page_url", "") or "", + }) + return out + + +def _query_jamendo(query: str, client_id: str) -> list: + """Query Jamendo API. Senza client_id ritorna [].""" + if not client_id or not query: + return [] + _throttle() + try: + resp = requests.get( + _JAMENDO_URL, + params={ + "client_id": client_id, + "format": "json", + "limit": 5, + "search": query, + }, + timeout=_HTTP_TIMEOUT, + ) + if resp.status_code != 200: + return [] + data = resp.json() + except Exception: + return [] + if not isinstance(data, dict): + return [] + status = (data.get("headers", {}) or {}).get("status", "") + if status and status != "success": + return [] + out = [] + for item in data.get("results", []) or []: + if not isinstance(item, dict): + continue + out.append({ + "source": "jamendo", + "title": item.get("name", "") or "", + "artist": item.get("artist_name", "") or "", + "license": item.get("license_name") or "Creative Commons", + "license_url": item.get("license_ccurl", "") or "", + "url": item.get("shareurl", "") or "", + }) + return out + + +# ------------------------------------------------------------------ +# Check singolo file +# ------------------------------------------------------------------ +def check_file(path: str, jamendo_client_id: str = "") -> dict: + """Verifica un singolo file. Ritorna: + {path, status, matches, query, artist, title[, error]} + status: 'royalty_free' | 'unknown' | 'error'. + """ + from core.metadata import read_metadata + from core.catalog import _parse_filename + + try: + md = read_metadata(path) + except Exception as e: + return { + "path": path, + "status": "error", + "matches": [], + "query": "", + "artist": "", + "title": "", + "error": str(e), + } + + artist = (md.get("artist") or "").strip() + title = (md.get("title") or "").strip() + if not artist or not title: + parsed_artist, parsed_title = _parse_filename(Path(path).stem) + artist = artist or parsed_artist + title = title or parsed_title + + query = f"{artist} {title}".strip() if artist else title + if not query: + return { + "path": path, + "status": "unknown", + "matches": [], + "query": "", + "artist": artist, + "title": title, + "error": "no query (tag + filename vuoti)", + } + + # Cache lookup + conn = sqlite3.connect(str(_cache_db_path())) + try: + _init_db(conn) + row = conn.execute( + "SELECT payload FROM lookups WHERE key=?", (query,) + ).fetchone() + if row: + try: + cached = json.loads(row[0]) + cached["path"] = path + # Preserva artist/title correnti nel caso siano stati + # ricavati dal filename ma la cache li aveva salvati gia' + return cached + except Exception: + pass + + candidates = _query_ccmixter(query) + candidates += _query_jamendo(query, jamendo_client_id) + + matches = [] + for c in candidates: + score_title = _similarity(title, c.get("title", "")) + score_artist = ( + _similarity(artist, c.get("artist", "")) + if artist and c.get("artist") else 0.0 + ) + if artist: + combined = score_title * 0.6 + score_artist * 0.4 + else: + combined = score_title + if combined >= _SIMILARITY_THRESHOLD: + enriched = dict(c) + enriched["score"] = round(combined, 2) + matches.append(enriched) + + # Ordina per score desc + matches.sort(key=lambda m: m.get("score", 0.0), reverse=True) + + status = "royalty_free" if matches else "unknown" + result = { + "path": path, + "status": status, + "matches": matches, + "query": query, + "artist": artist, + "title": title, + } + + # Scrivi in cache (senza il path, che e' per-file) + try: + payload = {k: v for k, v in result.items() if k != "path"} + conn.execute( + "INSERT OR REPLACE INTO lookups(key, payload, cached_at) " + "VALUES (?, ?, ?)", + (query, json.dumps(payload, ensure_ascii=False), time.time()), + ) + conn.commit() + except Exception: + pass + + return result + finally: + try: + conn.close() + except Exception: + pass + + +# ------------------------------------------------------------------ +# Scan folder +# ------------------------------------------------------------------ +def _emit_progress(cb: Optional[Callable], idx: int, total: int, + name: str, status: str, err: str = "") -> None: + """Chiama progress_callback in modo retrocompatibile (4 o 5 args).""" + if not cb: + return + try: + cb(idx, total, name, status, err) + except TypeError: + try: + cb(idx, total, name, status) + except Exception: + pass + except Exception: + pass + + +def scan_folder( + directory: str, + recursive: bool = True, + progress_callback: Optional[Callable] = None, + entry_callback: Optional[Callable] = None, +) -> list: + """Scansiona la cartella e verifica ogni file audio. + + Ritorna la lista dei risultati per file (vedi `check_file`). + `entry_callback(entry)` e' chiamato appena ogni file e' processato. + """ + reset_stop() + base = Path(directory) + if not base.exists() or not base.is_dir(): + _emit_progress(progress_callback, 0, 0, "", "completed", "") + return [] + + files: list = [] + iterator = base.rglob("*") if recursive else base.iterdir() + for f in iterator: + try: + if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS: + files.append(f) + except OSError: + continue + files.sort() + total = len(files) + + if total == 0: + _emit_progress(progress_callback, 0, 0, "", "completed", "") + return [] + + # Jamendo client_id (opzionale) + jamendo_id = "" + try: + from core.config import load_config + jamendo_id = (load_config().get("jamendo_client_id") or "").strip() + except Exception: + jamendo_id = "" + + results: list = [] + for i, fp in enumerate(files, start=1): + if is_stopped(): + _emit_progress(progress_callback, i - 1, total, "", "stopped", "") + return results + + _emit_progress(progress_callback, i, total, fp.name, "checking", "") + try: + r = check_file(str(fp), jamendo_client_id=jamendo_id) + except Exception as e: + r = { + "path": str(fp), + "status": "error", + "matches": [], + "query": "", + "artist": "", + "title": "", + "error": str(e), + } + _emit_progress(progress_callback, i, total, fp.name, "error", str(e)) + + results.append(r) + if entry_callback: + try: + entry_callback(r) + except Exception: + pass + + if not is_stopped(): + _emit_progress(progress_callback, total, total, "", "completed", "") + return results diff --git a/tests/test_royalty_check.py b/tests/test_royalty_check.py new file mode 100644 index 0000000..0afc9ac --- /dev/null +++ b/tests/test_royalty_check.py @@ -0,0 +1,373 @@ +"""Test per core.royalty_check — ccMixter + Jamendo + cache SQLite. + +Nessuna integrazione reale: tutte le chiamate di rete sono mockate con +`responses`. La cache SQLite e' isolata in tmp_path. +""" + +from __future__ import annotations + +from pathlib import Path +from unittest import mock + +import pytest +import responses + +from core import royalty_check as rc + + +# ------------------------------------------------------------------ +# Fixture: cache isolata per test +# ------------------------------------------------------------------ +@pytest.fixture +def patched_cache(tmp_path, monkeypatch): + db = tmp_path / "royalty_cache_test.db" + monkeypatch.setattr(rc, "_cache_db_path", lambda: db) + # Azzera il throttle globale per non rallentare i test + monkeypatch.setattr(rc, "_last_req", [0.0]) + monkeypatch.setattr(rc, "_RATE_LIMIT_SEC", 0.0) + return db + + +# ================================================================== +# _similarity +# ================================================================== +class TestSimilarity: + def test_similarity_perfect_match(self): + assert rc._similarity("hello world", "hello world") == pytest.approx(1.0) + + def test_similarity_case_insensitive(self): + assert rc._similarity("Hello World", "hello world") == pytest.approx(1.0) + + def test_similarity_empty_returns_zero(self): + assert rc._similarity("", "foo") == 0.0 + assert rc._similarity("foo", "") == 0.0 + assert rc._similarity("", "") == 0.0 + + def test_similarity_partial(self): + s = rc._similarity("Daft Punk - Around The World", "Around The World") + # SequenceMatcher usa chunks comuni: valore stabile e > 0.5 + assert 0.5 < s < 1.0 + + +# ================================================================== +# _query_ccmixter +# ================================================================== +class TestQueryCcmixter: + _SAMPLE = [ + { + "upload_name": "Around The World", + "user_real_name": "Jane Doe", + "user_name": "jane", + "license_name": "Attribution Noncommercial (3.0)", + "license_url": "https://creativecommons.org/licenses/by-nc/3.0/", + "file_page_url": "https://ccmixter.org/files/jane/1", + }, + { + "upload_name": "Other Song", + "user_real_name": "", + "user_name": "bob", + "license_name": "CC BY 4.0", + "license_url": "https://creativecommons.org/licenses/by/4.0/", + "file_page_url": "https://ccmixter.org/files/bob/2", + }, + ] + + @responses.activate + def test_query_ccmixter_parses_response(self, patched_cache): + responses.add( + responses.GET, + rc._CCMIXTER_URL, + json=self._SAMPLE, + status=200, + ) + res = rc._query_ccmixter("Around The World") + assert len(res) == 2 + assert res[0]["source"] == "ccmixter" + assert res[0]["title"] == "Around The World" + assert res[0]["artist"] == "Jane Doe" + assert "licenses" in res[0]["license_url"] + # Fallback a user_name quando user_real_name e' vuoto + assert res[1]["artist"] == "bob" + + @responses.activate + def test_query_ccmixter_empty_on_http_error(self, patched_cache): + responses.add( + responses.GET, + rc._CCMIXTER_URL, + json={"error": "boom"}, + status=500, + ) + assert rc._query_ccmixter("whatever") == [] + + @responses.activate + def test_query_ccmixter_empty_on_bad_json(self, patched_cache): + responses.add( + responses.GET, + rc._CCMIXTER_URL, + json={"not": "a list"}, + status=200, + ) + assert rc._query_ccmixter("whatever") == [] + + def test_query_ccmixter_empty_query(self, patched_cache): + # Nessuna chiamata di rete attesa — se la facesse responses fallirebbe + assert rc._query_ccmixter("") == [] + + +# ================================================================== +# _query_jamendo +# ================================================================== +class TestQueryJamendo: + _SAMPLE = { + "headers": {"status": "success"}, + "results": [ + { + "id": "123", + "name": "Freedom Song", + "artist_name": "Open Artist", + "license_name": "CC BY-SA", + "license_ccurl": "https://creativecommons.org/licenses/by-sa/4.0/", + "shareurl": "https://www.jamendo.com/track/123", + "audio": "https://.../123.mp3", + } + ], + } + + def test_query_jamendo_requires_client_id(self, patched_cache): + # Nessun client_id -> ritorno vuoto senza chiamata di rete + # (responses non e' attivato: una chiamata farebbe raise) + assert rc._query_jamendo("Freedom Song", client_id="") == [] + + @responses.activate + def test_query_jamendo_parses_response(self, patched_cache): + responses.add( + responses.GET, + rc._JAMENDO_URL, + json=self._SAMPLE, + status=200, + ) + res = rc._query_jamendo("Freedom Song", client_id="abcd1234") + assert len(res) == 1 + assert res[0]["source"] == "jamendo" + assert res[0]["title"] == "Freedom Song" + assert res[0]["artist"] == "Open Artist" + assert res[0]["license"] == "CC BY-SA" + assert res[0]["url"].startswith("https://www.jamendo.com/") + + @responses.activate + def test_query_jamendo_status_failure_returns_empty(self, patched_cache): + responses.add( + responses.GET, + rc._JAMENDO_URL, + json={"headers": {"status": "failed", "error_message": "bad key"}, + "results": []}, + status=200, + ) + assert rc._query_jamendo("Freedom Song", client_id="bad") == [] + + +# ================================================================== +# check_file +# ================================================================== +def _fake_metadata(artist: str, title: str): + """Factory per un dict compatibile con core.metadata.read_metadata.""" + return {"artist": artist, "title": title} + + +class TestCheckFile: + def test_check_file_matches_royalty_free(self, patched_cache, tmp_path): + # Prepara un path "esistente" solo per plausibilita' (il + # read_metadata e' mockato: non viene mai letto davvero). + p = tmp_path / "track.mp3" + p.write_bytes(b"\x00") + + cc_hit = [{ + "upload_name": "Around The World", + "user_real_name": "Jane Doe", + "user_name": "jane", + "license_name": "CC BY 3.0", + "license_url": "https://creativecommons.org/licenses/by/3.0/", + "file_page_url": "https://ccmixter.org/x/1", + }] + + with mock.patch.object(rc, "_query_ccmixter", return_value=[{ + "source": "ccmixter", + "title": "Around The World", + "artist": "Jane Doe", + "license": "CC BY 3.0", + "license_url": cc_hit[0]["license_url"], + "url": cc_hit[0]["file_page_url"], + }]), mock.patch.object(rc, "_query_jamendo", return_value=[]), \ + mock.patch("core.metadata.read_metadata", + return_value=_fake_metadata("Jane Doe", + "Around The World")): + res = rc.check_file(str(p)) + + assert res["status"] == "royalty_free" + assert len(res["matches"]) == 1 + assert res["matches"][0]["source"] == "ccmixter" + assert res["matches"][0]["score"] >= 0.8 + assert res["artist"] == "Jane Doe" + assert res["title"] == "Around The World" + + def test_check_file_no_match_returns_unknown(self, patched_cache, tmp_path): + p = tmp_path / "unknown.mp3" + p.write_bytes(b"\x00") + + with mock.patch.object(rc, "_query_ccmixter", return_value=[{ + "source": "ccmixter", + "title": "Completamente Diverso", + "artist": "Qualcun Altro", + "license": "CC BY", + "license_url": "", + "url": "", + }]), mock.patch.object(rc, "_query_jamendo", return_value=[]), \ + mock.patch("core.metadata.read_metadata", + return_value=_fake_metadata("Daft Punk", + "Harder Better Faster")): + res = rc.check_file(str(p)) + + assert res["status"] == "unknown" + assert res["matches"] == [] + # artist/title conservati dai tag + assert res["artist"] == "Daft Punk" + + def test_check_file_uses_cache(self, patched_cache, tmp_path): + p = tmp_path / "cached.mp3" + p.write_bytes(b"\x00") + + cc_mock = mock.MagicMock(return_value=[{ + "source": "ccmixter", + "title": "Hello", + "artist": "World", + "license": "CC BY", + "license_url": "", + "url": "", + }]) + jm_mock = mock.MagicMock(return_value=[]) + md_mock = mock.MagicMock(return_value=_fake_metadata("World", "Hello")) + + with mock.patch.object(rc, "_query_ccmixter", cc_mock), \ + mock.patch.object(rc, "_query_jamendo", jm_mock), \ + mock.patch("core.metadata.read_metadata", md_mock): + res1 = rc.check_file(str(p)) + res2 = rc.check_file(str(p)) + + assert res1["status"] == "royalty_free" + assert res2["status"] == "royalty_free" + # Seconda chiamata hitta la cache: le query API non vengono richiamate + assert cc_mock.call_count == 1 + assert jm_mock.call_count == 1 + + def test_check_file_fallback_to_filename_when_tags_missing( + self, patched_cache, tmp_path): + """Senza tag ID3, usa _parse_filename sul nome del file.""" + # Filename stile "Artista - Titolo.mp3" + p = tmp_path / "Jane Doe - Around The World.mp3" + p.write_bytes(b"\x00") + + with mock.patch.object(rc, "_query_ccmixter", return_value=[{ + "source": "ccmixter", + "title": "Around The World", + "artist": "Jane Doe", + "license": "CC BY", + "license_url": "", + "url": "", + }]), mock.patch.object(rc, "_query_jamendo", return_value=[]), \ + mock.patch("core.metadata.read_metadata", + return_value=_fake_metadata("", "")): + res = rc.check_file(str(p)) + + assert res["status"] == "royalty_free" + assert res["artist"] == "Jane Doe" + assert res["title"] == "Around The World" + + def test_check_file_metadata_error_returns_error_status( + self, patched_cache, tmp_path): + p = tmp_path / "broken.mp3" + p.write_bytes(b"\x00") + + with mock.patch("core.metadata.read_metadata", + side_effect=ValueError("formato illeggibile")): + res = rc.check_file(str(p)) + + assert res["status"] == "error" + assert "formato illeggibile" in res.get("error", "") + + +# ================================================================== +# scan_folder +# ================================================================== +class TestScanFolder: + def test_scan_folder_streams_entries(self, patched_cache, tmp_path): + # Crea 3 file audio nella cartella + for name in ("a.mp3", "b.mp3", "c.flac"): + (tmp_path / name).write_bytes(b"\x00") + + fake_result = { + "path": "", + "status": "unknown", + "matches": [], + "query": "q", + "artist": "A", + "title": "T", + } + + collected = [] + + def _entry_cb(entry): + collected.append(entry) + + with mock.patch.object(rc, "check_file", return_value=fake_result): + results = rc.scan_folder( + str(tmp_path), + recursive=False, + entry_callback=_entry_cb, + ) + + assert len(results) == 3 + assert len(collected) == 3 + + def test_scan_folder_filters_non_audio(self, patched_cache, tmp_path): + (tmp_path / "ok.mp3").write_bytes(b"\x00") + (tmp_path / "note.txt").write_text("hello") + (tmp_path / "image.png").write_bytes(b"\x00") + + with mock.patch.object(rc, "check_file", + return_value={"path": "", "status": "unknown", + "matches": [], "query": "", + "artist": "", "title": ""}): + results = rc.scan_folder(str(tmp_path), recursive=False) + assert len(results) == 1 + + def test_scan_folder_empty_directory(self, patched_cache, tmp_path): + empty = tmp_path / "empty" + empty.mkdir() + results = rc.scan_folder(str(empty), recursive=False) + assert results == [] + + def test_scan_folder_nonexistent_directory(self, patched_cache, tmp_path): + results = rc.scan_folder(str(tmp_path / "missing"), recursive=False) + assert results == [] + + def test_scan_folder_progress_callback_called(self, patched_cache, tmp_path): + (tmp_path / "a.mp3").write_bytes(b"\x00") + (tmp_path / "b.mp3").write_bytes(b"\x00") + + events = [] + + def _progress(idx, total, name, status, err=""): + events.append((idx, total, status)) + + fake_result = { + "path": "", "status": "unknown", "matches": [], + "query": "", "artist": "", "title": "", + } + + with mock.patch.object(rc, "check_file", return_value=fake_result): + rc.scan_folder(str(tmp_path), recursive=False, + progress_callback=_progress) + + # Attendo almeno "checking" per ciascun file + "completed" finale + assert any(e[2] == "completed" for e in events) + assert sum(1 for e in events if e[2] == "checking") >= 2