royalty: modulo check ccMixter + Jamendo con cache SQLite

Nuova feature "Diritti": verifica royalty-free status di brani locali
contro DB CC/open (ccMixter sempre attivo, Jamendo opzionale via
jamendo_client_id in config). Approccio honest-only:

- status 'royalty_free' SOLO con match Levenshtein combinato >= 0.80
- status 'unknown' NON implica automaticamente SIAE (= da verificare)
- status 'error' per fingerprint fail / network / tag illeggibili

Pipeline:
- Scan AUDIO_EXTENSIONS da core.upgrader (ricorsivo opzionale)
- Legge artist/title da ID3 (core.metadata.read_metadata), fallback
  a core.catalog._parse_filename sul nome file
- Query parallele ccMixter + Jamendo con throttle 0.3s
- Cache SQLite in <config_dir>/royalty_cache.db

21 test verdi (mock requests via responses, cache isolata in tmp_path).

Co-Authored-By: Claude Opus 4.7 <noreply@anthropic.com>
This commit is contained in:
lucianoandClaude Opus 4.7 committed 2026-10-09 14:25:10 +02:00
1 parent 9644ffec35
commit 018d310f53
2 files changed
+771

No files matched your search

+398
View File
@@ -0,0 +1,398 @@
"""Verifica royalty-free status di brani locali contro DB CC/open.
Approccio onesto: whitelist royalty-free. L'app segnala un brano come
"royalty_free" SOLO se trova un match affidabile in uno dei DB CC/open.
"Non trovato" (status 'unknown') NON implica automaticamente SIAE —
significa semplicemente "da verificare manualmente".
Pipeline:
1. Scansiona la cartella (filtra per `AUDIO_EXTENSIONS` di
`core.upgrader`).
2. Per ogni file legge i tag ID3 (artist + title) via
`core.metadata.read_metadata`; fallback a
`core.catalog._parse_filename` se i tag sono mancanti.
3. Query sorgenti royalty-free:
- ccMixter (nessuna auth, sempre attivo)
- Jamendo (richiede `jamendo_client_id` in config; se vuoto, skip)
4. Levenshtein combinato artist+title >= `_SIMILARITY_THRESHOLD` →
status 'royalty_free'. Altrimenti 'unknown'.
Progress callback firma:
(processed, total, filename, status[, err_msg])
Status: 'checking' | 'stopped' | 'completed' | 'error'.
`entry_callback(entry)` viene chiamato per ogni file processato
(streaming alla UI).
"""
from __future__ import annotations
import json
import sqlite3
import threading
import time
from difflib import SequenceMatcher
from pathlib import Path
from typing import Callable, Optional
import requests
from core.upgrader import AUDIO_EXTENSIONS
# ------------------------------------------------------------------
# Costanti
# ------------------------------------------------------------------
_CCMIXTER_URL = "https://ccmixter.org/api/query"
_JAMENDO_URL = "https://api.jamendo.com/v3.0/tracks/"
_HTTP_TIMEOUT = 15
_RATE_LIMIT_SEC = 0.3
_SIMILARITY_THRESHOLD = 0.80
# ------------------------------------------------------------------
# Stop / interrupt
# ------------------------------------------------------------------
_stop_event = threading.Event()
def request_stop() -> None:
"""Segnala al worker di interrompere la scansione al prossimo file."""
_stop_event.set()
def reset_stop() -> None:
"""Azzera il flag di stop prima di iniziare una nuova scansione."""
_stop_event.clear()
def is_stopped() -> bool:
return _stop_event.is_set()
# ------------------------------------------------------------------
# Cache SQLite
# ------------------------------------------------------------------
def _cache_db_path() -> Path:
"""Path del file di cache SQLite (patchable nei test)."""
from core.config import _get_config_dir
return _get_config_dir() / "royalty_cache.db"
def _init_db(conn: sqlite3.Connection) -> None:
conn.executescript(
"""
CREATE TABLE IF NOT EXISTS lookups(
key TEXT PRIMARY KEY,
payload TEXT,
cached_at REAL
);
"""
)
# ------------------------------------------------------------------
# Similarity
# ------------------------------------------------------------------
def _similarity(a: str, b: str) -> float:
"""Ritorna SequenceMatcher ratio (0.0 - 1.0) case-insensitive."""
if not a or not b:
return 0.0
return SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio()
# ------------------------------------------------------------------
# Rate limit (globale)
# ------------------------------------------------------------------
_last_req = [0.0]
def _throttle() -> None:
now = time.monotonic()
elapsed = now - _last_req[0]
if elapsed < _RATE_LIMIT_SEC:
time.sleep(_RATE_LIMIT_SEC - elapsed)
_last_req[0] = time.monotonic()
# ------------------------------------------------------------------
# Query sorgenti
# ------------------------------------------------------------------
def _query_ccmixter(query: str) -> list:
"""Query ccMixter API. Ritorna lista di candidati normalizzati."""
if not query:
return []
_throttle()
try:
resp = requests.get(
_CCMIXTER_URL,
params={"f": "json", "search": query, "limit": 10},
timeout=_HTTP_TIMEOUT,
)
if resp.status_code != 200:
return []
data = resp.json()
except Exception:
return []
if not isinstance(data, list):
return []
out = []
for item in data:
if not isinstance(item, dict):
continue
out.append({
"source": "ccmixter",
"title": item.get("upload_name", "") or "",
"artist": (item.get("user_real_name")
or item.get("user_name", "")
or ""),
"license": item.get("license_name", "") or "",
"license_url": item.get("license_url", "") or "",
"url": item.get("file_page_url", "") or "",
})
return out
def _query_jamendo(query: str, client_id: str) -> list:
"""Query Jamendo API. Senza client_id ritorna []."""
if not client_id or not query:
return []
_throttle()
try:
resp = requests.get(
_JAMENDO_URL,
params={
"client_id": client_id,
"format": "json",
"limit": 5,
"search": query,
},
timeout=_HTTP_TIMEOUT,
)
if resp.status_code != 200:
return []
data = resp.json()
except Exception:
return []
if not isinstance(data, dict):
return []
status = (data.get("headers", {}) or {}).get("status", "")
if status and status != "success":
return []
out = []
for item in data.get("results", []) or []:
if not isinstance(item, dict):
continue
out.append({
"source": "jamendo",
"title": item.get("name", "") or "",
"artist": item.get("artist_name", "") or "",
"license": item.get("license_name") or "Creative Commons",
"license_url": item.get("license_ccurl", "") or "",
"url": item.get("shareurl", "") or "",
})
return out
# ------------------------------------------------------------------
# Check singolo file
# ------------------------------------------------------------------
def check_file(path: str, jamendo_client_id: str = "") -> dict:
"""Verifica un singolo file. Ritorna:
{path, status, matches, query, artist, title[, error]}
status: 'royalty_free' | 'unknown' | 'error'.
"""
from core.metadata import read_metadata
from core.catalog import _parse_filename
try:
md = read_metadata(path)
except Exception as e:
return {
"path": path,
"status": "error",
"matches": [],
"query": "",
"artist": "",
"title": "",
"error": str(e),
}
artist = (md.get("artist") or "").strip()
title = (md.get("title") or "").strip()
if not artist or not title:
parsed_artist, parsed_title = _parse_filename(Path(path).stem)
artist = artist or parsed_artist
title = title or parsed_title
query = f"{artist} {title}".strip() if artist else title
if not query:
return {
"path": path,
"status": "unknown",
"matches": [],
"query": "",
"artist": artist,
"title": title,
"error": "no query (tag + filename vuoti)",
}
# Cache lookup
conn = sqlite3.connect(str(_cache_db_path()))
try:
_init_db(conn)
row = conn.execute(
"SELECT payload FROM lookups WHERE key=?", (query,)
).fetchone()
if row:
try:
cached = json.loads(row[0])
cached["path"] = path
# Preserva artist/title correnti nel caso siano stati
# ricavati dal filename ma la cache li aveva salvati gia'
return cached
except Exception:
pass
candidates = _query_ccmixter(query)
candidates += _query_jamendo(query, jamendo_client_id)
matches = []
for c in candidates:
score_title = _similarity(title, c.get("title", ""))
score_artist = (
_similarity(artist, c.get("artist", ""))
if artist and c.get("artist") else 0.0
)
if artist:
combined = score_title * 0.6 + score_artist * 0.4
else:
combined = score_title
if combined >= _SIMILARITY_THRESHOLD:
enriched = dict(c)
enriched["score"] = round(combined, 2)
matches.append(enriched)
# Ordina per score desc
matches.sort(key=lambda m: m.get("score", 0.0), reverse=True)
status = "royalty_free" if matches else "unknown"
result = {
"path": path,
"status": status,
"matches": matches,
"query": query,
"artist": artist,
"title": title,
}
# Scrivi in cache (senza il path, che e' per-file)
try:
payload = {k: v for k, v in result.items() if k != "path"}
conn.execute(
"INSERT OR REPLACE INTO lookups(key, payload, cached_at) "
"VALUES (?, ?, ?)",
(query, json.dumps(payload, ensure_ascii=False), time.time()),
)
conn.commit()
except Exception:
pass
return result
finally:
try:
conn.close()
except Exception:
pass
# ------------------------------------------------------------------
# Scan folder
# ------------------------------------------------------------------
def _emit_progress(cb: Optional[Callable], idx: int, total: int,
name: str, status: str, err: str = "") -> None:
"""Chiama progress_callback in modo retrocompatibile (4 o 5 args)."""
if not cb:
return
try:
cb(idx, total, name, status, err)
except TypeError:
try:
cb(idx, total, name, status)
except Exception:
pass
except Exception:
pass
def scan_folder(
directory: str,
recursive: bool = True,
progress_callback: Optional[Callable] = None,
entry_callback: Optional[Callable] = None,
) -> list:
"""Scansiona la cartella e verifica ogni file audio.
Ritorna la lista dei risultati per file (vedi `check_file`).
`entry_callback(entry)` e' chiamato appena ogni file e' processato.
"""
reset_stop()
base = Path(directory)
if not base.exists() or not base.is_dir():
_emit_progress(progress_callback, 0, 0, "", "completed", "")
return []
files: list = []
iterator = base.rglob("*") if recursive else base.iterdir()
for f in iterator:
try:
if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS:
files.append(f)
except OSError:
continue
files.sort()
total = len(files)
if total == 0:
_emit_progress(progress_callback, 0, 0, "", "completed", "")
return []
# Jamendo client_id (opzionale)
jamendo_id = ""
try:
from core.config import load_config
jamendo_id = (load_config().get("jamendo_client_id") or "").strip()
except Exception:
jamendo_id = ""
results: list = []
for i, fp in enumerate(files, start=1):
if is_stopped():
_emit_progress(progress_callback, i - 1, total, "", "stopped", "")
return results
_emit_progress(progress_callback, i, total, fp.name, "checking", "")
try:
r = check_file(str(fp), jamendo_client_id=jamendo_id)
except Exception as e:
r = {
"path": str(fp),
"status": "error",
"matches": [],
"query": "",
"artist": "",
"title": "",
"error": str(e),
}
_emit_progress(progress_callback, i, total, fp.name, "error", str(e))
results.append(r)
if entry_callback:
try:
entry_callback(r)
except Exception:
pass
if not is_stopped():
_emit_progress(progress_callback, total, total, "", "completed", "")
return results
+373
View File
@@ -0,0 +1,373 @@
"""Test per core.royalty_check — ccMixter + Jamendo + cache SQLite.
Nessuna integrazione reale: tutte le chiamate di rete sono mockate con
`responses`. La cache SQLite e' isolata in tmp_path.
"""
from __future__ import annotations
from pathlib import Path
from unittest import mock
import pytest
import responses
from core import royalty_check as rc
# ------------------------------------------------------------------
# Fixture: cache isolata per test
# ------------------------------------------------------------------
@pytest.fixture
def patched_cache(tmp_path, monkeypatch):
db = tmp_path / "royalty_cache_test.db"
monkeypatch.setattr(rc, "_cache_db_path", lambda: db)
# Azzera il throttle globale per non rallentare i test
monkeypatch.setattr(rc, "_last_req", [0.0])
monkeypatch.setattr(rc, "_RATE_LIMIT_SEC", 0.0)
return db
# ==================================================================
# _similarity
# ==================================================================
class TestSimilarity:
def test_similarity_perfect_match(self):
assert rc._similarity("hello world", "hello world") == pytest.approx(1.0)
def test_similarity_case_insensitive(self):
assert rc._similarity("Hello World", "hello world") == pytest.approx(1.0)
def test_similarity_empty_returns_zero(self):
assert rc._similarity("", "foo") == 0.0
assert rc._similarity("foo", "") == 0.0
assert rc._similarity("", "") == 0.0
def test_similarity_partial(self):
s = rc._similarity("Daft Punk - Around The World", "Around The World")
# SequenceMatcher usa chunks comuni: valore stabile e > 0.5
assert 0.5 < s < 1.0
# ==================================================================
# _query_ccmixter
# ==================================================================
class TestQueryCcmixter:
_SAMPLE = [
{
"upload_name": "Around The World",
"user_real_name": "Jane Doe",
"user_name": "jane",
"license_name": "Attribution Noncommercial (3.0)",
"license_url": "https://creativecommons.org/licenses/by-nc/3.0/",
"file_page_url": "https://ccmixter.org/files/jane/1",
},
{
"upload_name": "Other Song",
"user_real_name": "",
"user_name": "bob",
"license_name": "CC BY 4.0",
"license_url": "https://creativecommons.org/licenses/by/4.0/",
"file_page_url": "https://ccmixter.org/files/bob/2",
},
]
@responses.activate
def test_query_ccmixter_parses_response(self, patched_cache):
responses.add(
responses.GET,
rc._CCMIXTER_URL,
json=self._SAMPLE,
status=200,
)
res = rc._query_ccmixter("Around The World")
assert len(res) == 2
assert res[0]["source"] == "ccmixter"
assert res[0]["title"] == "Around The World"
assert res[0]["artist"] == "Jane Doe"
assert "licenses" in res[0]["license_url"]
# Fallback a user_name quando user_real_name e' vuoto
assert res[1]["artist"] == "bob"
@responses.activate
def test_query_ccmixter_empty_on_http_error(self, patched_cache):
responses.add(
responses.GET,
rc._CCMIXTER_URL,
json={"error": "boom"},
status=500,
)
assert rc._query_ccmixter("whatever") == []
@responses.activate
def test_query_ccmixter_empty_on_bad_json(self, patched_cache):
responses.add(
responses.GET,
rc._CCMIXTER_URL,
json={"not": "a list"},
status=200,
)
assert rc._query_ccmixter("whatever") == []
def test_query_ccmixter_empty_query(self, patched_cache):
# Nessuna chiamata di rete attesa — se la facesse responses fallirebbe
assert rc._query_ccmixter("") == []
# ==================================================================
# _query_jamendo
# ==================================================================
class TestQueryJamendo:
_SAMPLE = {
"headers": {"status": "success"},
"results": [
{
"id": "123",
"name": "Freedom Song",
"artist_name": "Open Artist",
"license_name": "CC BY-SA",
"license_ccurl": "https://creativecommons.org/licenses/by-sa/4.0/",
"shareurl": "https://www.jamendo.com/track/123",
"audio": "https://.../123.mp3",
}
],
}
def test_query_jamendo_requires_client_id(self, patched_cache):
# Nessun client_id -> ritorno vuoto senza chiamata di rete
# (responses non e' attivato: una chiamata farebbe raise)
assert rc._query_jamendo("Freedom Song", client_id="") == []
@responses.activate
def test_query_jamendo_parses_response(self, patched_cache):
responses.add(
responses.GET,
rc._JAMENDO_URL,
json=self._SAMPLE,
status=200,
)
res = rc._query_jamendo("Freedom Song", client_id="abcd1234")
assert len(res) == 1
assert res[0]["source"] == "jamendo"
assert res[0]["title"] == "Freedom Song"
assert res[0]["artist"] == "Open Artist"
assert res[0]["license"] == "CC BY-SA"
assert res[0]["url"].startswith("https://www.jamendo.com/")
@responses.activate
def test_query_jamendo_status_failure_returns_empty(self, patched_cache):
responses.add(
responses.GET,
rc._JAMENDO_URL,
json={"headers": {"status": "failed", "error_message": "bad key"},
"results": []},
status=200,
)
assert rc._query_jamendo("Freedom Song", client_id="bad") == []
# ==================================================================
# check_file
# ==================================================================
def _fake_metadata(artist: str, title: str):
"""Factory per un dict compatibile con core.metadata.read_metadata."""
return {"artist": artist, "title": title}
class TestCheckFile:
def test_check_file_matches_royalty_free(self, patched_cache, tmp_path):
# Prepara un path "esistente" solo per plausibilita' (il
# read_metadata e' mockato: non viene mai letto davvero).
p = tmp_path / "track.mp3"
p.write_bytes(b"\x00")
cc_hit = [{
"upload_name": "Around The World",
"user_real_name": "Jane Doe",
"user_name": "jane",
"license_name": "CC BY 3.0",
"license_url": "https://creativecommons.org/licenses/by/3.0/",
"file_page_url": "https://ccmixter.org/x/1",
}]
with mock.patch.object(rc, "_query_ccmixter", return_value=[{
"source": "ccmixter",
"title": "Around The World",
"artist": "Jane Doe",
"license": "CC BY 3.0",
"license_url": cc_hit[0]["license_url"],
"url": cc_hit[0]["file_page_url"],
}]), mock.patch.object(rc, "_query_jamendo", return_value=[]), \
mock.patch("core.metadata.read_metadata",
return_value=_fake_metadata("Jane Doe",
"Around The World")):
res = rc.check_file(str(p))
assert res["status"] == "royalty_free"
assert len(res["matches"]) == 1
assert res["matches"][0]["source"] == "ccmixter"
assert res["matches"][0]["score"] >= 0.8
assert res["artist"] == "Jane Doe"
assert res["title"] == "Around The World"
def test_check_file_no_match_returns_unknown(self, patched_cache, tmp_path):
p = tmp_path / "unknown.mp3"
p.write_bytes(b"\x00")
with mock.patch.object(rc, "_query_ccmixter", return_value=[{
"source": "ccmixter",
"title": "Completamente Diverso",
"artist": "Qualcun Altro",
"license": "CC BY",
"license_url": "",
"url": "",
}]), mock.patch.object(rc, "_query_jamendo", return_value=[]), \
mock.patch("core.metadata.read_metadata",
return_value=_fake_metadata("Daft Punk",
"Harder Better Faster")):
res = rc.check_file(str(p))
assert res["status"] == "unknown"
assert res["matches"] == []
# artist/title conservati dai tag
assert res["artist"] == "Daft Punk"
def test_check_file_uses_cache(self, patched_cache, tmp_path):
p = tmp_path / "cached.mp3"
p.write_bytes(b"\x00")
cc_mock = mock.MagicMock(return_value=[{
"source": "ccmixter",
"title": "Hello",
"artist": "World",
"license": "CC BY",
"license_url": "",
"url": "",
}])
jm_mock = mock.MagicMock(return_value=[])
md_mock = mock.MagicMock(return_value=_fake_metadata("World", "Hello"))
with mock.patch.object(rc, "_query_ccmixter", cc_mock), \
mock.patch.object(rc, "_query_jamendo", jm_mock), \
mock.patch("core.metadata.read_metadata", md_mock):
res1 = rc.check_file(str(p))
res2 = rc.check_file(str(p))
assert res1["status"] == "royalty_free"
assert res2["status"] == "royalty_free"
# Seconda chiamata hitta la cache: le query API non vengono richiamate
assert cc_mock.call_count == 1
assert jm_mock.call_count == 1
def test_check_file_fallback_to_filename_when_tags_missing(
self, patched_cache, tmp_path):
"""Senza tag ID3, usa _parse_filename sul nome del file."""
# Filename stile "Artista - Titolo.mp3"
p = tmp_path / "Jane Doe - Around The World.mp3"
p.write_bytes(b"\x00")
with mock.patch.object(rc, "_query_ccmixter", return_value=[{
"source": "ccmixter",
"title": "Around The World",
"artist": "Jane Doe",
"license": "CC BY",
"license_url": "",
"url": "",
}]), mock.patch.object(rc, "_query_jamendo", return_value=[]), \
mock.patch("core.metadata.read_metadata",
return_value=_fake_metadata("", "")):
res = rc.check_file(str(p))
assert res["status"] == "royalty_free"
assert res["artist"] == "Jane Doe"
assert res["title"] == "Around The World"
def test_check_file_metadata_error_returns_error_status(
self, patched_cache, tmp_path):
p = tmp_path / "broken.mp3"
p.write_bytes(b"\x00")
with mock.patch("core.metadata.read_metadata",
side_effect=ValueError("formato illeggibile")):
res = rc.check_file(str(p))
assert res["status"] == "error"
assert "formato illeggibile" in res.get("error", "")
# ==================================================================
# scan_folder
# ==================================================================
class TestScanFolder:
def test_scan_folder_streams_entries(self, patched_cache, tmp_path):
# Crea 3 file audio nella cartella
for name in ("a.mp3", "b.mp3", "c.flac"):
(tmp_path / name).write_bytes(b"\x00")
fake_result = {
"path": "",
"status": "unknown",
"matches": [],
"query": "q",
"artist": "A",
"title": "T",
}
collected = []
def _entry_cb(entry):
collected.append(entry)
with mock.patch.object(rc, "check_file", return_value=fake_result):
results = rc.scan_folder(
str(tmp_path),
recursive=False,
entry_callback=_entry_cb,
)
assert len(results) == 3
assert len(collected) == 3
def test_scan_folder_filters_non_audio(self, patched_cache, tmp_path):
(tmp_path / "ok.mp3").write_bytes(b"\x00")
(tmp_path / "note.txt").write_text("hello")
(tmp_path / "image.png").write_bytes(b"\x00")
with mock.patch.object(rc, "check_file",
return_value={"path": "", "status": "unknown",
"matches": [], "query": "",
"artist": "", "title": ""}):
results = rc.scan_folder(str(tmp_path), recursive=False)
assert len(results) == 1
def test_scan_folder_empty_directory(self, patched_cache, tmp_path):
empty = tmp_path / "empty"
empty.mkdir()
results = rc.scan_folder(str(empty), recursive=False)
assert results == []
def test_scan_folder_nonexistent_directory(self, patched_cache, tmp_path):
results = rc.scan_folder(str(tmp_path / "missing"), recursive=False)
assert results == []
def test_scan_folder_progress_callback_called(self, patched_cache, tmp_path):
(tmp_path / "a.mp3").write_bytes(b"\x00")
(tmp_path / "b.mp3").write_bytes(b"\x00")
events = []
def _progress(idx, total, name, status, err=""):
events.append((idx, total, status))
fake_result = {
"path": "", "status": "unknown", "matches": [],
"query": "", "artist": "", "title": "",
}
with mock.patch.object(rc, "check_file", return_value=fake_result):
rc.scan_folder(str(tmp_path), recursive=False,
progress_callback=_progress)
# Attendo almeno "checking" per ciascun file + "completed" finale
assert any(e[2] == "completed" for e in events)
assert sum(1 for e in events if e[2] == "checking") >= 2