"""Verifica royalty-free status di brani locali contro DB CC/open. Approccio onesto: whitelist royalty-free. L'app segnala un brano come "royalty_free" SOLO se trova un match affidabile in uno dei DB CC/open. "Non trovato" (status 'unknown') NON implica automaticamente SIAE — significa semplicemente "da verificare manualmente". Pipeline: 1. Scansiona la cartella (filtra per `AUDIO_EXTENSIONS` di `core.upgrader`). 2. Per ogni file legge i tag ID3 (artist + title) via `core.metadata.read_metadata`; fallback a `core.catalog._parse_filename` se i tag sono mancanti. 3. Query sorgenti royalty-free: - ccMixter (nessuna auth, sempre attivo) - Jamendo (richiede `jamendo_client_id` in config; se vuoto, skip) 4. Levenshtein combinato artist+title >= `_SIMILARITY_THRESHOLD` → status 'royalty_free'. Altrimenti 'unknown'. Progress callback firma: (processed, total, filename, status[, err_msg]) Status: 'checking' | 'stopped' | 'completed' | 'error'. `entry_callback(entry)` viene chiamato per ogni file processato (streaming alla UI). """ from __future__ import annotations import json import sqlite3 import threading import time from difflib import SequenceMatcher from pathlib import Path from typing import Callable, Optional import requests from core.upgrader import AUDIO_EXTENSIONS # ------------------------------------------------------------------ # Costanti # ------------------------------------------------------------------ _CCMIXTER_URL = "https://ccmixter.org/api/query" _JAMENDO_URL = "https://api.jamendo.com/v3.0/tracks/" _HTTP_TIMEOUT = 15 _RATE_LIMIT_SEC = 0.3 _SIMILARITY_THRESHOLD = 0.80 # ------------------------------------------------------------------ # Stop / interrupt # ------------------------------------------------------------------ _stop_event = threading.Event() def request_stop() -> None: """Segnala al worker di interrompere la scansione al prossimo file.""" _stop_event.set() def reset_stop() -> None: """Azzera il flag di stop prima di iniziare una nuova scansione.""" _stop_event.clear() def is_stopped() -> bool: return _stop_event.is_set() # ------------------------------------------------------------------ # Cache SQLite # ------------------------------------------------------------------ def _cache_db_path() -> Path: """Path del file di cache SQLite (patchable nei test).""" from core.config import _get_config_dir return _get_config_dir() / "royalty_cache.db" def _init_db(conn: sqlite3.Connection) -> None: conn.executescript( """ CREATE TABLE IF NOT EXISTS lookups( key TEXT PRIMARY KEY, payload TEXT, cached_at REAL ); """ ) # ------------------------------------------------------------------ # Similarity # ------------------------------------------------------------------ def _similarity(a: str, b: str) -> float: """Ritorna SequenceMatcher ratio (0.0 - 1.0) case-insensitive.""" if not a or not b: return 0.0 return SequenceMatcher(None, a.lower().strip(), b.lower().strip()).ratio() # ------------------------------------------------------------------ # Rate limit (globale) # ------------------------------------------------------------------ _last_req = [0.0] def _throttle() -> None: now = time.monotonic() elapsed = now - _last_req[0] if elapsed < _RATE_LIMIT_SEC: time.sleep(_RATE_LIMIT_SEC - elapsed) _last_req[0] = time.monotonic() # ------------------------------------------------------------------ # Query sorgenti # ------------------------------------------------------------------ def _query_ccmixter(query: str) -> list: """Query ccMixter API. Ritorna lista di candidati normalizzati.""" if not query: return [] _throttle() try: resp = requests.get( _CCMIXTER_URL, params={"f": "json", "search": query, "limit": 10}, timeout=_HTTP_TIMEOUT, ) if resp.status_code != 200: return [] data = resp.json() except Exception: return [] if not isinstance(data, list): return [] out = [] for item in data: if not isinstance(item, dict): continue out.append({ "source": "ccmixter", "title": item.get("upload_name", "") or "", "artist": (item.get("user_real_name") or item.get("user_name", "") or ""), "license": item.get("license_name", "") or "", "license_url": item.get("license_url", "") or "", "url": item.get("file_page_url", "") or "", }) return out def _query_jamendo(query: str, client_id: str) -> list: """Query Jamendo API. Senza client_id ritorna [].""" if not client_id or not query: return [] _throttle() try: resp = requests.get( _JAMENDO_URL, params={ "client_id": client_id, "format": "json", "limit": 5, "search": query, }, timeout=_HTTP_TIMEOUT, ) if resp.status_code != 200: return [] data = resp.json() except Exception: return [] if not isinstance(data, dict): return [] status = (data.get("headers", {}) or {}).get("status", "") if status and status != "success": return [] out = [] for item in data.get("results", []) or []: if not isinstance(item, dict): continue out.append({ "source": "jamendo", "title": item.get("name", "") or "", "artist": item.get("artist_name", "") or "", "license": item.get("license_name") or "Creative Commons", "license_url": item.get("license_ccurl", "") or "", "url": item.get("shareurl", "") or "", }) return out # ------------------------------------------------------------------ # Check singolo file # ------------------------------------------------------------------ def check_file(path: str, jamendo_client_id: str = "") -> dict: """Verifica un singolo file. Ritorna: {path, status, matches, query, artist, title[, error]} status: 'royalty_free' | 'unknown' | 'error'. """ from core.metadata import read_metadata from core.catalog import _parse_filename try: md = read_metadata(path) except Exception as e: return { "path": path, "status": "error", "matches": [], "query": "", "artist": "", "title": "", "error": str(e), } artist = (md.get("artist") or "").strip() title = (md.get("title") or "").strip() if not artist or not title: parsed_artist, parsed_title = _parse_filename(Path(path).stem) artist = artist or parsed_artist title = title or parsed_title query = f"{artist} {title}".strip() if artist else title if not query: return { "path": path, "status": "unknown", "matches": [], "query": "", "artist": artist, "title": title, "error": "no query (tag + filename vuoti)", } # Cache lookup conn = sqlite3.connect(str(_cache_db_path())) try: _init_db(conn) row = conn.execute( "SELECT payload FROM lookups WHERE key=?", (query,) ).fetchone() if row: try: cached = json.loads(row[0]) cached["path"] = path # Preserva artist/title correnti nel caso siano stati # ricavati dal filename ma la cache li aveva salvati gia' return cached except Exception: pass candidates = _query_ccmixter(query) candidates += _query_jamendo(query, jamendo_client_id) matches = [] for c in candidates: score_title = _similarity(title, c.get("title", "")) score_artist = ( _similarity(artist, c.get("artist", "")) if artist and c.get("artist") else 0.0 ) if artist: combined = score_title * 0.6 + score_artist * 0.4 else: combined = score_title if combined >= _SIMILARITY_THRESHOLD: enriched = dict(c) enriched["score"] = round(combined, 2) matches.append(enriched) # Ordina per score desc matches.sort(key=lambda m: m.get("score", 0.0), reverse=True) status = "royalty_free" if matches else "unknown" result = { "path": path, "status": status, "matches": matches, "query": query, "artist": artist, "title": title, } # Scrivi in cache (senza il path, che e' per-file) try: payload = {k: v for k, v in result.items() if k != "path"} conn.execute( "INSERT OR REPLACE INTO lookups(key, payload, cached_at) " "VALUES (?, ?, ?)", (query, json.dumps(payload, ensure_ascii=False), time.time()), ) conn.commit() except Exception: pass return result finally: try: conn.close() except Exception: pass # ------------------------------------------------------------------ # Scan folder # ------------------------------------------------------------------ def _emit_progress(cb: Optional[Callable], idx: int, total: int, name: str, status: str, err: str = "") -> None: """Chiama progress_callback in modo retrocompatibile (4 o 5 args).""" if not cb: return try: cb(idx, total, name, status, err) except TypeError: try: cb(idx, total, name, status) except Exception: pass except Exception: pass def scan_folder( directory: str, recursive: bool = True, progress_callback: Optional[Callable] = None, entry_callback: Optional[Callable] = None, ) -> list: """Scansiona la cartella e verifica ogni file audio. Ritorna la lista dei risultati per file (vedi `check_file`). `entry_callback(entry)` e' chiamato appena ogni file e' processato. """ reset_stop() base = Path(directory) if not base.exists() or not base.is_dir(): _emit_progress(progress_callback, 0, 0, "", "completed", "") return [] files: list = [] iterator = base.rglob("*") if recursive else base.iterdir() for f in iterator: try: if f.is_file() and f.suffix.lower() in AUDIO_EXTENSIONS: files.append(f) except OSError: continue files.sort() total = len(files) if total == 0: _emit_progress(progress_callback, 0, 0, "", "completed", "") return [] # Jamendo client_id (opzionale) jamendo_id = "" try: from core.config import load_config jamendo_id = (load_config().get("jamendo_client_id") or "").strip() except Exception: jamendo_id = "" results: list = [] for i, fp in enumerate(files, start=1): if is_stopped(): _emit_progress(progress_callback, i - 1, total, "", "stopped", "") return results _emit_progress(progress_callback, i, total, fp.name, "checking", "") try: r = check_file(str(fp), jamendo_client_id=jamendo_id) except Exception as e: r = { "path": str(fp), "status": "error", "matches": [], "query": "", "artist": "", "title": "", "error": str(e), } _emit_progress(progress_callback, i, total, fp.name, "error", str(e)) results.append(r) if entry_callback: try: entry_callback(r) except Exception: pass if not is_stopped(): _emit_progress(progress_callback, total, total, "", "completed", "") return results