spotify: search_artist_discography con dedupe top-tracks+album

This commit is contained in:
luciano committed 2026-07-14 17:16:19 +02:00
1 parent 252d76b907
commit 2771fa8075
2 files changed
+274

No files matched your search

+88
View File
@@ -3,6 +3,8 @@
from __future__ import annotations
import re
import time
import requests
@@ -314,3 +316,89 @@ def _track_to_dict(t: dict) -> dict:
"album": t.get("album", {}).get("name", ""),
"duration_sec": int(t.get("duration_ms", 0)) // 1000,
}
def search_artist_discography(token: str, artist_name: str) -> list:
"""Trova l'artista esatto (o il piu' popolare tra i match) e ritorna
tutti i suoi brani: top tracks + tracce di ogni album/single.
Deduplica per (name.lower().strip(), first_artist.lower().strip()).
Raises:
ValueError: se nessun artista trovato per il nome dato.
"""
headers = {"Authorization": f"Bearer {token}"}
# 1. Cerca artista
resp = requests.get(
"https://api.spotify.com/v1/search",
headers=headers,
params={"q": artist_name, "type": "artist", "limit": 5},
timeout=15,
)
resp.raise_for_status()
candidates = resp.json().get("artists", {}).get("items", [])
if not candidates:
raise ValueError(f"Artista '{artist_name}' non trovato")
# Match esatto (case-insensitive) se possibile, altrimenti piu' popolare
query_lower = artist_name.lower().strip()
exact = [c for c in candidates if c.get("name", "").lower().strip() == query_lower]
if exact:
artist = exact[0]
else:
artist = max(candidates, key=lambda c: c.get("popularity", 0))
artist_id = artist["id"]
collected: list = []
# 2. Top tracks
r_top = requests.get(
f"https://api.spotify.com/v1/artists/{artist_id}/top-tracks",
headers=headers,
params={"market": "IT"},
timeout=15,
)
r_top.raise_for_status()
for t in r_top.json().get("tracks", []):
collected.append(_track_to_dict(t))
# 3. Albums (album + single)
r_alb = requests.get(
f"https://api.spotify.com/v1/artists/{artist_id}/albums",
headers=headers,
params={"include_groups": "album,single", "limit": 50, "market": "IT"},
timeout=15,
)
r_alb.raise_for_status()
albums = r_alb.json().get("items", [])
# 4. Per ogni album, tracce (album/track object non ha "album" sub-field, iniettiamola)
for alb in albums:
alb_id = alb.get("id")
alb_name = alb.get("name", "")
if not alb_id:
continue
time.sleep(0.1) # rate limit interno
r_at = requests.get(
f"https://api.spotify.com/v1/albums/{alb_id}/tracks",
headers=headers,
params={"limit": 50},
timeout=15,
)
r_at.raise_for_status()
for t in r_at.json().get("items", []):
t = dict(t)
# Album tracks non hanno "album" nested; iniettiamo il nome
t.setdefault("album", {"name": alb_name})
collected.append(_track_to_dict(t))
# 5. Dedupe
seen: set = set()
unique: list = []
for t in collected:
key = (t["name"].lower().strip(), t["artists"].split(",")[0].lower().strip())
if key in seen:
continue
seen.add(key)
unique.append(t)
return unique