beatport: parse JSON → list[BeatportTrack]
This commit is contained in:
1 parent
f1e57905c1
commit
13c8e60712
2 files changed
+84
No files matched your search
@@ -110,3 +110,59 @@ def list_genres() -> list:
|
|||||||
]
|
]
|
||||||
result.sort(key=lambda g: g["name"].casefold())
|
result.sort(key=lambda g: g["name"].casefold())
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def _find_tracks_results(data: dict) -> list:
|
||||||
|
"""Cerca dentro le queries dehydrated il primo `results` che ha almeno 50 elementi
|
||||||
|
e la shape di una track (chiave `id` presente)."""
|
||||||
|
try:
|
||||||
|
queries = data["props"]["pageProps"]["dehydratedState"]["queries"]
|
||||||
|
except (KeyError, TypeError) as e:
|
||||||
|
raise BeatportParseError(
|
||||||
|
f"schema JSON inatteso, `results` non localizzabile (chiave mancante: {e})"
|
||||||
|
) from e
|
||||||
|
|
||||||
|
for q in queries:
|
||||||
|
state_data = (q or {}).get("state", {}).get("data")
|
||||||
|
if not isinstance(state_data, dict):
|
||||||
|
continue
|
||||||
|
results = state_data.get("results")
|
||||||
|
if isinstance(results, list) and len(results) >= 50:
|
||||||
|
if results and isinstance(results[0], dict) and "id" in results[0]:
|
||||||
|
return results
|
||||||
|
raise BeatportParseError("nessun `results` di 50+ track trovato in __NEXT_DATA__")
|
||||||
|
|
||||||
|
|
||||||
|
def _format_artists(artists_field: object) -> str:
|
||||||
|
"""Beatport ritorna artists come lista di dict {name, ...}.
|
||||||
|
Formatta come 'A, B & C' (& prima dell'ultimo)."""
|
||||||
|
if not isinstance(artists_field, list) or not artists_field:
|
||||||
|
return ""
|
||||||
|
names = [a.get("name", "") for a in artists_field if isinstance(a, dict)]
|
||||||
|
names = [n for n in names if n]
|
||||||
|
if not names:
|
||||||
|
return ""
|
||||||
|
if len(names) == 1:
|
||||||
|
return names[0]
|
||||||
|
return ", ".join(names[:-1]) + " & " + names[-1]
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_tracks(data: dict) -> list:
|
||||||
|
"""Trasforma i track dict di Beatport in BeatportTrack ordinati per posizione."""
|
||||||
|
raw = _find_tracks_results(data)
|
||||||
|
out: list = []
|
||||||
|
for i, item in enumerate(raw, 1):
|
||||||
|
try:
|
||||||
|
length_ms = int(item.get("length_ms") or 0)
|
||||||
|
track = BeatportTrack(
|
||||||
|
position=i,
|
||||||
|
title=str(item.get("name") or "").strip(),
|
||||||
|
mix=str(item.get("mix_name") or "").strip(),
|
||||||
|
artists=_format_artists(item.get("artists")),
|
||||||
|
duration_sec=length_ms // 1000,
|
||||||
|
beatport_id=int(item.get("id") or 0),
|
||||||
|
)
|
||||||
|
except (TypeError, ValueError) as e:
|
||||||
|
raise BeatportParseError(f"track[{i}] shape inattesa: {e}") from e
|
||||||
|
out.append(track)
|
||||||
|
return out
|
||||||
@@ -83,3 +83,31 @@ class TestExtractNextData:
|
|||||||
broken = '<script id="__NEXT_DATA__" type="application/json">{not: valid}</script>'
|
broken = '<script id="__NEXT_DATA__" type="application/json">{not: valid}</script>'
|
||||||
with pytest.raises(beatport.BeatportParseError, match="JSON malformato"):
|
with pytest.raises(beatport.BeatportParseError, match="JSON malformato"):
|
||||||
beatport._extract_next_data(broken)
|
beatport._extract_next_data(broken)
|
||||||
|
|
||||||
|
|
||||||
|
class TestParseTracks:
|
||||||
|
def test_extracts_100_tracks(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "beatport_melodic_top100.html").read_text()
|
||||||
|
data = beatport._extract_next_data(html)
|
||||||
|
tracks = beatport._parse_tracks(data)
|
||||||
|
assert len(tracks) == 100
|
||||||
|
|
||||||
|
def test_positions_are_sequential_1_to_100(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "beatport_melodic_top100.html").read_text()
|
||||||
|
tracks = beatport._parse_tracks(beatport._extract_next_data(html))
|
||||||
|
positions = [t.position for t in tracks]
|
||||||
|
assert positions == list(range(1, 101))
|
||||||
|
|
||||||
|
def test_track_shape(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "beatport_melodic_top100.html").read_text()
|
||||||
|
tracks = beatport._parse_tracks(beatport._extract_next_data(html))
|
||||||
|
first = tracks[0]
|
||||||
|
assert isinstance(first, beatport.BeatportTrack)
|
||||||
|
assert first.title
|
||||||
|
assert first.artists
|
||||||
|
assert first.duration_sec > 0
|
||||||
|
assert first.beatport_id > 0
|
||||||
|
|
||||||
|
def test_schema_missing_results_raises(self):
|
||||||
|
with pytest.raises(beatport.BeatportParseError, match="results"):
|
||||||
|
beatport._parse_tracks({"props": {"pageProps": {}}})
|
||||||
Reference in new issue
Block a user