traxsource: _discover_top100_url + _parse_tracks con BeautifulSoup
This commit is contained in:
1 parent
e1e629f8a2
commit
784b56682f
2 files changed
+126
No files matched your search
@@ -104,3 +104,88 @@ def _large_cover(url: str) -> str:
|
|||||||
if not url:
|
if not url:
|
||||||
return ""
|
return ""
|
||||||
return _SIZE_RE.sub("/500x500/", url)
|
return _SIZE_RE.sub("/500x500/", url)
|
||||||
|
|
||||||
|
|
||||||
|
_TOP100_LINK_RE = re.compile(r'href="(/title/\d+/top-100-[a-z0-9-]+)"')
|
||||||
|
|
||||||
|
|
||||||
|
def _discover_top100_url(genre_html: str) -> str:
|
||||||
|
"""Estrae il path relativo della playlist Top 100 corrente dalla pagina di un genere."""
|
||||||
|
m = _TOP100_LINK_RE.search(genre_html)
|
||||||
|
if not m:
|
||||||
|
raise TraxsourceParseError("link Top 100 non trovato nella pagina genere")
|
||||||
|
return m.group(1)
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_tracks(html: str) -> list:
|
||||||
|
"""Parsa la pagina Top 100 (title playlist) e ritorna list[TraxsourceTrack].
|
||||||
|
|
||||||
|
Selettori (verificati su fixture tech-house 2026-07):
|
||||||
|
row = div.trk-row.play-trk (data-trid=<int>)
|
||||||
|
position = div.tnum (inside div.tnum-pos)
|
||||||
|
title <a> = div.trk-cell.title a[href^="/track/"]
|
||||||
|
version = span.version (contiene child span.duration da rimuovere)
|
||||||
|
artists = a.com-artists (uno o piu)
|
||||||
|
label <a> = div.trk-cell.label a
|
||||||
|
cover img = div.trk-cell.thumb img (src /scripts/image.php/52x52/...)
|
||||||
|
"""
|
||||||
|
# Lazy import — bs4 non e' hard-dep del modulo (importato solo quando serve).
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
|
|
||||||
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
|
rows = soup.select("div.trk-row.play-trk")
|
||||||
|
if not rows:
|
||||||
|
raise TraxsourceParseError("nessuna track (div.trk-row.play-trk) trovata")
|
||||||
|
|
||||||
|
out: list = []
|
||||||
|
for i, row in enumerate(rows, 1):
|
||||||
|
try:
|
||||||
|
trid = int(row.get("data-trid") or 0)
|
||||||
|
|
||||||
|
pos_el = row.select_one("div.tnum")
|
||||||
|
position = i # fallback su enumerate se pos manca / non e' un numero
|
||||||
|
if pos_el:
|
||||||
|
pos_txt = pos_el.get_text(strip=True)
|
||||||
|
if pos_txt.isdigit():
|
||||||
|
position = int(pos_txt)
|
||||||
|
|
||||||
|
title_a = row.select_one('div.trk-cell.title a[href^="/track/"]')
|
||||||
|
if not title_a:
|
||||||
|
continue
|
||||||
|
title = title_a.get_text(strip=True)
|
||||||
|
href = title_a.get("href") or ""
|
||||||
|
slug = href.rsplit("/", 1)[-1]
|
||||||
|
|
||||||
|
# Mix version: contenuto di span.version, escluso span.duration
|
||||||
|
mix = ""
|
||||||
|
version_el = row.select_one("span.version")
|
||||||
|
if version_el:
|
||||||
|
dur_el = version_el.select_one("span.duration")
|
||||||
|
if dur_el:
|
||||||
|
dur_el.extract()
|
||||||
|
mix = version_el.get_text(strip=True)
|
||||||
|
|
||||||
|
artist_names = [a.get_text(strip=True) for a in row.select("a.com-artists")]
|
||||||
|
artists = _format_artists(artist_names)
|
||||||
|
|
||||||
|
label_a = row.select_one("div.trk-cell.label a")
|
||||||
|
label = label_a.get_text(strip=True) if label_a else ""
|
||||||
|
|
||||||
|
img = row.select_one('div.trk-cell.thumb img[src*="/scripts/image.php/"]')
|
||||||
|
image_url = (img.get("src") or "") if img else ""
|
||||||
|
cover_large = _large_cover(image_url)
|
||||||
|
|
||||||
|
out.append(TraxsourceTrack(
|
||||||
|
position=position,
|
||||||
|
title=title,
|
||||||
|
mix=mix,
|
||||||
|
artists=artists,
|
||||||
|
label=label,
|
||||||
|
traxsource_id=trid,
|
||||||
|
slug=slug,
|
||||||
|
image_url=image_url,
|
||||||
|
cover_url_large=cover_large,
|
||||||
|
))
|
||||||
|
except Exception as e:
|
||||||
|
raise TraxsourceParseError(f"errore parse track[{i}]: {e}") from e
|
||||||
|
return out
|
||||||
@@ -95,3 +95,44 @@ class TestExceptions:
|
|||||||
def test_exceptions_are_subclasses(self):
|
def test_exceptions_are_subclasses(self):
|
||||||
assert issubclass(traxsource.TraxsourceUnreachableError, traxsource.TraxsourceError)
|
assert issubclass(traxsource.TraxsourceUnreachableError, traxsource.TraxsourceError)
|
||||||
assert issubclass(traxsource.TraxsourceParseError, traxsource.TraxsourceError)
|
assert issubclass(traxsource.TraxsourceParseError, traxsource.TraxsourceError)
|
||||||
|
|
||||||
|
|
||||||
|
class TestDiscoverTop100Url:
|
||||||
|
def test_extracts_link_from_genre_page(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "traxsource_tech_house_genre.html").read_text()
|
||||||
|
url = traxsource._discover_top100_url(html)
|
||||||
|
assert url.startswith("/title/")
|
||||||
|
assert "top-100" in url
|
||||||
|
|
||||||
|
def test_raises_when_no_link(self):
|
||||||
|
with pytest.raises(traxsource.TraxsourceParseError, match="Top 100"):
|
||||||
|
traxsource._discover_top100_url("<html>nulla</html>")
|
||||||
|
|
||||||
|
|
||||||
|
class TestParseTracks:
|
||||||
|
def test_parses_100_tracks(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "traxsource_tech_house_top100.html").read_text()
|
||||||
|
tracks = traxsource._parse_tracks(html)
|
||||||
|
assert len(tracks) == 100
|
||||||
|
|
||||||
|
def test_positions_sequential_1_to_100(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "traxsource_tech_house_top100.html").read_text()
|
||||||
|
tracks = traxsource._parse_tracks(html)
|
||||||
|
positions = [t.position for t in tracks]
|
||||||
|
assert positions == list(range(1, 101))
|
||||||
|
|
||||||
|
def test_track_shape(self, fixtures_dir):
|
||||||
|
html = (fixtures_dir / "traxsource_tech_house_top100.html").read_text()
|
||||||
|
tracks = traxsource._parse_tracks(html)
|
||||||
|
first = tracks[0]
|
||||||
|
assert first.title
|
||||||
|
assert first.artists
|
||||||
|
assert first.traxsource_id > 0
|
||||||
|
assert first.slug
|
||||||
|
assert first.image_url.startswith("https://")
|
||||||
|
assert first.cover_url_large.startswith("https://")
|
||||||
|
assert "500x500" in first.cover_url_large
|
||||||
|
|
||||||
|
def test_raises_when_no_tracks(self):
|
||||||
|
with pytest.raises(traxsource.TraxsourceParseError, match="track"):
|
||||||
|
traxsource._parse_tracks("<html>vuoto</html>")
|
||||||
Reference in new issue
Block a user