diff --git a/core/traxsource.py b/core/traxsource.py index 86bcaee..f49af28 100644 --- a/core/traxsource.py +++ b/core/traxsource.py @@ -7,6 +7,7 @@ Spec: docs/superpowers/specs/2026-08-01-traxsource-charts-design.md. from __future__ import annotations import re +import time from dataclasses import dataclass @@ -189,3 +190,88 @@ def _parse_tracks(html: str) -> list: except Exception as e: raise TraxsourceParseError(f"errore parse track[{i}]: {e}") from e return out + + +# --- fetch_top100 con session curl_cffi + retry + cache ------------------ + +_IMPERSONATE = "chrome131" +_REQUEST_TIMEOUT = 15 +_MAX_ATTEMPTS = 3 +_BACKOFF_SEC = [1, 3] +_CACHE_TTL_SEC = 15 * 60 + +_cache: dict = {} +_session_singleton = None + + +def _session(): + """Ritorna la Session curl_cffi singleton, preriscaldata con GET a /.""" + global _session_singleton + if _session_singleton is None: + from curl_cffi import requests as _cffi + _session_singleton = _cffi.Session(impersonate=_IMPERSONATE) + try: + _session_singleton.get("https://www.traxsource.com/", timeout=_REQUEST_TIMEOUT) + except Exception: + pass # cookie CF possono arrivare comunque + return _session_singleton + + +def _do_get(session, url: str) -> str: + """GET con retry + backoff. Include Referer per pagine interne.""" + last_exc = None + for attempt in range(_MAX_ATTEMPTS): + try: + resp = session.get( + url, + timeout=_REQUEST_TIMEOUT, + headers={"Referer": "https://www.traxsource.com/"}, + ) + if resp.status_code >= 500 or resp.status_code == 403: + raise Exception(f"HTTP {resp.status_code}") + resp.raise_for_status() + return resp.text + except Exception as e: + last_exc = e + if attempt < _MAX_ATTEMPTS - 1: + time.sleep(_BACKOFF_SEC[attempt]) + raise TraxsourceUnreachableError( + f"Traxsource irraggiungibile dopo {_MAX_ATTEMPTS} tentativi: {last_exc}" + ) + + +def fetch_top100(slug: str, force_refresh: bool = False) -> list: + """Fetches Top 100 corrente per il genere. + + 1. Verifica slug in GENRES + 2. Fetch pagina genre -> _discover_top100_url + 3. Fetch pagina Top 100 -> _parse_tracks + 4. Cache 15 min + + Raises: + ValueError: slug non in GENRES + TraxsourceUnreachableError: rete/5xx dopo retry + TraxsourceParseError: HTML non conforme + """ + if slug not in GENRES: + raise ValueError(f"slug genere non valido: {slug!r}") + + now = time.time() + if not force_refresh: + cached = _cache.get(slug) + if cached and (now - cached[0]) < _CACHE_TTL_SEC: + return cached[1] + + gid, _name = GENRES[slug] + sess = _session() + + genre_url = f"https://www.traxsource.com/genre/{gid}/{slug}" + genre_html = _do_get(sess, genre_url) + top100_path = _discover_top100_url(genre_html) + + top100_url = "https://www.traxsource.com" + top100_path + top100_html = _do_get(sess, top100_url) + tracks = _parse_tracks(top100_html) + + _cache[slug] = (now, tracks) + return tracks diff --git a/tests/test_traxsource.py b/tests/test_traxsource.py index 1c8ecf8..f1fdd98 100644 --- a/tests/test_traxsource.py +++ b/tests/test_traxsource.py @@ -136,3 +136,82 @@ class TestParseTracks: def test_raises_when_no_tracks(self): with pytest.raises(traxsource.TraxsourceParseError, match="track"): traxsource._parse_tracks("vuoto") + + +from unittest.mock import patch, MagicMock + + +def _mock_response(text: str, status_code: int = 200) -> MagicMock: + resp = MagicMock() + resp.text = text + resp.status_code = status_code + + def _raise(): + if status_code >= 400: + raise Exception(f"HTTP {status_code}") + resp.raise_for_status = _raise + return resp + + +class TestFetchTop100: + @pytest.fixture + def genre_html(self, fixtures_dir): + return (fixtures_dir / "traxsource_tech_house_genre.html").read_text() + + @pytest.fixture + def top100_html(self, fixtures_dir): + return (fixtures_dir / "traxsource_tech_house_top100.html").read_text() + + def test_success_returns_100_tracks(self, genre_html, top100_html): + traxsource._cache.clear() + mock_sess = MagicMock() + mock_sess.get.side_effect = [ + _mock_response(genre_html, 200), + _mock_response(top100_html, 200), + ] + with patch("core.traxsource._session", return_value=mock_sess): + tracks = traxsource.fetch_top100("tech-house") + assert len(tracks) == 100 + + def test_invalid_slug_raises_value_error(self): + with pytest.raises(ValueError, match="slug"): + traxsource.fetch_top100("not-a-genre") + + def test_5xx_retries_and_raises_unreachable(self): + traxsource._cache.clear() + mock_sess = MagicMock() + mock_sess.get.return_value = _mock_response("", 503) + with patch("core.traxsource._session", return_value=mock_sess), \ + patch("core.traxsource.time.sleep"): + with pytest.raises(traxsource.TraxsourceUnreachableError): + traxsource.fetch_top100("tech-house") + # 3 tentativi + assert mock_sess.get.call_count == 3 + + def test_cache_hit_within_ttl(self, genre_html, top100_html): + traxsource._cache.clear() + mock_sess = MagicMock() + mock_sess.get.side_effect = [ + _mock_response(genre_html, 200), + _mock_response(top100_html, 200), + ] + with patch("core.traxsource._session", return_value=mock_sess): + traxsource.fetch_top100("tech-house") + traxsource.fetch_top100("tech-house") + # 2 chiamate al primo fetch (genre + top100), 0 al secondo + assert mock_sess.get.call_count == 2 + + def test_force_refresh_bypasses_cache(self, genre_html, top100_html): + traxsource._cache.clear() + mock_sess = MagicMock() + # 4 risposte (2 fetch x 2 richieste ciascuno) + mock_sess.get.side_effect = [ + _mock_response(genre_html, 200), + _mock_response(top100_html, 200), + _mock_response(genre_html, 200), + _mock_response(top100_html, 200), + ] + with patch("core.traxsource._session", return_value=mock_sess): + traxsource.fetch_top100("tech-house") + traxsource.fetch_top100("tech-house", force_refresh=True) + assert mock_sess.get.call_count == 4