diff --git a/.env.example b/.env.example deleted file mode 100644 index 27f5a90..0000000 --- a/.env.example +++ /dev/null @@ -1,9 +0,0 @@ -# Copy to `.env` and fill in, then `source .env` before running. -# Get these from https://developer.spotify.com/dashboard (your app's settings). -# Register `http://127.0.0.1:8888` as a Redirect URI in that app. -export SPOTIFY_CLIENT_ID=your_client_id_here -export SPOTIFY_CLIENT_SECRET=your_client_secret_here - -# Optional overrides: -# export MUSIC_LIBRARY_PATH=~/Music -# export WORK_DIR=./work_dir/new diff --git a/.gitignore b/.gitignore index 7b75b45..880f9c3 100644 --- a/.gitignore +++ b/.gitignore @@ -2,8 +2,6 @@ prefetched* __pycache__ .idea secret -.env -.venv work_dir old tmp \ No newline at end of file diff --git a/Env.py b/Env.py index c609347..f717d8b 100644 --- a/Env.py +++ b/Env.py @@ -1,13 +1,7 @@ -import os from pathlib import Path -# Local config. Paths can be overridden via environment variables. -local_library_path = Path(os.environ.get("MUSIC_LIBRARY_PATH", "~/Music")).expanduser() -work_dir = Path(os.environ.get("WORK_DIR", "./work_dir/new")) +local_library_path = Path("/home/auser/Music/") +work_dir = Path("./work_dir/new") -# Spotify app credentials. NEVER hardcode these here — set them in the -# environment (see .env.example) so they don't end up in version control. -# export SPOTIFY_CLIENT_ID=... -# export SPOTIFY_CLIENT_SECRET=... -client_id = os.environ.get("SPOTIFY_CLIENT_ID") -client_secret = os.environ.get("SPOTIFY_CLIENT_SECRET") +client_id = '3b4db45bb66b45e9a2532989c8034332' +client_secret = "ed03dce79ec049e59b1bf20967aba418" diff --git a/Flow.py b/Flow.py index 9ced6ae..9cd7a04 100644 --- a/Flow.py +++ b/Flow.py @@ -26,7 +26,7 @@ class Flow: itunes_library : Library = None @staticmethod - def fetch_spotify(): + def fetch_spotify(self): fetch_all(client_id, client_secret, work_dir) @staticmethod diff --git a/README.md b/README.md index 62d8bd7..4a45af3 100644 --- a/README.md +++ b/README.md @@ -1,100 +1,3 @@ -# MusicIndexer - -Pull your Spotify library and listening stats via the Spotify Web API, index a -local music library (audio tags + iTunes XML), and match the two so you can see -which tracks you own, what's missing, and your top artists/tracks/playlists — in -a Streamlit GUI. - -## Features - -- **Spotify backend** — OAuth (Authorization Code) login, then fetch the user - profile, liked tracks, all playlists and their tracks, long-term top tracks - and top artists, and playlist/profile cover art. Resilient paginated fetching - (follows `next`, retries dropped connections, honours `429` rate limits) with - progress bars. -- **Local backend** — index a folder of audio files via `mutagen` tags. -- **iTunes backend** — parse an iTunes Library XML export. -- **Matching** — map local tracks to Spotify tracks (fuzzy matcher; ISRC/MBID - matching planned). -- **Stats & lyrics** — generate listening stats and fetch lyrics. -- **GUI** — browse libraries as artist → album → track tables. - -## Setup - -Requires Python 3.13+. - -```bash -python -m venv .venv -source .venv/bin/activate -pip install -r requirements.txt -``` - -### Credentials - -Create an app at https://developer.spotify.com/dashboard and add -`http://127.0.0.1:8888` as a **Redirect URI**. Credentials are read from the -environment — never commit them. - -```bash -cp .env.example .env -# edit .env with your client id/secret -source .env -``` - -`Env.py` reads `SPOTIFY_CLIENT_ID` / `SPOTIFY_CLIENT_SECRET` (plus optional -`MUSIC_LIBRARY_PATH` and `WORK_DIR`) from the environment. - -## Usage - -Edit `main.py` to uncomment the steps you want, then run: - -```bash -python main.py -``` - -The flow steps (see `Flow.py`) include: - -| Step | What it does | -|------|--------------| -| `fetch_spotify()` | OAuth login + download all Spotify data into `work_dir/spotify/` | -| `fetch_local()` | index local audio files into `work_dir/local/` | -| `parse_spotify_library()` / `parse_itunes_library()` / `parse_local_library()` | build normalized libraries | -| `load_libraries()` / `save_libraries()` | (de)serialize libraries to `work_dir` | -| `map_local_to_spotify()` | match local tracks against the Spotify library | -| `fetch_and_save_lyrics()` | fetch lyrics for tracks | -| `log_stats()` | generate stats | - -On first `fetch_spotify()` a browser opens for the Spotify login; a local server -on `127.0.0.1:8888` catches the redirect and exchanges the code for a token. - -### GUI - -```bash -./gui.sh # runs: .venv/bin/streamlit run Gui.py -``` - -## Layout - -``` -main.py entry point (CLI flow) -Gui.py / gui.sh Streamlit GUI -Flow.py orchestrates fetch / parse / map / stats -Env.py config (reads credentials from the environment) -src/ - Library.py data model (Library / Artist / Album / Track) - LibrarySerializer.py load/save libraries - backends/ - spotify/ Web API client, OAuth, parser - local/ local-file indexer, parser, playlist generator - itunes/ iTunes XML parser - navidrome/ Navidrome backend - mappers/ local↔Spotify matching (fuzzy / search) - lyrics/ lyrics fetching -work_dir/ fetched data and serialized libraries (gitignored) -``` - -## Notes - -- Data is cached as JSON under `work_dir/` (gitignored); re-running reuses it. -- The deprecated Spotify endpoints (audio-features, recommendations, etc.) are - not used — only currently-supported endpoints. +# spotifyFetcher +loads all data through web api
+client secret - 4ced4e4575094b749834a60996680c6c diff --git a/gui.sh b/gui.sh index 9be5f9d..eb55739 100755 --- a/gui.sh +++ b/gui.sh @@ -1,5 +1 @@ -#!/usr/bin/env bash -# Run from the script's directory using the project virtualenv, -# so it works regardless of whether the venv is activated / on PATH. -cd "$(dirname "$0")" -exec .venv/bin/streamlit run Gui.py "$@" +streamlit run Gui.py diff --git a/main.py b/main.py index 68707c7..26588d1 100644 --- a/main.py +++ b/main.py @@ -17,17 +17,17 @@ def test1(): def main(): flow = Flow.Flow() - flow.fetch_spotify() + # flow.fetch_spotify() # flow.fetch_local() flow.parse_spotify_library() - #flow.parse_itunes_library() - #flow.parse_local_library() + flow.parse_itunes_library() + flow.parse_local_library() - #flow.load_libraries() + flow.load_libraries() - #flow.map_local_to_spotify() - #flow.save_libraries() + flow.map_local_to_spotify() + flow.save_libraries() # flow.log_stats() diff --git a/requirements.txt b/requirements.txt deleted file mode 100644 index 9a1d950..0000000 --- a/requirements.txt +++ /dev/null @@ -1,9 +0,0 @@ -streamlit==1.58.0 -streamlit-aggrid==1.2.1.post2 -pandas==3.0.3 -Pillow==12.2.0 -mutagen==1.47.0 -rapidfuzz==3.14.5 -requests==2.34.2 -tqdm==4.68.2 -xmltodict==1.0.4 diff --git a/src/backends/spotify/SpotifyWebAPI.py b/src/backends/spotify/SpotifyWebAPI.py index e795d8a..73fe61b 100644 --- a/src/backends/spotify/SpotifyWebAPI.py +++ b/src/backends/spotify/SpotifyWebAPI.py @@ -1,10 +1,7 @@ import requests import urllib.parse -from requests.adapters import HTTPAdapter -from urllib3.util.retry import Retry from PIL import Image from io import BytesIO -from tqdm import tqdm from src.backends.spotify.SpotifyAuthenticator import authenticate @@ -12,31 +9,6 @@ from src.helpers import * FETCH_STEP = 50 token = '' -REQUEST_TIMEOUT = 30 - - -def _build_session(): - """Session that retries transient failures (dropped connections, 5xx) and - honours Spotify's 429 Retry-After, so a single reset mid-fetch doesn't - abort the whole run.""" - session = requests.Session() - retry = Retry( - total=6, - connect=6, - read=6, - backoff_factor=1.5, # sleeps ~0, 1.5, 3, 6, 12, 24s between attempts - status_forcelist=(429, 500, 502, 503, 504), - allowed_methods=frozenset({"GET", "POST"}), - respect_retry_after_header=True, - raise_on_status=False, - ) - adapter = HTTPAdapter(max_retries=retry) - session.mount("https://", adapter) - session.mount("http://", adapter) - return session - - -_session = _build_session() def fetch(endpoint, method='GET', body=None): @@ -45,27 +17,23 @@ def fetch(endpoint, method='GET', body=None): 'Authorization': f'Bearer {token}', } - response = _session.request(method, url, headers=headers, json=body, timeout=REQUEST_TIMEOUT) + response = requests.request(method, url, headers=headers, json=body) response.raise_for_status() return response.json() -def fetch_paginated(endpoint, desc="Fetching", leave=True): +def fetch_paginated(endpoint): """Collect every item of a paging object by following its `next` field, which is Spotify's recommended way to page (more robust than tracking - offset/total manually). Shows a tqdm progress bar driven by the paging - object's `total`.""" + offset/total manually).""" items = [] page = fetch(endpoint) - with tqdm(total=page.get('total'), desc=desc, unit="item", leave=leave) as pbar: - while True: - batch = page.get('items', []) - items += batch - pbar.update(len(batch)) - next_url = page.get('next') - if not next_url: - break - page = fetch(next_url) + while True: + items += page.get('items', []) + next_url = page.get('next') + if not next_url: + break + page = fetch(next_url) return items @@ -75,31 +43,31 @@ def find_song(song_pattern): def fetch_playlists(user_id): - return fetch_paginated(f"v1/me/playlists?limit={FETCH_STEP}", desc="Fetching playlists") + print("Fetching playlists") + return fetch_paginated(f"v1/me/playlists?limit={FETCH_STEP}") def fetch_playlist_items(playlist_id): - return fetch_paginated( - f"v1/playlists/{playlist_id}/tracks?limit={FETCH_STEP}", - desc="Fetching playlist tracks", - leave=False, - ) + return fetch_paginated(f"v1/playlists/{playlist_id}/tracks?limit={FETCH_STEP}") def fetch_tracks(user_id): - tracks = fetch_paginated(f"v1/me/tracks?limit={FETCH_STEP}&market=ES", desc="Fetching liked tracks") + print("Fetching tracks") + tracks = fetch_paginated(f"v1/me/tracks?limit={FETCH_STEP}&market=ES") save_data(tracks, "tracks") return tracks def fetch_tracks_top(): - tracks = fetch_paginated(f"v1/me/top/tracks?limit={FETCH_STEP}&time_range=long_term", desc="Fetching top tracks") + print("Fetching top tracks") + tracks = fetch_paginated(f"v1/me/top/tracks?limit={FETCH_STEP}&time_range=long_term") save_data(tracks, "top_tracks") return tracks def fetch_artists_top(): - artists = fetch_paginated(f"v1/me/top/artists?limit={FETCH_STEP}&time_range=long_term", desc="Fetching top artists") + print("Fetching top artists") + artists = fetch_paginated(f"v1/me/top/artists?limit={FETCH_STEP}&time_range=long_term") save_data(artists, "top_artists") return artists