Files
2026-05-17 19:15:03 +02:00

449 lines
16 KiB
Python
Executable File

#!/usr/bin/env python3
"""
bandcamp-sync — Download new Bandcamp purchases (WAV) to the local music library,
then invoke bandcamp-tag.py to fill in metadata.
Usage:
bandcamp-sync.py [MUSIC_DIR] # sync new purchases to MUSIC_DIR
bandcamp-sync.py --dry-run # show what would be downloaded, no writes
bandcamp-sync.py --list # list full collection, no download
Credentials (first match wins):
1. Env vars BANDCAMP_EMAIL / BANDCAMP_PASSWORD
2. bandcamp-api service at http://localhost:8091/credentials
3. JSON file ~/.config/audioserver/bandcamp.json { "email": ..., "password": ... }
"""
import argparse
import html
import json
import os
import re
import shutil
import subprocess
import sys
import time
import zipfile
from pathlib import Path
from urllib.parse import unquote
import requests
# ── Config ────────────────────────────────────────────────────────────────────
DEFAULT_MUSIC_DIR = Path("/mnt/local/music")
SCRIPT_DIR = Path(__file__).parent
TAGGER = SCRIPT_DIR / "bandcamp-tag.py"
REQUEST_DELAY = 1.0 # seconds between Bandcamp requests
COLLECTION_BATCH = 50
SESSION = requests.Session()
SESSION.headers["User-Agent"] = (
"Mozilla/5.0 (X11; Linux aarch64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
)
# ── Credentials ───────────────────────────────────────────────────────────────
def load_credentials() -> tuple[str, str, str]:
"""Returns (email, password, cookies). Cookies take priority when present."""
email = os.environ.get("BANDCAMP_EMAIL", "")
password = os.environ.get("BANDCAMP_PASSWORD", "")
cookies = ""
if email and password:
return email, password, cookies
try:
r = requests.get("http://localhost:8091/credentials", timeout=3)
if r.ok:
data = r.json()
if data.get("email"):
email = data["email"]
except Exception:
pass
cfg_file = Path.home() / ".config" / "audioserver" / "bandcamp.json"
if cfg_file.exists():
with open(cfg_file) as f:
cfg = json.load(f)
email = cfg.get("email", email)
password = cfg.get("password", password)
cookies = cfg.get("cookies", cookies)
docker_cfg = Path("/var/lib/docker/volumes/audioserver_bandcamp_config/_data/bandcamp.json")
if not password and not cookies and docker_cfg.exists():
try:
with open(docker_cfg) as f:
cfg = json.load(f)
email = cfg.get("email", email)
password = cfg.get("password", password)
cookies = cfg.get("cookies", cookies)
except Exception:
pass
return email, password, cookies
# ── Bandcamp auth ─────────────────────────────────────────────────────────────
def bc_login(email: str, password: str) -> dict:
"""Login to Bandcamp. Returns account dict with fan_id."""
r = SESSION.get("https://bandcamp.com/login", timeout=15)
m = re.search(r'<meta\s+name=["\']csrf-token["\']\s+content=["\']([^"\']+)["\']', r.text, re.IGNORECASE)
if not m:
m = re.search(r'<meta\s+content=["\']([^"\']+)["\']\s+name=["\']csrf-token["\']', r.text, re.IGNORECASE)
if not m:
raise RuntimeError("Could not get CSRF token from Bandcamp login page")
csrf = m.group(1)
time.sleep(REQUEST_DELAY)
r = SESSION.post(
"https://bandcamp.com/login_cb",
data={
"user.name": email,
"login.password": password,
"login.twofactor": "",
"login.twofactor_remember": "",
"login.from": "",
"to_band_path": "",
},
headers={
"X-CSRF-TOKEN": csrf,
"Referer": "https://bandcamp.com/login",
"Accept": "application/json, text/javascript, */*",
"Origin": "https://bandcamp.com",
},
timeout=15,
)
payload = r.json()
if payload.get("ok") != 1:
errors = payload.get("errors")
if errors and isinstance(errors, list):
err = "; ".join(f"{e.get('field','')}: {e.get('reason','')}" for e in errors if isinstance(e, dict))
else:
err = payload.get("error") or payload.get("error_message") or json.dumps(payload)
raise RuntimeError(f"Bandcamp login failed: {err}")
return payload.get("account", {})
def apply_cookies(session: requests.Session, cookie_str: str):
for part in cookie_str.split(';'):
part = part.strip()
if '=' in part:
name, _, value = part.partition('=')
session.cookies.set(name.strip(), value.strip(), domain='.bandcamp.com')
def get_session_info(session: requests.Session) -> tuple[int, str]:
"""Returns (fan_id, username) from session cookies or homepage."""
js_raw = session.cookies.get('js_account_details', '')
fan_id = None
username = ''
if js_raw:
try:
data = json.loads(unquote(js_raw))
fan_id = data.get('fan_id') or data.get('id')
username = data.get('username', '')
if fan_id:
return int(fan_id), username
except Exception:
pass
time.sleep(REQUEST_DELAY)
r = session.get("https://bandcamp.com/", timeout=15)
m = re.search(r'"fan_id"\s*:\s*(\d+)', r.text)
if m:
fan_id = int(m.group(1))
if not username:
m2 = re.search(r'"username"\s*:\s*"([^"]+)"', r.text)
if m2:
username = m2.group(1)
if not fan_id:
raise RuntimeError("Could not determine fan_id from session cookies — they may have expired")
return fan_id, username
def get_fan_id(account: dict) -> int:
fan_id = account.get("fan_id") or account.get("id")
if fan_id:
return int(fan_id)
username = account.get("username", "")
if username:
time.sleep(REQUEST_DELAY)
r = SESSION.get(f"https://bandcamp.com/{username}", timeout=15)
m = re.search(r'"fan_id"\s*:\s*(\d+)', r.text)
if m:
return int(m.group(1))
m = re.search(r'data-fan-id="(\d+)"', r.text)
if m:
return int(m.group(1))
raise RuntimeError("Could not determine fan_id — check credentials")
# ── Collection ────────────────────────────────────────────────────────────────
def fetch_collection(fan_id: int) -> list[dict]:
"""Fetch all purchased items from the fan collection."""
items = []
token = "9999999999:0:a::"
while True:
time.sleep(REQUEST_DELAY)
r = SESSION.post(
"https://bandcamp.com/api/fancollection/1/collection_items",
json={"fan_id": fan_id, "older_than_token": token, "count": COLLECTION_BATCH},
timeout=15,
)
r.raise_for_status()
data = r.json()
batch = data.get("items") or data.get("redownload_urls") or []
items.extend(batch)
if not data.get("more_available") or not batch:
break
token = data.get("last_token", "")
if not token:
break
return items
# ── Local library index ───────────────────────────────────────────────────────
def _norm(s: str) -> str:
"""Normalize for loose matching (lowercase, strip punctuation/spaces)."""
return re.sub(r"[^a-z0-9]", "", s.lower())
def index_local(music_dir: Path) -> set[str]:
"""Return a set of normalized 'artist||album' keys for existing folders."""
keys = set()
for entry in music_dir.iterdir():
if not entry.is_dir():
continue
name = entry.name
if " - " in name:
artist, album = name.split(" - ", 1)
keys.add(f"{_norm(artist)}||{_norm(album)}")
else:
keys.add(f"||{_norm(name)}")
return keys
def item_key(item: dict) -> str:
artist = item.get("band_name", "")
album = item.get("album_title", "") or item.get("item_title", "")
return f"{_norm(artist)}||{_norm(album)}"
def item_label(item: dict) -> str:
artist = item.get("band_name", "")
album = item.get("album_title", "") or item.get("item_title", "")
return f"{artist} - {album}"
# ── Download ──────────────────────────────────────────────────────────────────
def _parse_blob(html_text: str) -> dict:
"""Extract and parse the data-blob JSON from a Bandcamp page."""
m = re.search(r'data-blob="([^"]+)"', html_text)
if not m:
m = re.search(r"data-blob='([^']+)'", html_text)
if not m:
raise RuntimeError("data-blob not found on download page")
return json.loads(html.unescape(m.group(1)))
def get_wav_url(sale_item_id: int) -> str | None:
"""Return the WAV download URL for a purchase, or None if unavailable."""
time.sleep(REQUEST_DELAY)
r = SESSION.get(
"https://bandcamp.com/download",
params={
"from": "collection",
"payment_id": sale_item_id,
"stp": "gen",
"type": "album",
},
timeout=15,
)
if r.status_code != 200:
return None
try:
blob = _parse_blob(r.text)
except Exception as exc:
print(f" ✗ Could not parse data-blob: {exc}")
return None
for dl_item in blob.get("download_items", []):
downloads = dl_item.get("downloads", {})
for fmt in ("wav", "WAV"):
entry = downloads.get(fmt)
if entry and entry.get("url"):
return entry["url"]
return None
def download_and_extract(url: str, dest_dir: Path, label: str) -> Path | None:
"""Download a ZIP from url, extract to dest_dir/<Artist - Album>/."""
tmp_zip = dest_dir / f"_download_{os.getpid()}.zip"
tmp_extract = dest_dir / f"_extract_{os.getpid()}"
try:
print(f" Downloading {label}", end="", flush=True)
with SESSION.get(url, stream=True, timeout=120, allow_redirects=True) as r:
r.raise_for_status()
total = int(r.headers.get("Content-Length", 0))
downloaded = 0
with open(tmp_zip, "wb") as f:
for chunk in r.iter_content(chunk_size=1024 * 256):
f.write(chunk)
downloaded += len(chunk)
print(f" {downloaded // 1024 // 1024} MB")
tmp_extract.mkdir(parents=True, exist_ok=True)
with zipfile.ZipFile(tmp_zip) as z:
z.extractall(tmp_extract)
# Find the album folder inside the extracted ZIP
subdirs = [p for p in tmp_extract.iterdir() if p.is_dir()]
if subdirs:
album_src = subdirs[0]
else:
album_src = tmp_extract # files in root of ZIP
# Build destination folder name from the label
safe_label = re.sub(r'[<>:"/\\|?*]', "-", label)
final_dest = dest_dir / safe_label
if final_dest.exists():
shutil.rmtree(final_dest)
shutil.move(str(album_src), str(final_dest))
return final_dest
except Exception as exc:
print(f" ✗ Download failed: {exc}")
return None
finally:
if tmp_zip.exists():
tmp_zip.unlink()
if tmp_extract.exists():
shutil.rmtree(tmp_extract, ignore_errors=True)
def run_tagger(album_dir: Path):
"""Run bandcamp-tag.py on a single album folder."""
if not TAGGER.exists():
return
print(f" Tagging…")
result = subprocess.run(
[sys.executable, str(TAGGER), "--album", album_dir.name, str(album_dir.parent)],
capture_output=True, text=True,
)
for line in result.stdout.strip().splitlines():
print(f" {line}")
if result.returncode != 0:
for line in result.stderr.strip().splitlines():
print(f" ERR {line}")
# ── Main ──────────────────────────────────────────────────────────────────────
def main():
parser = argparse.ArgumentParser(description="Sync Bandcamp purchases to local library")
parser.add_argument("music_dir", nargs="?", type=Path, default=DEFAULT_MUSIC_DIR)
parser.add_argument("--dry-run", action="store_true", help="Show plan only, no downloads")
parser.add_argument("--list", action="store_true", help="List collection, exit")
parser.add_argument("--no-tag", action="store_true", help="Skip bandcamp-tag.py after download")
args = parser.parse_args()
music_dir = args.music_dir
email, password, cookies = load_credentials()
if cookies:
apply_cookies(SESSION, cookies)
print("Using saved session cookies…")
try:
fan_id, username = get_session_info(SESSION)
print(f"✓ Session active — {username or 'unknown'} (fan_id={fan_id})")
except RuntimeError as exc:
print(f"{exc}")
sys.exit(1)
elif email and password:
print(f"Logging in as {email}")
account = bc_login(email, password)
fan_id = get_fan_id(account)
username = account.get("username", str(fan_id))
print(f"✓ Logged in — {username} (fan_id={fan_id})")
else:
print("✗ No credentials found. Save session cookies or email/password in Settings → Bandcamp,")
print(" or set BANDCAMP_EMAIL / BANDCAMP_PASSWORD environment variables.")
sys.exit(1)
print("Fetching collection…")
items = fetch_collection(fan_id)
print(f" {len(items)} purchases found")
if args.list:
for item in items:
print(f" {item_label(item)}")
return
if not music_dir.exists():
print(f"✗ Music directory not found: {music_dir}")
sys.exit(1)
local = index_local(music_dir)
print(f" {len(local)} albums already in {music_dir}")
# Only consider album purchases (not singles / merch)
album_items = [i for i in items if i.get("item_type") in ("album", None)]
new_items = [i for i in album_items if item_key(i) not in local]
if not new_items:
print("✓ Library is up to date — nothing to download")
return
print(f"\n{len(new_items)} new album(s) to download:")
for item in new_items:
print(f"{item_label(item)}")
if args.dry_run:
return
print()
ok = 0
for item in new_items:
label = item_label(item)
sale_id = item.get("sale_item_id") or item.get("payment_id")
if not sale_id:
print(f" [{label}] — no sale_item_id, skipping")
continue
print(f" [{label}]")
wav_url = get_wav_url(int(sale_id))
if not wav_url:
print(f" ✗ No WAV download available (check Bandcamp — may need to choose format)")
continue
album_dir = download_and_extract(wav_url, music_dir, label)
if album_dir is None:
continue
if not args.no_tag and TAGGER.exists():
run_tagger(album_dir)
ok += 1
print(f"\n{ok}/{len(new_items)} album(s) downloaded")
if __name__ == "__main__":
main()