From 121f738b0eda1119068207fb620be8ea12dd70e6 Mon Sep 17 00:00:00 2001 From: yosh Date: Thu, 9 Jan 2025 14:10:22 -0500 Subject: [PATCH 1/5] decouple downloading logic from external features the `bc_free_downloader.py` module now only does the logic of getting info from bandcamp and downloading from bandcamp. the task of checking if an album is already downloaded is delegated to the __main__ file --- free_bandcamp_downloader/__main__.py | 637 +++++------------- .../bandcamp_http_adapter.py | 1 + .../bc_free_downloader.py | 451 +++++++++++++ 3 files changed, 614 insertions(+), 475 deletions(-) create mode 100644 free_bandcamp_downloader/bc_free_downloader.py diff --git a/free_bandcamp_downloader/__main__.py b/free_bandcamp_downloader/__main__.py index ce8602c..4cf370a 100644 --- a/free_bandcamp_downloader/__main__.py +++ b/free_bandcamp_downloader/__main__.py @@ -1,20 +1,22 @@ """Download free albums and tracks from Bandcamp + Usage: - bcdl-free [--debug] [--force] [--no-unzip] [-al] - [-d ] [-e ] [-z ] [-c ] [-f ] - [--cookies ] [--identity ] URL... bcdl-free setdefault [-d ] [-e ] [-z ] [-c ] [-f ] bcdl-free defaults bcdl-free clear bcdl-free -h | --help | --version + bcdl-free [--debug] [--force] [--no-unzip] [-al] + [-d ] [-e ] [-z ] [-c ] [-f ] + [--cookies ] [--identity ] [--download-history-file ] + URL... Arguments: URL URL to download. Can be a link to a label or release page Subcommands: - defaults list default configuration options setdefaults set default configuration options + defaults list default configuration options clear clear default configuration options Options: @@ -31,6 +33,7 @@ Options: -f --format Set format --cookies Path to cookies.txt file so albums in your collection can be downloaded --identity Value of identity cookie so albums in your collection can be downloaded + --download-history-file Path to history file containing downloaded albums Formats: - FLAC @@ -43,424 +46,48 @@ Formats: - AIFF """ -import atexit import dataclasses -import glob -import html -import json -import logging +import sys import os import pprint -import re -import sys -import time -import zipfile -from configparser import ConfigParser -from dataclasses import dataclass -from http.cookiejar import MozillaCookieJar -from typing import Dict, Optional, Set -from urllib.parse import urljoin - -import mutagen -import pyrfc6266 -import requests -from bs4 import BeautifulSoup +from typing import Set, Tuple from docopt import docopt -from tqdm import tqdm -from guerrillamail import GuerrillaMailSession +from configparser import ConfigParser -from free_bandcamp_downloader import __version__, logger -from free_bandcamp_downloader.bandcamp_http_adapter import BandcampHTTPAdapter +from free_bandcamp_downloader import __version__ +from free_bandcamp_downloader.bc_free_downloader import * +from free_bandcamp_downloader import logger -@dataclass -class BCFreeDownloaderOptions: - country: str = "United States" - zipcode: str = "00000" - email: str = "auto" - format: str = "FLAC" - dir: str = "." - - -@dataclass -class BCFreeDownloaderAlbumData: - about: Optional[str] - credits: Optional[str] - tags: Optional[str] - id: str - title: Optional[str] - - -class BCFreeDownloadError(Exception): - pass - - -class BCFreeDownloader: - CHUNK_SIZE = 1024 * 1024 - LINK_REGEX = re.compile(r'') - RETRY_URL_REGEX = re.compile(r'"retry_url":"(?P[^"]*)"') - FORMATS = { - "FLAC": "flac", - "V0MP3": "mp3-v0", - "320MP3": "mp3-320", - "AAC": "aac-hi", - "Ogg": "vorbis", - "ALAC": "alac", - "WAV": "wav", - "AIFF": "aiff-lossless", - } - - def __init__( - self, - options: BCFreeDownloaderOptions, - config_dir: str, - download_history_file: str, - unzip: bool = True, - cookies_file: Optional[str] = None, - identity: Optional[str] = None, - ): - self.options = options - self.config_dir = config_dir - self.download_history_file = download_history_file - self.downloaded: Set[str] = set() # can be URL or ID - self.mail_session = None - self.mail_album_data: Dict[str, BCFreeDownloaderAlbumData] = {} - self.unzip = unzip - self._init_downloaded() - self._init_session(cookies_file, identity) - - def _init_email(self): - self.mail_session = GuerrillaMailSession() - self.options.email = self.mail_session.get_session_state()["email_address"] - - def _init_downloaded(self): - if self.download_history_file: - with open(self.download_history_file, "r") as f: - for line in f: - self.downloaded.add(line.strip()) - - def _init_session(self, cookies_file: Optional[str], identity: Optional[str]): - self.session = requests.Session() - self.session.mount("https://", BandcampHTTPAdapter()) - if cookies_file: - cj = MozillaCookieJar(cookies_file) - cj.load() - self.session.cookies = cj - if identity: - self.session.cookies.set("identity", identity) - - def _download_file( - self, - download_page_url: str, - format: str, - album_data: Optional[BCFreeDownloaderAlbumData] = None, - ) -> str: - r = self.session.get(download_page_url) - r.raise_for_status() - soup = BeautifulSoup(r.text, "html.parser") - album_url = soup.find("div", class_="download-artwork").find("a").attrs["href"] - if album_data is None: - r = self.session.get(album_url) - r.raise_for_status() - id = self._get_album_data_from_soup(BeautifulSoup(r.text, "html.parser")).id - album_data = self.mail_album_data[id] - - data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"]) - download_url = data["digital_items"][0]["downloads"][self.FORMATS[format]][ - "url" - ] - - def download(download_url: str) -> str: - with self.session.get(download_url, stream=True) as r: - r.raise_for_status() - size = int(r.headers["content-length"]) - name = pyrfc6266.requests_response_to_filename(r) - file_name = os.path.join(self.options.dir, name) - with tqdm(total=size, unit="iB", unit_scale=True) as pbar: - with open(file_name, "wb") as f: - for chunk in r.iter_content(chunk_size=self.CHUNK_SIZE): - f.write(chunk) - pbar.update(len(chunk)) - return file_name - - try: - file_name = download(download_url) - except Exception: - statdownload_url = download_url.replace("/download/", "/statdownload/") - with self.session.get(statdownload_url) as r: - r.raise_for_status() - download_url = self.RETRY_URL_REGEX.search(r.text).group("retry_url") - if download_url: - file_name = download(download_url) - else: - # retry requires email address - raise BCFreeDownloadError( - "Download expired. Make sure your payment email is linked " - "to your fan account (Settings > Fan > Payment email addresses)" - ) - - logger.info(f"Downloaded {file_name}") - - if file_name.endswith("zip") and self.unzip: - # Unzip archive - dir_name = file_name[:-4] - with zipfile.ZipFile(file_name, "r") as f: - f.extractall(dir_name) - logger.info(f"Unzipped to {dir_name}. Use --no-unzip to prevent this") - os.remove(file_name) - files = glob.glob(os.path.join(dir_name, "*")) - else: - files = [file_name] - # Tag downloaded audio files with url & comment - logger.info("Setting tags...") - for file in files: - f = mutagen.File(file) - if f is None: - continue - f["website"] = album_url - if album_data.tags: - f["genre"] = album_data.tags - comment = "" - if album_data.about: - comment += album_data.about - if album_data.about and album_data.credits: - comment += "\n\n" - if album_data.credits: - comment += album_data.credits - f["comment"] = comment - f.save() - # successfully downloaded file, add to download history - self.downloaded.add(album_data.id) - if self.download_history_file: - with open(self.download_history_file, "a") as f: - f.write(f"{album_data.id}\n") - - return album_data.id - - @staticmethod - def _get_album_data_from_soup(soup: BeautifulSoup) -> BCFreeDownloaderAlbumData: - about = soup.find("div", class_="tralbum-about") - credits = soup.find("div", class_="tralbum-credits") - tags = [tag.get_text() for tag in soup.find_all("a", class_="tag")] - properties = json.loads( - soup.find("meta", attrs={"name": "bc-page-properties"})["content"] - ) - id = f"{properties['item_type']}:{properties['item_id']}" - - return BCFreeDownloaderAlbumData( - about=about.get_text("\n") if about else None, - credits=credits.get_text("\n") if credits else None, - tags=",".join(sorted(tags)), - id=id, - title=None, +class Config: + def __init__(self): + config_dir = get_config_dir() + # initialize with default config first + # this is combined from the dataclass and CLI params + self.parser = ConfigParser(allow_no_value=True) + self.parser["free-bandcamp-downloader"] = {} + for field in dataclasses.fields(BCFreeDownloaderOptions): + self.parser["free-bandcamp-downloader"][field.name] = field.default + self.parser["free-bandcamp-downloader"]["force"] = "false" + self.parser["free-bandcamp-downloader"]["no-unzip"] = "false" + self.parser["free-bandcamp-downloader"]["download-history-file"] = ( + get_data_dir() + "/downloaded.txt" ) - def _download_purchased_album( - self, user_id: int, album_data: BCFreeDownloaderAlbumData - ): - logger.info("Downloading album from collection...") - logger.debug(f"Searching for album: '{album_data.title}'") - data = { - "fan_id": user_id, - "search_key": album_data.title, - "search_type": "collection", - } - r = self.session.post( - "https://bandcamp.com/api/fancollection/1/search_items", json=data - ) - r.raise_for_status() - results = r.json() - tralbums = results["tralbums"] - redownload_urls = results["redownload_urls"] - try: - tralbum = next( - filter( - lambda tralbum: f"{tralbum['tralbum_type']}:{tralbum['tralbum_id']}" - == album_data.id, - tralbums, - ) - ) - except StopIteration: - raise BCFreeDownloadError("Could not find album in collection") - sale_id = f"{tralbum['sale_item_type']}{tralbum['sale_item_id']}" - if sale_id not in redownload_urls: - raise BCFreeDownloadError("Could not find album download URL in collection") - download_url = redownload_urls[sale_id] - logger.debug(f"Got download URL: {download_url}") - self._download_file(download_url, self.options.format, album_data) - - # check if the album was already downloaded given an id and url - # format of id is [at]: - def is_downloaded(self, id: str, url: str = ""): - return url in self.downloaded or id in self.downloaded - - def _download_album(self, soup: BeautifulSoup, force: bool = False): - album_data = self._get_album_data_from_soup(soup) - url = soup.head.find("meta", attrs={"property": "og:url"})["content"] - if not force and self.is_downloaded(album_data.id, url): - raise BCFreeDownloadError( - f"{url} already downloaded. To download anyways, use option --force" - ) - - logger.debug(f"Album data: {album_data}") - - tralbum_data = soup.find("script", {"data-tralbum": True}).attrs["data-tralbum"] - tralbum_data = json.loads(tralbum_data) - - if not tralbum_data["hasAudio"]: - raise BCFreeDownloadError(f"{url} has no audio. Skipping...") - - head_data = soup.head.find( - "script", {"type": "application/ld+json"}, recursive=False - ).string - - head_data = json.loads(head_data) - head_id = head_data.get("@id") - # fallback if a track link was provided - # track releases have this inAlbum key even if they're standalone - head_data = head_data.get("inAlbum", head_data)["albumRelease"] - # find the albumRelease object that matches the overall album @id link - # this will ensure that strictly the page release is downloaded - head_data = next(obj for obj in head_data if obj["@id"] == head_id) - if "offers" not in head_data: - raise BCFreeDownloadError(f"{url} has no digital download. Skipping...") - - album_data.title = head_data["name"] - - if head_data["offers"]["price"] == 0.0: - if tralbum_data["current"]["require_email"]: - if not self.options.email or self.options.email == "auto": - self._init_email() - logger.info(f"{url} requires email") - email_post_url = urljoin(url, "/email_download") - r = self.session.post( - email_post_url, - data={ - "encoding_name": "none", - "item_id": tralbum_data["current"]["id"], - "item_type": tralbum_data["current"]["type"], - "address": self.options.email, - "country": self.options.country, - "postcode": self.options.zipcode, - }, - ) - r.raise_for_status() - r = r.json() - if not r["ok"]: - raise ValueError(f"Bad response when sending email address: {r}") - self.mail_album_data[album_data.id] = album_data - else: - logger.info(f"{url} does not require email") - self._download_file( - tralbum_data["freeDownloadPage"], self.options.format, album_data - ) - else: - if tralbum_data["is_purchased"]: - collection_info = soup.find( - "script", {"data-tralbum-collect-info": True} - ).attrs["data-tralbum-collect-info"] - collection_info = json.loads(collection_info) - self._download_purchased_album(collection_info["fan_id"], album_data) - else: - raise BCFreeDownloadError( - f"{url} is not free. If you have purchased this album, " - "use the --cookies flag or --identity flag to pass your login cookie." - ) - - def _download_label(self, soup: BeautifulSoup, force: bool = False): - albums = [] - baseurl = soup.head.find("meta", attrs={"property": "og:url"})["content"] - - # bandcamp splits the album between this music-grid html and some json blob - grid = soup.find("ol", id="music-grid") - for li in grid.find_all("li"): - if "display:none" in li.get("style", ""): - continue - data = li["data-item-id"].split("-") - albums += [{"url": li.a["href"], "id": f"{data[0][0]}:{data[1]}"}] - for obj in json.loads(html.unescape(grid.get("data-client-items", {}))): - if obj.get("filtered"): - continue - albums += [{"url": obj["page_url"], "id": f"{obj['type'][0]}:{obj['id']}"}] - - for album in albums: - if album["url"][0] == "/": - album["url"] = urljoin(baseurl, album["url"]) - - # perform a check here for already-downloaded album to prevent mass requests - # to bandcamp for large labels and getting rate limited as a result - if not force and self.is_downloaded(album["id"], album["url"]): - logger.info( - f"{album['url']} already downloaded. To download anyways, use option --force" - ) - continue - - logger.info(f"Downloading {album['url']}") - r = self.session.get(album["url"]) - r.raise_for_status() - soup = BeautifulSoup(r.text, "html.parser") - try: - self._download_album(soup, force) - except BCFreeDownloadError as ex: - logger.info(ex) - - def download_url(self, url: str, force: bool = False): - # detect whether it's a release or label page - r = self.session.get(url) - r.raise_for_status() - soup = BeautifulSoup(r.text, "html.parser") - - try: - url_type = soup.head.find("meta", attrs={"property": "og:type"})["content"] - except Exception: - raise BCFreeDownloadError(f"{url} does not have an og:type property.") - - if url_type == "album" or url_type == "song": - self._download_album(soup, force) - elif url_type == "band": - self._download_label(soup, force) - else: - raise BCFreeDownloadError(f"{url} does not have a valid og:type value") - - def wait_for_email_downloads(self): - checked_ids = set() - while (expected_emails := len(self.mail_album_data)) > 0: - logger.info(f"Waiting for {expected_emails} emails from Bandcamp...") - time.sleep(5) - for email in self.mail_session.get_email_list(): - email_id = email.guid - if email_id not in checked_ids: - checked_ids.add(email_id) - if ( - email.sender == "noreply@bandcamp.com" - and "download" in email.subject - ): - logger.info(f'Received email "{email.subject}"') - content = self.mail_session.get_email(email_id).body - match = self.LINK_REGEX.search(content) - if match: - download_url = match.group("url") - album_id = self._download_file( - download_url, self.options.format - ) - self.mail_album_data.pop(album_id) - else: - logger.error( - f"Could not find download URL in body: {content}" - ) - - -class BCFreeDownloaderConfig: - def __init__(self, config_path: str): - self.config_path = config_path - self.parser = ConfigParser() - self.parser.read(config_path) - atexit.register(self.save) + # read config file + self.config_path = os.path.join(config_dir, "free-bandcamp-downloader.cfg") + if not os.path.exists(self.config_path): + with open(self.config_path, "w") as f: + self.parser.write(f) + return + self.parser.read(self.config_path) def get(self, key): - return self.parser["free-bandcamp-downloader"].get(key, None) + return self.parser["free-bandcamp-downloader"].get(key) def set(self, key, value): + if value is not None: + value = str(value) self.parser["free-bandcamp-downloader"][key] = value def save(self): @@ -471,7 +98,14 @@ class BCFreeDownloaderConfig: return pprint.pformat(dict(self.parser["free-bandcamp-downloader"]), indent=2) -def get_config_dir(): +def options_from_config(config: Config): + options = BCFreeDownloaderOptions() + for field in dataclasses.fields(options): + setattr(options, field.name, config.parser["free-bandcamp-downloader"][field.name]) + return options + + +def get_config_dir() -> str: if "XDG_CONFIG_HOME" in os.environ: config_dir = os.path.join( os.environ["XDG_CONFIG_HOME"], "free-bandcamp-downloader" @@ -480,84 +114,154 @@ def get_config_dir(): config_dir = os.path.join( os.path.expanduser("~"), ".config", "free-bandcamp-downloader" ) + if not os.path.exists(config_dir): + os.makedirs(config_dir) return config_dir -def get_data_dir(): +def get_data_dir() -> str: if "XDG_DATA_HOME" in os.environ: data_dir = os.path.join(os.environ["XDG_DATA_HOME"], "free-bandcamp-downloader") else: data_dir = os.path.join( os.path.expanduser("~"), ".local", "share", "free-bandcamp-downloader" ) + if not os.path.exists(data_dir): + os.makedirs(data_dir) return data_dir -def get_config(data_dir: str, config_dir: str): - download_history_file = os.path.join(data_dir, "downloaded.txt") - default_config = f"""[free-bandcamp-downloader] - country = United States - zipcode = 00000 - email = auto - format = FLAC - dir = . - download_history_file = {download_history_file}""" - config_file = os.path.join(config_dir, "free-bandcamp-downloader.cfg") - if not os.path.exists(config_file): - if not os.path.exists(config_dir): - os.makedirs(config_dir) - with open(config_file, "w") as f: - f.write(default_config) - if not os.path.exists(download_history_file): - if not os.path.exists(data_dir): - os.makedirs(data_dir) - with open(download_history_file, "w") as f: +def is_downloaded(downloaded_set, id: Tuple[str, int], url: str = None) -> bool: + return id in downloaded_set or url in downloaded_set + + +def add_to_dl_file(config: Config, id: Tuple[str, int]): + history_file = config.parser["free-bandcamp-downloader"]["download-history-file"] + with open(history_file, "a") as f: + f.write(f"{id[0][0]}:{id[1]}\n") + + +def get_downloaded(config: Config) -> Set[Tuple[str, int | str]]: + history_file = config.parser["free-bandcamp-downloader"]["download-history-file"] + if not os.path.exists(history_file): + with open(history_file, "w") as f: pass - config = BCFreeDownloaderConfig(config_file) - return config + + downloaded = set() + with open(history_file, "r") as f: + for line in f: + type = line.strip()[:2] + if type == "a:": + type = "album" + data = int(line[2:]) + elif type == "t:": + type = "track" + data = int(line[2:]) + else: + type = "url" + data = line.strip() + downloaded.add((type, data)) + return downloaded + + +def post_download(album_info: AlbumInfo, config: Config): + file_name = album_info["file_name"] + # file list for setting tags + files = [file_name] + unzip = not config.parser.getboolean("free-bandcamp-downloader", "no-unzip") + + # unzip if needed + if unzip and file_name.endswith(".zip"): + files = BCFreeDownloader.unzip_album(file_name) + + logger.info("Setting tags...") + for file in files: + BCFreeDownloader.tag_file(file, album_info["head_data"]) + + +def download_urls(urls: List[str], config: Config): + downloader = BCFreeDownloader(options_from_config(config)) + downloaded = get_downloaded(config) + force = config.parser.getboolean("free-bandcamp-downloader", "force") + + for url in urls: + soup = downloader.get_url_soup(url) + url_info = downloader.get_page_info(soup) + + urltype = url_info and url_info.get("type") + if urltype == "album" or urltype == "song": + tralbum = url_info["info"]["tralbum_data"] + type = tralbum["current"]["type"] + id = tralbum["current"]["id"] + url = tralbum["url"] + if not force and is_downloaded(downloaded, (type, id), url): + logger.error( + f"{url} already downloaded. To download anyways, use --force." + ) + continue + ret = downloader.download_album(soup) + if ret["is_downloaded"]: + add_to_dl_file(config, (type, id)) + downloaded.add((type, id)) + post_download(ret, config) + elif urltype == "band": + for rel in url_info["info"]["releases"]: + type = rel["type"] + id = rel["id"] + url = rel["url"] + if not force and is_downloaded(downloaded, (type, id), url): + logger.error( + f"{url} already downloaded. To download anyways, use --force." + ) + continue + soup = downloader.get_url_soup(url) + ret = downloader.download_album(soup) + if ret["is_downloaded"]: + add_to_dl_file(config, (type, id)) + downloaded.add((type, id)) + post_download(ret, config) + else: + continue + + # finish up downloading + ret = downloader.flush_email_downloads() + for album_info in ret: + type = album_info["tralbum_data"]["current"]["type"] + id = album_info["tralbum_data"]["current"]["id"] + add_to_dl_file(config, (type, id)) + downloaded.add((type, id)) + post_download(album_info, config) def main(): - data_dir = get_data_dir() - config_dir = get_config_dir() - config = get_config(data_dir, config_dir) - options = BCFreeDownloaderOptions() + config = Config() arguments = docopt(__doc__, version=__version__) if arguments["--debug"]: logger.setLevel(logging.DEBUG) - # set options if needed + # set config if needed if arguments["URL"] or arguments["setdefault"]: - for field in dataclasses.fields(options): - option = field.name + for option in config.parser["free-bandcamp-downloader"].keys(): arg = f"--{option}" - if arguments[arg]: - setattr(options, option, arguments[arg]) - else: - setattr(options, option, config.get(option)) - if not getattr(options, option): - logger.error( - f'{option} is not set, use "bcdl-free setdefault {arg} <{option}>"' - ) - sys.exit(1) - if options.format not in BCFreeDownloader.FORMATS: + if arguments.get(arg): + config.set(option, arguments[arg]) + if ( + config.parser["free-bandcamp-downloader"]["format"] + not in BCFreeDownloader.FORMATS + ): logger.error( - f'{options["format"]} is not a valid format. See "bcdl-free -h" for valid formats' + f'{config.parser.get("format")} is not a valid format. See "bcdl-free -h" for valid formats' ) sys.exit(1) + # write to config file if arguments["setdefault"]: - # write arguments to config - for field in dataclasses.fields(options): - option = field.name - arg = f"--{option}" - if arguments[arg]: - config.set(option, arguments[arg]) + config.save() sys.exit(0) if arguments["clear"]: - with open(config.get("download_history_file"), "w"): + with open(config.config_path, "w"): pass sys.exit(0) @@ -566,24 +270,7 @@ def main(): sys.exit(0) if arguments["URL"]: - # init downloader - downloader = BCFreeDownloader( - options, - config_dir, - config.get("download_history_file"), - not arguments["--no-unzip"], - arguments["--cookies"], - arguments["--identity"], - ) - - for url in arguments["URL"]: - try: - downloader.download_url(url) - except BCFreeDownloadError as ex: - logger.info(ex) - - # finish up downloading - downloader.wait_for_email_downloads() + download_urls(arguments["URL"], config) if __name__ == "__main__": diff --git a/free_bandcamp_downloader/bandcamp_http_adapter.py b/free_bandcamp_downloader/bandcamp_http_adapter.py index 5916b7d..48b0c2b 100644 --- a/free_bandcamp_downloader/bandcamp_http_adapter.py +++ b/free_bandcamp_downloader/bandcamp_http_adapter.py @@ -1,4 +1,5 @@ from requests.adapters import HTTPAdapter +from urllib3.util import Retry from urllib3.util.ssl_ import create_urllib3_context diff --git a/free_bandcamp_downloader/bc_free_downloader.py b/free_bandcamp_downloader/bc_free_downloader.py new file mode 100644 index 0000000..0b7b795 --- /dev/null +++ b/free_bandcamp_downloader/bc_free_downloader.py @@ -0,0 +1,451 @@ +import glob +import html +import json +import os +import re +import time +import zipfile +import mutagen +import pyrfc6266 +import requests + +from bs4 import BeautifulSoup +from tqdm import tqdm +from dataclasses import dataclass +from http.cookiejar import MozillaCookieJar +from typing import Dict, List, Optional, Tuple, TypedDict +from urllib.parse import urljoin +from guerrillamail import GuerrillaMailSession + +from free_bandcamp_downloader import logger +from free_bandcamp_downloader.bandcamp_http_adapter import * + + +class DownloadRet(TypedDict): + id: Tuple[str, int] + file_name: str + + +class AlbumInfo(TypedDict): + tralbum_data: Dict + head_data: Dict + is_downloaded: Optional[bool] + email_queued: Optional[bool] + file_name: Optional[str] + + +class LabelReleaseInfo(TypedDict): + type: str + id: Tuple[str, int] + band_id: int + url: str + release_info: Optional[AlbumInfo] + + +class LabelInfo(TypedDict): + label_info: Dict + releases: List[LabelReleaseInfo] + + +class PageInfo(TypedDict): + type: str + info: LabelInfo | AlbumInfo + + +@dataclass +class BCFreeDownloaderOptions: + country: str = "United States" + zipcode: str = "00000" + email: str = "auto" + format: str = "FLAC" + dir: str = "." + cookies: Optional[str] = None + identity: Optional[str] = None + + +class BCFreeDownloadError(Exception): + pass + + +class BCFreeDownloader: + CHUNK_SIZE = 1024 * 1024 + LINK_REGEX = re.compile(r'') + RETRY_URL_REGEX = re.compile(r'"retry_url":"(?P[^"]*)"') + FORMATS = { + "FLAC": "flac", + "V0MP3": "mp3-v0", + "320MP3": "mp3-320", + "AAC": "aac-hi", + "Ogg": "vorbis", + "ALAC": "alac", + "WAV": "wav", + "AIFF": "aiff-lossless", + } + + def __init__(self, options: BCFreeDownloaderOptions): + self.options = options + self.mail_session = None + self.queued_emails = {} # { ("album"|"track", id): {info} } + self.session = None + self.email = None + self._init_session() + + def _init_email(self): + logger.info("Starting mail session...") + if not self.options.email: + self.options.email = "auto" + if self.options.email == "auto": + self.mail_session = GuerrillaMailSession() + self.options.email = self.mail_session.get_session_state()["email_address"] + + def _init_session(self): + self.session = requests.Session() + retries = Retry( + total = 10, + backoff_factor = 10, + backoff_max = 60, + allowed_methods = {"POST", "GET"} + ) + self.session.mount("https://", BandcampHTTPAdapter(max_retries=retries)) + if self.options.cookies: + cj = MozillaCookieJar(self.options.cookies) + cj.load() + self.session.cookies = cj + if self.options.identity: + self.session.cookies.set("identity", self.options.identity) + + def _download_file(self, download_page_url: str, format: str) -> DownloadRet: + soup = self.get_url_soup(download_page_url) + album_url = soup.find("div", class_="download-artwork").find("a").attrs["href"] + + data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"])[ + "digital_items" + ][0] + download_url = data["downloads"][self.FORMATS[format]]["url"] + id = (data["type"], int(data["item_id"])) + + def download(download_url: str) -> str: + with self.get_url(download_url, stream=True) as r: + size = int(r.headers["content-length"]) + name = pyrfc6266.requests_response_to_filename(r) + file_name = os.path.join(self.options.dir, name) + with tqdm(total=size, unit="iB", unit_scale=True) as pbar: + with open(file_name, "wb") as f: + for chunk in r.iter_content(chunk_size=self.CHUNK_SIZE): + f.write(chunk) + pbar.update(len(chunk)) + return file_name + + try: + file_name = download(download_url) + except Exception: + statdownload_url = download_url.replace("/download/", "/statdownload/") + with self.get_url(statdownload_url) as r: + download_url = self.RETRY_URL_REGEX.search(r.text).group("retry_url") + if download_url: + file_name = download(download_url) + else: + # retry requires email address + raise BCFreeDownloadError( + "Download expired. Make sure your payment email is linked " + "to your fan account (Settings > Fan > Payment email addresses)" + ) + + logger.info(f"Downloaded {file_name}") + + return {"id": id, "file_name": file_name} + + # unzip the provided file and return all file paths + @staticmethod + def unzip_album(file_name: str) -> List[str]: + dir_name = file_name[:-4] + with zipfile.ZipFile(file_name, "r") as f: + f.extractall(dir_name) + logger.info(f"Unzipped {file_name}.") + os.remove(file_name) + return glob.glob(os.path.join(dir_name, "*")) + + # Tag downloaded audio file with url & comment + @staticmethod + def tag_file(file_name: str, head_data: Dict): + try: + f = mutagen.File(file_name) + if f is None: + return + + f["website"] = head_data["@id"] + if head_data.get("keywords"): + f["genre"] = head_data["keywords"] + comment = "" + comment += head_data.get("description", "").strip() + comment += "\n\n" + head_data.get("creditText", "") + f["comment"] = comment.strip() + f.save() + except Exception: + # only should happen if the file doesn't support tags + pass + + def _download_purchased_album( + self, user_id: int, tralbum_data: Dict + ) -> DownloadRet: + logger.info("Downloading album from collection...") + logger.debug(f"Searching for album: '{tralbum_data["current"]["title"]}'") + data = { + "fan_id": user_id, + "search_key": tralbum_data["current"]["title"], + "search_type": "collection", + } + results = self.post_url_json( + "https://bandcamp.com/api/fancollection/1/search_items", json=data + ) + tralbums = results["tralbums"] + redownload_urls = results["redownload_urls"] + wanted_id = f"{tralbum_data["item_type"][0]}:{tralbum_data["id"]}" + try: + tralbum = next( + filter( + lambda tralbum: f"{tralbum["tralbum_type"]}:{tralbum["tralbum_id"]}" + == wanted_id, + tralbums, + ) + ) + except StopIteration: + raise BCFreeDownloadError("Could not find album in collection") + sale_id = f"{tralbum["sale_item_type"]}{tralbum["sale_item_id"]}" + if sale_id not in redownload_urls: + raise BCFreeDownloadError("Could not find album download URL in collection") + download_url = redownload_urls[sale_id] + logger.debug(f"Got download URL: {download_url}") + return self._download_file(download_url, self.options.format) + + # download from release page + def download_album(self, soup: BeautifulSoup) -> AlbumInfo: + album_data = BCFreeDownloader.get_album_info(soup) + tralbum_data = album_data["tralbum_data"] + head_data = album_data["head_data"] + album_data["is_downloaded"] = False + album_data["email_queued"] = False + url = tralbum_data["url"] + + logger.debug(f"tralbum data: {tralbum_data}") + logger.debug(f"album head data: {head_data}") + + if not tralbum_data["hasAudio"]: + logger.error(f"{url} has no audio.") + return album_data + + head_id = head_data.get("@id") + # fallback if a track link was provided + # track releases have this inAlbum key even if they're standalone + album_release = head_data.get("inAlbum", head_data)["albumRelease"] + # find the albumRelease object that matches the overall album @id link + # this will ensure that strictly the page release is downloaded + album_release = next(obj for obj in album_release if obj["@id"] == head_id) + + if "offers" not in album_release: + logger.error(f"{url} has no available offers.") + + if tralbum_data["freeDownloadPage"]: + logger.info(f"{url} does not require email") + dlret = self._download_file( + tralbum_data["freeDownloadPage"], self.options.format + ) + elif album_release["offers"]["price"] == 0.0: + logger.info(f"{url} requires email") + if self.mail_session is None: + self._init_email() + email_post_url = urljoin(url, "/email_download") + r = self.post_url_json( + email_post_url, + data={ + "encoding_name": "none", + "item_id": tralbum_data["current"]["id"], + "item_type": tralbum_data["current"]["type"], + "address": self.options.email, + "country": self.options.country, + "postcode": self.options.zipcode, + }, + ) + + if not r["ok"]: + raise ValueError(f"Bad response when sending email address: {r}") + type = tralbum_data["current"]["type"] + id = tralbum_data["current"]["id"] + album_data["email_queued"] = True + self.queued_emails[(type, id)] = album_data + return album_data + elif tralbum_data["is_purchased"]: + collection_info = soup.find( + "script", {"data-tralbum-collect-info": True} + ).attrs["data-tralbum-collect-info"] + collection_info = json.loads(collection_info) + dlret = self._download_purchased_album( + collection_info["fan_id"], tralbum_data + ) + else: + logger.error( + f"{url} is not free. If you have purchased this album, " + "use the --cookies flag or --identity flag to pass your login cookie." + ) + return album_data + + album_data["is_downloaded"] = True + album_data["file_name"] = dlret["file_name"] + + return album_data + + # unconditionally download from release page + def download_label(self, soup: BeautifulSoup) -> LabelInfo: + info = BCFreeDownloader.get_label_info(soup) + + for release in info["releases"]: + logger.info(f"Downloading {release["url"]}") + + soup = self.get_url_soup(release["url"]) + try: + ret = self.download_album(soup) + release["release_info"] = ret + except BCFreeDownloadError as ex: + logger.info(ex) + + return info + + # unconditionally downloads the provided url + # returns either the result of download_album or download_label + # with the `page_type` set to album|song|band + # exception if download error + def download_url(self, url: str, force: bool = False): + soup = self.get_url_soup(url) + + page_info = self.get_page_info(soup) + page_type = page_info.get("type") + if page_type == "album" or page_type == "song": + ret = self.download_album(soup, force) + else: + ret = self.download_label(soup, force) + + ret["page_type"] = page_type + + return ret + + def flush_email_downloads(self) -> List[AlbumInfo]: + checked_ids = set() + downloaded = [] + while len(self.queued_emails) > 0: + logger.info( + f"Waiting for {len(self.queued_emails)} emails from Bandcamp..." + ) + time.sleep(5) + for email in self.mail_session.get_email_list(): + email_id = email.guid + if email_id in checked_ids: + continue + + checked_ids.add(email_id) + if ( + email.sender == "noreply@bandcamp.com" + and "download" in email.subject + ): + logger.info(f'Received email "{email.subject}"') + content = self.mail_session.get_email(email_id).body + match = self.LINK_REGEX.search(content) + if match: + download_url = match.group("url") + dlret = self._download_file(download_url, self.options.format) + self.queued_emails[dlret["id"]]["file_name"] = dlret[ + "file_name" + ] + downloaded.append(self.queued_emails[dlret["id"]]) + self.queued_emails.pop(dlret["id"]) + else: + logger.error(f"Could not find download URL in body: {content}") + return downloaded + + # get_url_x can't be staticmethods because of special session context + def get_url(self, url: str, **kwargs) -> requests.Response: + r = self.session.get(url, **kwargs) + r.raise_for_status() + return r + + def get_url_soup(self, url: str, **kwargs) -> BeautifulSoup: + return BeautifulSoup(self.get_url(url, **kwargs).text, "html.parser") + + def get_url_info(self, url: str, **kwargs) -> PageInfo: + soup = self.get_url_soup(url, **kwargs) + try: + return self.get_page_info(soup) + except Exception: + raise BCFreeDownloadError(f"Could not get page info for {url}") + + def post_url(self, url: str, **kwargs) -> requests.Response: + r = self.session.post(url, **kwargs) + r.raise_for_status() + return r + + def post_url_json(self, url: str, **kwargs) -> Dict: + return self.post_url(url, **kwargs).json() + + # get the album/label info of a bandcamp page + @staticmethod + def get_page_info(soup: BeautifulSoup) -> PageInfo: + page_type = soup.head.find("meta", attrs={"property": "og:type"}).get("content") + + if page_type == "album" or page_type == "song": + return {"type": page_type, "info": BCFreeDownloader.get_album_info(soup)} + if page_type == "band": + return {"type": page_type, "info": BCFreeDownloader.get_label_info(soup)} + else: + # only bandcamp pages are supported + raise BCFreeDownloadError("Page does not have a valid og:type value") + + @staticmethod + def get_label_info(soup: BeautifulSoup) -> LabelInfo: + label_info = soup.find("script", attrs={"data-band": True}) + if label_info is None: + raise BCFreeDownloadError("Page has no data-band script.") + label_info = json.loads(label_info["data-band"]) + + releases = [] + # needed for releases + local_url = label_info["local_url"] + + # bandcamp splits the release between this music-grid html and some json blob + grid = soup.find("ol", id="music-grid") + for li in grid.find_all("li"): + if "display:none" in li.get("style", ""): + continue + + data = li["data-item-id"].split("-") + # most important fields + releases.append( + { + "type": data[0], + "id": int(data[1]), + "url": li.a["href"], + "band_id": int(li["data-band-id"]), + } + ) + for obj in json.loads(html.unescape(grid.get("data-client-items", {}))): + if obj.get("filtered"): + continue + # normalize to fit the other half + obj["url"] = obj.pop("page_url") + releases.append(obj) + + # fixup local urls into global ones + for release in releases: + if release["url"][0] == "/": + release["url"] = urljoin(local_url, release["url"]) + + return {"label_info": label_info, "releases": releases} + + @staticmethod + def get_album_info(soup: BeautifulSoup) -> AlbumInfo: + tralbum_data = soup.find("script", {"data-tralbum": True}).attrs["data-tralbum"] + tralbum_data = json.loads(tralbum_data) + head_data = soup.head.find( + "script", {"type": "application/ld+json"}, recursive=False + ).string + head_data = json.loads(head_data) + + return {"tralbum_data": tralbum_data, "head_data": head_data} From 5b1a926e49c86da71f56888bec5a3ed3a4e8e3f0 Mon Sep 17 00:00:00 2001 From: 7x11x13 Date: Fri, 21 Mar 2025 13:45:58 -0400 Subject: [PATCH 2/5] Lint --- .gitignore | 3 ++- free_bandcamp_downloader/__main__.py | 13 ++++++++++--- .../bandcamp_http_adapter.py | 1 - .../bc_free_downloader.py | 19 ++++++++----------- 4 files changed, 20 insertions(+), 16 deletions(-) diff --git a/.gitignore b/.gitignore index f935c1b..aa87166 100644 --- a/.gitignore +++ b/.gitignore @@ -2,4 +2,5 @@ env *.egg-info *.log __pycache__ -notes \ No newline at end of file +notes +.DS_Store \ No newline at end of file diff --git a/free_bandcamp_downloader/__main__.py b/free_bandcamp_downloader/__main__.py index 4cf370a..71cd647 100644 --- a/free_bandcamp_downloader/__main__.py +++ b/free_bandcamp_downloader/__main__.py @@ -47,15 +47,20 @@ Formats: """ import dataclasses +import logging import sys import os import pprint -from typing import Set, Tuple +from typing import List, Set, Tuple from docopt import docopt from configparser import ConfigParser from free_bandcamp_downloader import __version__ -from free_bandcamp_downloader.bc_free_downloader import * +from free_bandcamp_downloader.bc_free_downloader import ( + AlbumInfo, + BCFreeDownloader, + BCFreeDownloaderOptions, +) from free_bandcamp_downloader import logger @@ -101,7 +106,9 @@ class Config: def options_from_config(config: Config): options = BCFreeDownloaderOptions() for field in dataclasses.fields(options): - setattr(options, field.name, config.parser["free-bandcamp-downloader"][field.name]) + setattr( + options, field.name, config.parser["free-bandcamp-downloader"][field.name] + ) return options diff --git a/free_bandcamp_downloader/bandcamp_http_adapter.py b/free_bandcamp_downloader/bandcamp_http_adapter.py index 48b0c2b..5916b7d 100644 --- a/free_bandcamp_downloader/bandcamp_http_adapter.py +++ b/free_bandcamp_downloader/bandcamp_http_adapter.py @@ -1,5 +1,4 @@ from requests.adapters import HTTPAdapter -from urllib3.util import Retry from urllib3.util.ssl_ import create_urllib3_context diff --git a/free_bandcamp_downloader/bc_free_downloader.py b/free_bandcamp_downloader/bc_free_downloader.py index 0b7b795..cf5f669 100644 --- a/free_bandcamp_downloader/bc_free_downloader.py +++ b/free_bandcamp_downloader/bc_free_downloader.py @@ -16,9 +16,10 @@ from http.cookiejar import MozillaCookieJar from typing import Dict, List, Optional, Tuple, TypedDict from urllib.parse import urljoin from guerrillamail import GuerrillaMailSession +from urllib3 import Retry from free_bandcamp_downloader import logger -from free_bandcamp_downloader.bandcamp_http_adapter import * +from free_bandcamp_downloader.bandcamp_http_adapter import BandcampHTTPAdapter class DownloadRet(TypedDict): @@ -101,10 +102,7 @@ class BCFreeDownloader: def _init_session(self): self.session = requests.Session() retries = Retry( - total = 10, - backoff_factor = 10, - backoff_max = 60, - allowed_methods = {"POST", "GET"} + total=10, backoff_factor=10, backoff_max=60, allowed_methods={"POST", "GET"} ) self.session.mount("https://", BandcampHTTPAdapter(max_retries=retries)) if self.options.cookies: @@ -116,7 +114,6 @@ class BCFreeDownloader: def _download_file(self, download_page_url: str, format: str) -> DownloadRet: soup = self.get_url_soup(download_page_url) - album_url = soup.find("div", class_="download-artwork").find("a").attrs["href"] data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"])[ "digital_items" @@ -189,7 +186,7 @@ class BCFreeDownloader: self, user_id: int, tralbum_data: Dict ) -> DownloadRet: logger.info("Downloading album from collection...") - logger.debug(f"Searching for album: '{tralbum_data["current"]["title"]}'") + logger.debug(f"Searching for album: '{tralbum_data['current']['title']}'") data = { "fan_id": user_id, "search_key": tralbum_data["current"]["title"], @@ -200,18 +197,18 @@ class BCFreeDownloader: ) tralbums = results["tralbums"] redownload_urls = results["redownload_urls"] - wanted_id = f"{tralbum_data["item_type"][0]}:{tralbum_data["id"]}" + wanted_id = f"{tralbum_data['item_type'][0]}:{tralbum_data['id']}" try: tralbum = next( filter( - lambda tralbum: f"{tralbum["tralbum_type"]}:{tralbum["tralbum_id"]}" + lambda tralbum: f"{tralbum['tralbum_type']}:{tralbum['tralbum_id']}" == wanted_id, tralbums, ) ) except StopIteration: raise BCFreeDownloadError("Could not find album in collection") - sale_id = f"{tralbum["sale_item_type"]}{tralbum["sale_item_id"]}" + sale_id = f"{tralbum['sale_item_type']}{tralbum['sale_item_id']}" if sale_id not in redownload_urls: raise BCFreeDownloadError("Could not find album download URL in collection") download_url = redownload_urls[sale_id] @@ -299,7 +296,7 @@ class BCFreeDownloader: info = BCFreeDownloader.get_label_info(soup) for release in info["releases"]: - logger.info(f"Downloading {release["url"]}") + logger.info(f"Downloading {release['url']}") soup = self.get_url_soup(release["url"]) try: From cd2825e1e6d3c0498661e0949b30b4e66216141c Mon Sep 17 00:00:00 2001 From: 7x11x13 Date: Fri, 21 Mar 2025 13:47:03 -0400 Subject: [PATCH 3/5] Update README --- README.md | 58 +++++++++++++++++++++++++++++++------------------------ 1 file changed, 33 insertions(+), 25 deletions(-) diff --git a/README.md b/README.md index f581d92..dfe11ac 100644 --- a/README.md +++ b/README.md @@ -1,12 +1,13 @@ # free-bandcamp-downloader -Download free and $0 minimum name-your-price albums and tracks from Bandcamp (including ones that are sent to email), +Download free and $0 minimum name-your-price albums and tracks from Bandcamp (including ones that are sent to email), and tag them with data from the Bandcamp page. Also able to download items in your collection, if login cookies are supplied using the `--cookies` or `--identity` argument. ## Installation Install with pip + ``` pip install free-bandcamp-downloader ``` @@ -21,33 +22,40 @@ argument which you must supply the value of your "identity" cookie. ``` Usage: - bcdl-free (-a | -l )[--force][--no-unzip][-d | --dir ][-e | --email ] - [-z | --zipcode ][-c | --country ][-f | --format ] - [--cookies ][--identity ][--debug] - bcdl-free setdefault [-d | --dir ][-e | --email ][-z | --zipcode ] - [-c | --country ][-f | --format ] + bcdl-free setdefault [-d ] [-e ] [-z ] + [-c ] [-f ] bcdl-free defaults bcdl-free clear - bcdl-free (-h | --help) - bcdl-free --version + bcdl-free -h | --help | --version + bcdl-free [--debug] [--force] [--no-unzip] [-al] + [-d ] [-e ] [-z ] [-c ] [-f ] + [--cookies ] [--identity ] [--download-history-file ] + URL... + +Arguments: + URL URL to download. Can be a link to a label or release page + +Subcommands: + setdefaults set default configuration options + defaults list default configuration options + clear clear default configuration options + Options: - -h --help Show this screen - --version Show version - -a Download the album at URL - -l Download all free albums of the label at URL - --force Download even if album has been downloaded before - --no-unzip Don't unzip downloaded albums - setdefault Set default options - defaults List the default options - clear Clear download history - -d --dir Set download directory - -c --country Set country - -z --zipcode Set zipcode - -e --email Set email (set to 'auto' to automatically download from a disposable email) - -f --format Set format - --cookies Path to cookies.txt file so albums in your collection can be downloaded - --identity Value of identity cookie so albums in your collection can be downloaded - --debug Set loglevel to debug + -h --help Show this screen + --version Show version + --force Download even if album has been downloaded before + --no-unzip Don't unzip downloaded albums + --debug Set loglevel to debug + -a -l Dummy options, for backwards compatibility + -d --dir Set download directory + -c --country Set country + -z --zipcode Set zipcode + -e --email Set email (set to 'auto' to automatically download from a disposable email) + -f --format Set format + --cookies Path to cookies.txt file so albums in your collection can be downloaded + --identity Value of identity cookie so albums in your collection can be downloaded + --download-history-file Path to history file containing downloaded albums + Formats: - FLAC - V0MP3 From 7be18d9d8e6a5d52d307b92a3dcf814b1e066a64 Mon Sep 17 00:00:00 2001 From: 7x11x13 Date: Fri, 21 Mar 2025 13:50:02 -0400 Subject: [PATCH 4/5] Update README --- README.md | 11 +++++++++-- 1 file changed, 9 insertions(+), 2 deletions(-) diff --git a/README.md b/README.md index dfe11ac..230dd02 100644 --- a/README.md +++ b/README.md @@ -6,10 +6,17 @@ supplied using the `--cookies` or `--identity` argument. ## Installation -Install with pip +With pip: ``` -pip install free-bandcamp-downloader +$ pip install free-bandcamp-downloader +$ bcdl-free +``` + +With [uv](https://docs.astral.sh/uv/getting-started/installation/): + +``` +$ uvx --from free-bandcamp-downloader bcdl-free ``` ## Note on passing cookies From 49ffe5626033ca2e4d59df63fad82dcd23cdb216 Mon Sep 17 00:00:00 2001 From: 7x11x13 Date: Fri, 21 Mar 2025 13:51:01 -0400 Subject: [PATCH 5/5] Update version --- pyproject.toml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/pyproject.toml b/pyproject.toml index e2e94ef..3faa9ae 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [tool.poetry] name = "free-bandcamp-downloader" -version = "0.4.3" +version = "0.5.0" description = "Download free and name-your-price albums from Bandcamp in lossless quality" authors = ["7x11x13 "] license = "MIT"