diff --git a/free_bandcamp_downloader/__main__.py b/free_bandcamp_downloader/__main__.py index ce8602c..4cf370a 100644 --- a/free_bandcamp_downloader/__main__.py +++ b/free_bandcamp_downloader/__main__.py @@ -1,20 +1,22 @@ """Download free albums and tracks from Bandcamp + Usage: - bcdl-free [--debug] [--force] [--no-unzip] [-al] - [-d ] [-e ] [-z ] [-c ] [-f ] - [--cookies ] [--identity ] URL... bcdl-free setdefault [-d ] [-e ] [-z ] [-c ] [-f ] bcdl-free defaults bcdl-free clear bcdl-free -h | --help | --version + bcdl-free [--debug] [--force] [--no-unzip] [-al] + [-d ] [-e ] [-z ] [-c ] [-f ] + [--cookies ] [--identity ] [--download-history-file ] + URL... Arguments: URL URL to download. Can be a link to a label or release page Subcommands: - defaults list default configuration options setdefaults set default configuration options + defaults list default configuration options clear clear default configuration options Options: @@ -31,6 +33,7 @@ Options: -f --format Set format --cookies Path to cookies.txt file so albums in your collection can be downloaded --identity Value of identity cookie so albums in your collection can be downloaded + --download-history-file Path to history file containing downloaded albums Formats: - FLAC @@ -43,424 +46,48 @@ Formats: - AIFF """ -import atexit import dataclasses -import glob -import html -import json -import logging +import sys import os import pprint -import re -import sys -import time -import zipfile -from configparser import ConfigParser -from dataclasses import dataclass -from http.cookiejar import MozillaCookieJar -from typing import Dict, Optional, Set -from urllib.parse import urljoin - -import mutagen -import pyrfc6266 -import requests -from bs4 import BeautifulSoup +from typing import Set, Tuple from docopt import docopt -from tqdm import tqdm -from guerrillamail import GuerrillaMailSession +from configparser import ConfigParser -from free_bandcamp_downloader import __version__, logger -from free_bandcamp_downloader.bandcamp_http_adapter import BandcampHTTPAdapter +from free_bandcamp_downloader import __version__ +from free_bandcamp_downloader.bc_free_downloader import * +from free_bandcamp_downloader import logger -@dataclass -class BCFreeDownloaderOptions: - country: str = "United States" - zipcode: str = "00000" - email: str = "auto" - format: str = "FLAC" - dir: str = "." - - -@dataclass -class BCFreeDownloaderAlbumData: - about: Optional[str] - credits: Optional[str] - tags: Optional[str] - id: str - title: Optional[str] - - -class BCFreeDownloadError(Exception): - pass - - -class BCFreeDownloader: - CHUNK_SIZE = 1024 * 1024 - LINK_REGEX = re.compile(r'') - RETRY_URL_REGEX = re.compile(r'"retry_url":"(?P[^"]*)"') - FORMATS = { - "FLAC": "flac", - "V0MP3": "mp3-v0", - "320MP3": "mp3-320", - "AAC": "aac-hi", - "Ogg": "vorbis", - "ALAC": "alac", - "WAV": "wav", - "AIFF": "aiff-lossless", - } - - def __init__( - self, - options: BCFreeDownloaderOptions, - config_dir: str, - download_history_file: str, - unzip: bool = True, - cookies_file: Optional[str] = None, - identity: Optional[str] = None, - ): - self.options = options - self.config_dir = config_dir - self.download_history_file = download_history_file - self.downloaded: Set[str] = set() # can be URL or ID - self.mail_session = None - self.mail_album_data: Dict[str, BCFreeDownloaderAlbumData] = {} - self.unzip = unzip - self._init_downloaded() - self._init_session(cookies_file, identity) - - def _init_email(self): - self.mail_session = GuerrillaMailSession() - self.options.email = self.mail_session.get_session_state()["email_address"] - - def _init_downloaded(self): - if self.download_history_file: - with open(self.download_history_file, "r") as f: - for line in f: - self.downloaded.add(line.strip()) - - def _init_session(self, cookies_file: Optional[str], identity: Optional[str]): - self.session = requests.Session() - self.session.mount("https://", BandcampHTTPAdapter()) - if cookies_file: - cj = MozillaCookieJar(cookies_file) - cj.load() - self.session.cookies = cj - if identity: - self.session.cookies.set("identity", identity) - - def _download_file( - self, - download_page_url: str, - format: str, - album_data: Optional[BCFreeDownloaderAlbumData] = None, - ) -> str: - r = self.session.get(download_page_url) - r.raise_for_status() - soup = BeautifulSoup(r.text, "html.parser") - album_url = soup.find("div", class_="download-artwork").find("a").attrs["href"] - if album_data is None: - r = self.session.get(album_url) - r.raise_for_status() - id = self._get_album_data_from_soup(BeautifulSoup(r.text, "html.parser")).id - album_data = self.mail_album_data[id] - - data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"]) - download_url = data["digital_items"][0]["downloads"][self.FORMATS[format]][ - "url" - ] - - def download(download_url: str) -> str: - with self.session.get(download_url, stream=True) as r: - r.raise_for_status() - size = int(r.headers["content-length"]) - name = pyrfc6266.requests_response_to_filename(r) - file_name = os.path.join(self.options.dir, name) - with tqdm(total=size, unit="iB", unit_scale=True) as pbar: - with open(file_name, "wb") as f: - for chunk in r.iter_content(chunk_size=self.CHUNK_SIZE): - f.write(chunk) - pbar.update(len(chunk)) - return file_name - - try: - file_name = download(download_url) - except Exception: - statdownload_url = download_url.replace("/download/", "/statdownload/") - with self.session.get(statdownload_url) as r: - r.raise_for_status() - download_url = self.RETRY_URL_REGEX.search(r.text).group("retry_url") - if download_url: - file_name = download(download_url) - else: - # retry requires email address - raise BCFreeDownloadError( - "Download expired. Make sure your payment email is linked " - "to your fan account (Settings > Fan > Payment email addresses)" - ) - - logger.info(f"Downloaded {file_name}") - - if file_name.endswith("zip") and self.unzip: - # Unzip archive - dir_name = file_name[:-4] - with zipfile.ZipFile(file_name, "r") as f: - f.extractall(dir_name) - logger.info(f"Unzipped to {dir_name}. Use --no-unzip to prevent this") - os.remove(file_name) - files = glob.glob(os.path.join(dir_name, "*")) - else: - files = [file_name] - # Tag downloaded audio files with url & comment - logger.info("Setting tags...") - for file in files: - f = mutagen.File(file) - if f is None: - continue - f["website"] = album_url - if album_data.tags: - f["genre"] = album_data.tags - comment = "" - if album_data.about: - comment += album_data.about - if album_data.about and album_data.credits: - comment += "\n\n" - if album_data.credits: - comment += album_data.credits - f["comment"] = comment - f.save() - # successfully downloaded file, add to download history - self.downloaded.add(album_data.id) - if self.download_history_file: - with open(self.download_history_file, "a") as f: - f.write(f"{album_data.id}\n") - - return album_data.id - - @staticmethod - def _get_album_data_from_soup(soup: BeautifulSoup) -> BCFreeDownloaderAlbumData: - about = soup.find("div", class_="tralbum-about") - credits = soup.find("div", class_="tralbum-credits") - tags = [tag.get_text() for tag in soup.find_all("a", class_="tag")] - properties = json.loads( - soup.find("meta", attrs={"name": "bc-page-properties"})["content"] - ) - id = f"{properties['item_type']}:{properties['item_id']}" - - return BCFreeDownloaderAlbumData( - about=about.get_text("\n") if about else None, - credits=credits.get_text("\n") if credits else None, - tags=",".join(sorted(tags)), - id=id, - title=None, +class Config: + def __init__(self): + config_dir = get_config_dir() + # initialize with default config first + # this is combined from the dataclass and CLI params + self.parser = ConfigParser(allow_no_value=True) + self.parser["free-bandcamp-downloader"] = {} + for field in dataclasses.fields(BCFreeDownloaderOptions): + self.parser["free-bandcamp-downloader"][field.name] = field.default + self.parser["free-bandcamp-downloader"]["force"] = "false" + self.parser["free-bandcamp-downloader"]["no-unzip"] = "false" + self.parser["free-bandcamp-downloader"]["download-history-file"] = ( + get_data_dir() + "/downloaded.txt" ) - def _download_purchased_album( - self, user_id: int, album_data: BCFreeDownloaderAlbumData - ): - logger.info("Downloading album from collection...") - logger.debug(f"Searching for album: '{album_data.title}'") - data = { - "fan_id": user_id, - "search_key": album_data.title, - "search_type": "collection", - } - r = self.session.post( - "https://bandcamp.com/api/fancollection/1/search_items", json=data - ) - r.raise_for_status() - results = r.json() - tralbums = results["tralbums"] - redownload_urls = results["redownload_urls"] - try: - tralbum = next( - filter( - lambda tralbum: f"{tralbum['tralbum_type']}:{tralbum['tralbum_id']}" - == album_data.id, - tralbums, - ) - ) - except StopIteration: - raise BCFreeDownloadError("Could not find album in collection") - sale_id = f"{tralbum['sale_item_type']}{tralbum['sale_item_id']}" - if sale_id not in redownload_urls: - raise BCFreeDownloadError("Could not find album download URL in collection") - download_url = redownload_urls[sale_id] - logger.debug(f"Got download URL: {download_url}") - self._download_file(download_url, self.options.format, album_data) - - # check if the album was already downloaded given an id and url - # format of id is [at]: - def is_downloaded(self, id: str, url: str = ""): - return url in self.downloaded or id in self.downloaded - - def _download_album(self, soup: BeautifulSoup, force: bool = False): - album_data = self._get_album_data_from_soup(soup) - url = soup.head.find("meta", attrs={"property": "og:url"})["content"] - if not force and self.is_downloaded(album_data.id, url): - raise BCFreeDownloadError( - f"{url} already downloaded. To download anyways, use option --force" - ) - - logger.debug(f"Album data: {album_data}") - - tralbum_data = soup.find("script", {"data-tralbum": True}).attrs["data-tralbum"] - tralbum_data = json.loads(tralbum_data) - - if not tralbum_data["hasAudio"]: - raise BCFreeDownloadError(f"{url} has no audio. Skipping...") - - head_data = soup.head.find( - "script", {"type": "application/ld+json"}, recursive=False - ).string - - head_data = json.loads(head_data) - head_id = head_data.get("@id") - # fallback if a track link was provided - # track releases have this inAlbum key even if they're standalone - head_data = head_data.get("inAlbum", head_data)["albumRelease"] - # find the albumRelease object that matches the overall album @id link - # this will ensure that strictly the page release is downloaded - head_data = next(obj for obj in head_data if obj["@id"] == head_id) - if "offers" not in head_data: - raise BCFreeDownloadError(f"{url} has no digital download. Skipping...") - - album_data.title = head_data["name"] - - if head_data["offers"]["price"] == 0.0: - if tralbum_data["current"]["require_email"]: - if not self.options.email or self.options.email == "auto": - self._init_email() - logger.info(f"{url} requires email") - email_post_url = urljoin(url, "/email_download") - r = self.session.post( - email_post_url, - data={ - "encoding_name": "none", - "item_id": tralbum_data["current"]["id"], - "item_type": tralbum_data["current"]["type"], - "address": self.options.email, - "country": self.options.country, - "postcode": self.options.zipcode, - }, - ) - r.raise_for_status() - r = r.json() - if not r["ok"]: - raise ValueError(f"Bad response when sending email address: {r}") - self.mail_album_data[album_data.id] = album_data - else: - logger.info(f"{url} does not require email") - self._download_file( - tralbum_data["freeDownloadPage"], self.options.format, album_data - ) - else: - if tralbum_data["is_purchased"]: - collection_info = soup.find( - "script", {"data-tralbum-collect-info": True} - ).attrs["data-tralbum-collect-info"] - collection_info = json.loads(collection_info) - self._download_purchased_album(collection_info["fan_id"], album_data) - else: - raise BCFreeDownloadError( - f"{url} is not free. If you have purchased this album, " - "use the --cookies flag or --identity flag to pass your login cookie." - ) - - def _download_label(self, soup: BeautifulSoup, force: bool = False): - albums = [] - baseurl = soup.head.find("meta", attrs={"property": "og:url"})["content"] - - # bandcamp splits the album between this music-grid html and some json blob - grid = soup.find("ol", id="music-grid") - for li in grid.find_all("li"): - if "display:none" in li.get("style", ""): - continue - data = li["data-item-id"].split("-") - albums += [{"url": li.a["href"], "id": f"{data[0][0]}:{data[1]}"}] - for obj in json.loads(html.unescape(grid.get("data-client-items", {}))): - if obj.get("filtered"): - continue - albums += [{"url": obj["page_url"], "id": f"{obj['type'][0]}:{obj['id']}"}] - - for album in albums: - if album["url"][0] == "/": - album["url"] = urljoin(baseurl, album["url"]) - - # perform a check here for already-downloaded album to prevent mass requests - # to bandcamp for large labels and getting rate limited as a result - if not force and self.is_downloaded(album["id"], album["url"]): - logger.info( - f"{album['url']} already downloaded. To download anyways, use option --force" - ) - continue - - logger.info(f"Downloading {album['url']}") - r = self.session.get(album["url"]) - r.raise_for_status() - soup = BeautifulSoup(r.text, "html.parser") - try: - self._download_album(soup, force) - except BCFreeDownloadError as ex: - logger.info(ex) - - def download_url(self, url: str, force: bool = False): - # detect whether it's a release or label page - r = self.session.get(url) - r.raise_for_status() - soup = BeautifulSoup(r.text, "html.parser") - - try: - url_type = soup.head.find("meta", attrs={"property": "og:type"})["content"] - except Exception: - raise BCFreeDownloadError(f"{url} does not have an og:type property.") - - if url_type == "album" or url_type == "song": - self._download_album(soup, force) - elif url_type == "band": - self._download_label(soup, force) - else: - raise BCFreeDownloadError(f"{url} does not have a valid og:type value") - - def wait_for_email_downloads(self): - checked_ids = set() - while (expected_emails := len(self.mail_album_data)) > 0: - logger.info(f"Waiting for {expected_emails} emails from Bandcamp...") - time.sleep(5) - for email in self.mail_session.get_email_list(): - email_id = email.guid - if email_id not in checked_ids: - checked_ids.add(email_id) - if ( - email.sender == "noreply@bandcamp.com" - and "download" in email.subject - ): - logger.info(f'Received email "{email.subject}"') - content = self.mail_session.get_email(email_id).body - match = self.LINK_REGEX.search(content) - if match: - download_url = match.group("url") - album_id = self._download_file( - download_url, self.options.format - ) - self.mail_album_data.pop(album_id) - else: - logger.error( - f"Could not find download URL in body: {content}" - ) - - -class BCFreeDownloaderConfig: - def __init__(self, config_path: str): - self.config_path = config_path - self.parser = ConfigParser() - self.parser.read(config_path) - atexit.register(self.save) + # read config file + self.config_path = os.path.join(config_dir, "free-bandcamp-downloader.cfg") + if not os.path.exists(self.config_path): + with open(self.config_path, "w") as f: + self.parser.write(f) + return + self.parser.read(self.config_path) def get(self, key): - return self.parser["free-bandcamp-downloader"].get(key, None) + return self.parser["free-bandcamp-downloader"].get(key) def set(self, key, value): + if value is not None: + value = str(value) self.parser["free-bandcamp-downloader"][key] = value def save(self): @@ -471,7 +98,14 @@ class BCFreeDownloaderConfig: return pprint.pformat(dict(self.parser["free-bandcamp-downloader"]), indent=2) -def get_config_dir(): +def options_from_config(config: Config): + options = BCFreeDownloaderOptions() + for field in dataclasses.fields(options): + setattr(options, field.name, config.parser["free-bandcamp-downloader"][field.name]) + return options + + +def get_config_dir() -> str: if "XDG_CONFIG_HOME" in os.environ: config_dir = os.path.join( os.environ["XDG_CONFIG_HOME"], "free-bandcamp-downloader" @@ -480,84 +114,154 @@ def get_config_dir(): config_dir = os.path.join( os.path.expanduser("~"), ".config", "free-bandcamp-downloader" ) + if not os.path.exists(config_dir): + os.makedirs(config_dir) return config_dir -def get_data_dir(): +def get_data_dir() -> str: if "XDG_DATA_HOME" in os.environ: data_dir = os.path.join(os.environ["XDG_DATA_HOME"], "free-bandcamp-downloader") else: data_dir = os.path.join( os.path.expanduser("~"), ".local", "share", "free-bandcamp-downloader" ) + if not os.path.exists(data_dir): + os.makedirs(data_dir) return data_dir -def get_config(data_dir: str, config_dir: str): - download_history_file = os.path.join(data_dir, "downloaded.txt") - default_config = f"""[free-bandcamp-downloader] - country = United States - zipcode = 00000 - email = auto - format = FLAC - dir = . - download_history_file = {download_history_file}""" - config_file = os.path.join(config_dir, "free-bandcamp-downloader.cfg") - if not os.path.exists(config_file): - if not os.path.exists(config_dir): - os.makedirs(config_dir) - with open(config_file, "w") as f: - f.write(default_config) - if not os.path.exists(download_history_file): - if not os.path.exists(data_dir): - os.makedirs(data_dir) - with open(download_history_file, "w") as f: +def is_downloaded(downloaded_set, id: Tuple[str, int], url: str = None) -> bool: + return id in downloaded_set or url in downloaded_set + + +def add_to_dl_file(config: Config, id: Tuple[str, int]): + history_file = config.parser["free-bandcamp-downloader"]["download-history-file"] + with open(history_file, "a") as f: + f.write(f"{id[0][0]}:{id[1]}\n") + + +def get_downloaded(config: Config) -> Set[Tuple[str, int | str]]: + history_file = config.parser["free-bandcamp-downloader"]["download-history-file"] + if not os.path.exists(history_file): + with open(history_file, "w") as f: pass - config = BCFreeDownloaderConfig(config_file) - return config + + downloaded = set() + with open(history_file, "r") as f: + for line in f: + type = line.strip()[:2] + if type == "a:": + type = "album" + data = int(line[2:]) + elif type == "t:": + type = "track" + data = int(line[2:]) + else: + type = "url" + data = line.strip() + downloaded.add((type, data)) + return downloaded + + +def post_download(album_info: AlbumInfo, config: Config): + file_name = album_info["file_name"] + # file list for setting tags + files = [file_name] + unzip = not config.parser.getboolean("free-bandcamp-downloader", "no-unzip") + + # unzip if needed + if unzip and file_name.endswith(".zip"): + files = BCFreeDownloader.unzip_album(file_name) + + logger.info("Setting tags...") + for file in files: + BCFreeDownloader.tag_file(file, album_info["head_data"]) + + +def download_urls(urls: List[str], config: Config): + downloader = BCFreeDownloader(options_from_config(config)) + downloaded = get_downloaded(config) + force = config.parser.getboolean("free-bandcamp-downloader", "force") + + for url in urls: + soup = downloader.get_url_soup(url) + url_info = downloader.get_page_info(soup) + + urltype = url_info and url_info.get("type") + if urltype == "album" or urltype == "song": + tralbum = url_info["info"]["tralbum_data"] + type = tralbum["current"]["type"] + id = tralbum["current"]["id"] + url = tralbum["url"] + if not force and is_downloaded(downloaded, (type, id), url): + logger.error( + f"{url} already downloaded. To download anyways, use --force." + ) + continue + ret = downloader.download_album(soup) + if ret["is_downloaded"]: + add_to_dl_file(config, (type, id)) + downloaded.add((type, id)) + post_download(ret, config) + elif urltype == "band": + for rel in url_info["info"]["releases"]: + type = rel["type"] + id = rel["id"] + url = rel["url"] + if not force and is_downloaded(downloaded, (type, id), url): + logger.error( + f"{url} already downloaded. To download anyways, use --force." + ) + continue + soup = downloader.get_url_soup(url) + ret = downloader.download_album(soup) + if ret["is_downloaded"]: + add_to_dl_file(config, (type, id)) + downloaded.add((type, id)) + post_download(ret, config) + else: + continue + + # finish up downloading + ret = downloader.flush_email_downloads() + for album_info in ret: + type = album_info["tralbum_data"]["current"]["type"] + id = album_info["tralbum_data"]["current"]["id"] + add_to_dl_file(config, (type, id)) + downloaded.add((type, id)) + post_download(album_info, config) def main(): - data_dir = get_data_dir() - config_dir = get_config_dir() - config = get_config(data_dir, config_dir) - options = BCFreeDownloaderOptions() + config = Config() arguments = docopt(__doc__, version=__version__) if arguments["--debug"]: logger.setLevel(logging.DEBUG) - # set options if needed + # set config if needed if arguments["URL"] or arguments["setdefault"]: - for field in dataclasses.fields(options): - option = field.name + for option in config.parser["free-bandcamp-downloader"].keys(): arg = f"--{option}" - if arguments[arg]: - setattr(options, option, arguments[arg]) - else: - setattr(options, option, config.get(option)) - if not getattr(options, option): - logger.error( - f'{option} is not set, use "bcdl-free setdefault {arg} <{option}>"' - ) - sys.exit(1) - if options.format not in BCFreeDownloader.FORMATS: + if arguments.get(arg): + config.set(option, arguments[arg]) + if ( + config.parser["free-bandcamp-downloader"]["format"] + not in BCFreeDownloader.FORMATS + ): logger.error( - f'{options["format"]} is not a valid format. See "bcdl-free -h" for valid formats' + f'{config.parser.get("format")} is not a valid format. See "bcdl-free -h" for valid formats' ) sys.exit(1) + # write to config file if arguments["setdefault"]: - # write arguments to config - for field in dataclasses.fields(options): - option = field.name - arg = f"--{option}" - if arguments[arg]: - config.set(option, arguments[arg]) + config.save() sys.exit(0) if arguments["clear"]: - with open(config.get("download_history_file"), "w"): + with open(config.config_path, "w"): pass sys.exit(0) @@ -566,24 +270,7 @@ def main(): sys.exit(0) if arguments["URL"]: - # init downloader - downloader = BCFreeDownloader( - options, - config_dir, - config.get("download_history_file"), - not arguments["--no-unzip"], - arguments["--cookies"], - arguments["--identity"], - ) - - for url in arguments["URL"]: - try: - downloader.download_url(url) - except BCFreeDownloadError as ex: - logger.info(ex) - - # finish up downloading - downloader.wait_for_email_downloads() + download_urls(arguments["URL"], config) if __name__ == "__main__": diff --git a/free_bandcamp_downloader/bandcamp_http_adapter.py b/free_bandcamp_downloader/bandcamp_http_adapter.py index 5916b7d..48b0c2b 100644 --- a/free_bandcamp_downloader/bandcamp_http_adapter.py +++ b/free_bandcamp_downloader/bandcamp_http_adapter.py @@ -1,4 +1,5 @@ from requests.adapters import HTTPAdapter +from urllib3.util import Retry from urllib3.util.ssl_ import create_urllib3_context diff --git a/free_bandcamp_downloader/bc_free_downloader.py b/free_bandcamp_downloader/bc_free_downloader.py new file mode 100644 index 0000000..0b7b795 --- /dev/null +++ b/free_bandcamp_downloader/bc_free_downloader.py @@ -0,0 +1,451 @@ +import glob +import html +import json +import os +import re +import time +import zipfile +import mutagen +import pyrfc6266 +import requests + +from bs4 import BeautifulSoup +from tqdm import tqdm +from dataclasses import dataclass +from http.cookiejar import MozillaCookieJar +from typing import Dict, List, Optional, Tuple, TypedDict +from urllib.parse import urljoin +from guerrillamail import GuerrillaMailSession + +from free_bandcamp_downloader import logger +from free_bandcamp_downloader.bandcamp_http_adapter import * + + +class DownloadRet(TypedDict): + id: Tuple[str, int] + file_name: str + + +class AlbumInfo(TypedDict): + tralbum_data: Dict + head_data: Dict + is_downloaded: Optional[bool] + email_queued: Optional[bool] + file_name: Optional[str] + + +class LabelReleaseInfo(TypedDict): + type: str + id: Tuple[str, int] + band_id: int + url: str + release_info: Optional[AlbumInfo] + + +class LabelInfo(TypedDict): + label_info: Dict + releases: List[LabelReleaseInfo] + + +class PageInfo(TypedDict): + type: str + info: LabelInfo | AlbumInfo + + +@dataclass +class BCFreeDownloaderOptions: + country: str = "United States" + zipcode: str = "00000" + email: str = "auto" + format: str = "FLAC" + dir: str = "." + cookies: Optional[str] = None + identity: Optional[str] = None + + +class BCFreeDownloadError(Exception): + pass + + +class BCFreeDownloader: + CHUNK_SIZE = 1024 * 1024 + LINK_REGEX = re.compile(r'') + RETRY_URL_REGEX = re.compile(r'"retry_url":"(?P[^"]*)"') + FORMATS = { + "FLAC": "flac", + "V0MP3": "mp3-v0", + "320MP3": "mp3-320", + "AAC": "aac-hi", + "Ogg": "vorbis", + "ALAC": "alac", + "WAV": "wav", + "AIFF": "aiff-lossless", + } + + def __init__(self, options: BCFreeDownloaderOptions): + self.options = options + self.mail_session = None + self.queued_emails = {} # { ("album"|"track", id): {info} } + self.session = None + self.email = None + self._init_session() + + def _init_email(self): + logger.info("Starting mail session...") + if not self.options.email: + self.options.email = "auto" + if self.options.email == "auto": + self.mail_session = GuerrillaMailSession() + self.options.email = self.mail_session.get_session_state()["email_address"] + + def _init_session(self): + self.session = requests.Session() + retries = Retry( + total = 10, + backoff_factor = 10, + backoff_max = 60, + allowed_methods = {"POST", "GET"} + ) + self.session.mount("https://", BandcampHTTPAdapter(max_retries=retries)) + if self.options.cookies: + cj = MozillaCookieJar(self.options.cookies) + cj.load() + self.session.cookies = cj + if self.options.identity: + self.session.cookies.set("identity", self.options.identity) + + def _download_file(self, download_page_url: str, format: str) -> DownloadRet: + soup = self.get_url_soup(download_page_url) + album_url = soup.find("div", class_="download-artwork").find("a").attrs["href"] + + data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"])[ + "digital_items" + ][0] + download_url = data["downloads"][self.FORMATS[format]]["url"] + id = (data["type"], int(data["item_id"])) + + def download(download_url: str) -> str: + with self.get_url(download_url, stream=True) as r: + size = int(r.headers["content-length"]) + name = pyrfc6266.requests_response_to_filename(r) + file_name = os.path.join(self.options.dir, name) + with tqdm(total=size, unit="iB", unit_scale=True) as pbar: + with open(file_name, "wb") as f: + for chunk in r.iter_content(chunk_size=self.CHUNK_SIZE): + f.write(chunk) + pbar.update(len(chunk)) + return file_name + + try: + file_name = download(download_url) + except Exception: + statdownload_url = download_url.replace("/download/", "/statdownload/") + with self.get_url(statdownload_url) as r: + download_url = self.RETRY_URL_REGEX.search(r.text).group("retry_url") + if download_url: + file_name = download(download_url) + else: + # retry requires email address + raise BCFreeDownloadError( + "Download expired. Make sure your payment email is linked " + "to your fan account (Settings > Fan > Payment email addresses)" + ) + + logger.info(f"Downloaded {file_name}") + + return {"id": id, "file_name": file_name} + + # unzip the provided file and return all file paths + @staticmethod + def unzip_album(file_name: str) -> List[str]: + dir_name = file_name[:-4] + with zipfile.ZipFile(file_name, "r") as f: + f.extractall(dir_name) + logger.info(f"Unzipped {file_name}.") + os.remove(file_name) + return glob.glob(os.path.join(dir_name, "*")) + + # Tag downloaded audio file with url & comment + @staticmethod + def tag_file(file_name: str, head_data: Dict): + try: + f = mutagen.File(file_name) + if f is None: + return + + f["website"] = head_data["@id"] + if head_data.get("keywords"): + f["genre"] = head_data["keywords"] + comment = "" + comment += head_data.get("description", "").strip() + comment += "\n\n" + head_data.get("creditText", "") + f["comment"] = comment.strip() + f.save() + except Exception: + # only should happen if the file doesn't support tags + pass + + def _download_purchased_album( + self, user_id: int, tralbum_data: Dict + ) -> DownloadRet: + logger.info("Downloading album from collection...") + logger.debug(f"Searching for album: '{tralbum_data["current"]["title"]}'") + data = { + "fan_id": user_id, + "search_key": tralbum_data["current"]["title"], + "search_type": "collection", + } + results = self.post_url_json( + "https://bandcamp.com/api/fancollection/1/search_items", json=data + ) + tralbums = results["tralbums"] + redownload_urls = results["redownload_urls"] + wanted_id = f"{tralbum_data["item_type"][0]}:{tralbum_data["id"]}" + try: + tralbum = next( + filter( + lambda tralbum: f"{tralbum["tralbum_type"]}:{tralbum["tralbum_id"]}" + == wanted_id, + tralbums, + ) + ) + except StopIteration: + raise BCFreeDownloadError("Could not find album in collection") + sale_id = f"{tralbum["sale_item_type"]}{tralbum["sale_item_id"]}" + if sale_id not in redownload_urls: + raise BCFreeDownloadError("Could not find album download URL in collection") + download_url = redownload_urls[sale_id] + logger.debug(f"Got download URL: {download_url}") + return self._download_file(download_url, self.options.format) + + # download from release page + def download_album(self, soup: BeautifulSoup) -> AlbumInfo: + album_data = BCFreeDownloader.get_album_info(soup) + tralbum_data = album_data["tralbum_data"] + head_data = album_data["head_data"] + album_data["is_downloaded"] = False + album_data["email_queued"] = False + url = tralbum_data["url"] + + logger.debug(f"tralbum data: {tralbum_data}") + logger.debug(f"album head data: {head_data}") + + if not tralbum_data["hasAudio"]: + logger.error(f"{url} has no audio.") + return album_data + + head_id = head_data.get("@id") + # fallback if a track link was provided + # track releases have this inAlbum key even if they're standalone + album_release = head_data.get("inAlbum", head_data)["albumRelease"] + # find the albumRelease object that matches the overall album @id link + # this will ensure that strictly the page release is downloaded + album_release = next(obj for obj in album_release if obj["@id"] == head_id) + + if "offers" not in album_release: + logger.error(f"{url} has no available offers.") + + if tralbum_data["freeDownloadPage"]: + logger.info(f"{url} does not require email") + dlret = self._download_file( + tralbum_data["freeDownloadPage"], self.options.format + ) + elif album_release["offers"]["price"] == 0.0: + logger.info(f"{url} requires email") + if self.mail_session is None: + self._init_email() + email_post_url = urljoin(url, "/email_download") + r = self.post_url_json( + email_post_url, + data={ + "encoding_name": "none", + "item_id": tralbum_data["current"]["id"], + "item_type": tralbum_data["current"]["type"], + "address": self.options.email, + "country": self.options.country, + "postcode": self.options.zipcode, + }, + ) + + if not r["ok"]: + raise ValueError(f"Bad response when sending email address: {r}") + type = tralbum_data["current"]["type"] + id = tralbum_data["current"]["id"] + album_data["email_queued"] = True + self.queued_emails[(type, id)] = album_data + return album_data + elif tralbum_data["is_purchased"]: + collection_info = soup.find( + "script", {"data-tralbum-collect-info": True} + ).attrs["data-tralbum-collect-info"] + collection_info = json.loads(collection_info) + dlret = self._download_purchased_album( + collection_info["fan_id"], tralbum_data + ) + else: + logger.error( + f"{url} is not free. If you have purchased this album, " + "use the --cookies flag or --identity flag to pass your login cookie." + ) + return album_data + + album_data["is_downloaded"] = True + album_data["file_name"] = dlret["file_name"] + + return album_data + + # unconditionally download from release page + def download_label(self, soup: BeautifulSoup) -> LabelInfo: + info = BCFreeDownloader.get_label_info(soup) + + for release in info["releases"]: + logger.info(f"Downloading {release["url"]}") + + soup = self.get_url_soup(release["url"]) + try: + ret = self.download_album(soup) + release["release_info"] = ret + except BCFreeDownloadError as ex: + logger.info(ex) + + return info + + # unconditionally downloads the provided url + # returns either the result of download_album or download_label + # with the `page_type` set to album|song|band + # exception if download error + def download_url(self, url: str, force: bool = False): + soup = self.get_url_soup(url) + + page_info = self.get_page_info(soup) + page_type = page_info.get("type") + if page_type == "album" or page_type == "song": + ret = self.download_album(soup, force) + else: + ret = self.download_label(soup, force) + + ret["page_type"] = page_type + + return ret + + def flush_email_downloads(self) -> List[AlbumInfo]: + checked_ids = set() + downloaded = [] + while len(self.queued_emails) > 0: + logger.info( + f"Waiting for {len(self.queued_emails)} emails from Bandcamp..." + ) + time.sleep(5) + for email in self.mail_session.get_email_list(): + email_id = email.guid + if email_id in checked_ids: + continue + + checked_ids.add(email_id) + if ( + email.sender == "noreply@bandcamp.com" + and "download" in email.subject + ): + logger.info(f'Received email "{email.subject}"') + content = self.mail_session.get_email(email_id).body + match = self.LINK_REGEX.search(content) + if match: + download_url = match.group("url") + dlret = self._download_file(download_url, self.options.format) + self.queued_emails[dlret["id"]]["file_name"] = dlret[ + "file_name" + ] + downloaded.append(self.queued_emails[dlret["id"]]) + self.queued_emails.pop(dlret["id"]) + else: + logger.error(f"Could not find download URL in body: {content}") + return downloaded + + # get_url_x can't be staticmethods because of special session context + def get_url(self, url: str, **kwargs) -> requests.Response: + r = self.session.get(url, **kwargs) + r.raise_for_status() + return r + + def get_url_soup(self, url: str, **kwargs) -> BeautifulSoup: + return BeautifulSoup(self.get_url(url, **kwargs).text, "html.parser") + + def get_url_info(self, url: str, **kwargs) -> PageInfo: + soup = self.get_url_soup(url, **kwargs) + try: + return self.get_page_info(soup) + except Exception: + raise BCFreeDownloadError(f"Could not get page info for {url}") + + def post_url(self, url: str, **kwargs) -> requests.Response: + r = self.session.post(url, **kwargs) + r.raise_for_status() + return r + + def post_url_json(self, url: str, **kwargs) -> Dict: + return self.post_url(url, **kwargs).json() + + # get the album/label info of a bandcamp page + @staticmethod + def get_page_info(soup: BeautifulSoup) -> PageInfo: + page_type = soup.head.find("meta", attrs={"property": "og:type"}).get("content") + + if page_type == "album" or page_type == "song": + return {"type": page_type, "info": BCFreeDownloader.get_album_info(soup)} + if page_type == "band": + return {"type": page_type, "info": BCFreeDownloader.get_label_info(soup)} + else: + # only bandcamp pages are supported + raise BCFreeDownloadError("Page does not have a valid og:type value") + + @staticmethod + def get_label_info(soup: BeautifulSoup) -> LabelInfo: + label_info = soup.find("script", attrs={"data-band": True}) + if label_info is None: + raise BCFreeDownloadError("Page has no data-band script.") + label_info = json.loads(label_info["data-band"]) + + releases = [] + # needed for releases + local_url = label_info["local_url"] + + # bandcamp splits the release between this music-grid html and some json blob + grid = soup.find("ol", id="music-grid") + for li in grid.find_all("li"): + if "display:none" in li.get("style", ""): + continue + + data = li["data-item-id"].split("-") + # most important fields + releases.append( + { + "type": data[0], + "id": int(data[1]), + "url": li.a["href"], + "band_id": int(li["data-band-id"]), + } + ) + for obj in json.loads(html.unescape(grid.get("data-client-items", {}))): + if obj.get("filtered"): + continue + # normalize to fit the other half + obj["url"] = obj.pop("page_url") + releases.append(obj) + + # fixup local urls into global ones + for release in releases: + if release["url"][0] == "/": + release["url"] = urljoin(local_url, release["url"]) + + return {"label_info": label_info, "releases": releases} + + @staticmethod + def get_album_info(soup: BeautifulSoup) -> AlbumInfo: + tralbum_data = soup.find("script", {"data-tralbum": True}).attrs["data-tralbum"] + tralbum_data = json.loads(tralbum_data) + head_data = soup.head.find( + "script", {"type": "application/ld+json"}, recursive=False + ).string + head_data = json.loads(head_data) + + return {"tralbum_data": tralbum_data, "head_data": head_data}