Merge pull request #20 from yoshiyoshyosh/decoupling

decouple downloading logic from cli logic, along with other refactoring
This commit is contained in:
gavin
2025-03-21 13:52:02 -04:00
committed by GitHub
5 changed files with 661 additions and 500 deletions
+1
View File
@@ -3,3 +3,4 @@ env
*.log *.log
__pycache__ __pycache__
notes notes
.DS_Store
+35 -20
View File
@@ -6,9 +6,17 @@ supplied using the `--cookies` or `--identity` argument.
## Installation ## Installation
Install with pip With pip:
``` ```
pip install free-bandcamp-downloader $ pip install free-bandcamp-downloader
$ bcdl-free
```
With [uv](https://docs.astral.sh/uv/getting-started/installation/):
```
$ uvx --from free-bandcamp-downloader bcdl-free
``` ```
## Note on passing cookies ## Note on passing cookies
@@ -21,33 +29,40 @@ argument which you must supply the value of your "identity" cookie.
``` ```
Usage: Usage:
bcdl-free (-a <URL> | -l <URL>)[--force][--no-unzip][-d | --dir <dir>][-e | --email <email>] bcdl-free setdefault [-d <dir>] [-e <email>] [-z <zipcode>]
[-z | --zipcode <zipcode>][-c | --country <country>][-f | --format <format>] [-c <country>] [-f <format>]
[--cookies <file>][--identity <value>][--debug]
bcdl-free setdefault [-d | --dir <dir>][-e | --email <email>][-z | --zipcode <zipcode>]
[-c | --country <country>][-f | --format <format>]
bcdl-free defaults bcdl-free defaults
bcdl-free clear bcdl-free clear
bcdl-free (-h | --help) bcdl-free -h | --help | --version
bcdl-free --version bcdl-free [--debug] [--force] [--no-unzip] [-al]
[-d <dir>] [-e <email>] [-z <zipcode>] [-c <country>] [-f <format>]
[--cookies <file>] [--identity <value>] [--download-history-file <file>]
URL...
Arguments:
URL URL to download. Can be a link to a label or release page
Subcommands:
setdefaults set default configuration options
defaults list default configuration options
clear clear default configuration options
Options: Options:
-h --help Show this screen -h --help Show this screen
--version Show version --version Show version
-a <URL> Download the album at URL
-l <URL> Download all free albums of the label at URL
--force Download even if album has been downloaded before --force Download even if album has been downloaded before
--no-unzip Don't unzip downloaded albums --no-unzip Don't unzip downloaded albums
setdefault Set default options --debug Set loglevel to debug
defaults List the default options -a -l Dummy options, for backwards compatibility
clear Clear download history -d <dir> --dir <dir> Set download directory
-d --dir <dir> Set download directory -c <country> --country <country> Set country
-c --country <country> Set country -z <zipcode> --zipcode <zipcode> Set zipcode
-z --zipcode <zipcode> Set zipcode -e <email> --email <email> Set email (set to 'auto' to automatically download from a disposable email)
-e --email <email> Set email (set to 'auto' to automatically download from a disposable email) -f <format> --format <format> Set format
-f --format <format> Set format
--cookies <file> Path to cookies.txt file so albums in your collection can be downloaded --cookies <file> Path to cookies.txt file so albums in your collection can be downloaded
--identity <value> Value of identity cookie so albums in your collection can be downloaded --identity <value> Value of identity cookie so albums in your collection can be downloaded
--debug Set loglevel to debug --download-history-file <file> Path to history file containing downloaded albums
Formats: Formats:
- FLAC - FLAC
- V0MP3 - V0MP3
+168 -471
View File
@@ -1,4 +1,5 @@
"""Download free albums and tracks from Bandcamp """Download free albums and tracks from Bandcamp
Usage: Usage:
bcdl-free setdefault [-d <dir>] [-e <email>] [-z <zipcode>] bcdl-free setdefault [-d <dir>] [-e <email>] [-z <zipcode>]
[-c <country>] [-f <format>] [-c <country>] [-f <format>]
@@ -8,13 +9,17 @@ Usage:
[-d <dir>] [-e <email>] [-z <zipcode>] [-c <country>] [-f <format>] [-d <dir>] [-e <email>] [-z <zipcode>] [-c <country>] [-f <format>]
[--cookies <file>] [--identity <value>] URL... [--cookies <file>] [--identity <value>] URL...
bcdl-free -h | --help | --version bcdl-free -h | --help | --version
bcdl-free [--debug] [--force] [--no-unzip] [-al]
[-d <dir>] [-e <email>] [-z <zipcode>] [-c <country>] [-f <format>]
[--cookies <file>] [--identity <value>] [--download-history-file <file>]
URL...
Arguments: Arguments:
URL URL to download. Can be a link to a label or release page URL URL to download. Can be a link to a label or release page
Subcommands: Subcommands:
defaults list default configuration options
setdefaults set default configuration options setdefaults set default configuration options
defaults list default configuration options
clear clear default configuration options clear clear default configuration options
Options: Options:
@@ -31,6 +36,7 @@ Options:
-f <format> --format <format> Set format -f <format> --format <format> Set format
--cookies <file> Path to cookies.txt file so albums in your collection can be downloaded --cookies <file> Path to cookies.txt file so albums in your collection can be downloaded
--identity <value> Value of identity cookie so albums in your collection can be downloaded --identity <value> Value of identity cookie so albums in your collection can be downloaded
--download-history-file <file> Path to history file containing downloaded albums
Formats: Formats:
- FLAC - FLAC
@@ -43,424 +49,53 @@ Formats:
- AIFF - AIFF
""" """
import atexit
import dataclasses import dataclasses
import glob
import html
import json
import logging import logging
import sys
import os import os
import pprint import pprint
import re from typing import List, Set, Tuple
import sys
import time
import zipfile
from configparser import ConfigParser
from dataclasses import dataclass
from http.cookiejar import MozillaCookieJar
from typing import Dict, Optional, Set
from urllib.parse import urljoin
import mutagen
import pyrfc6266
import requests
from bs4 import BeautifulSoup
from docopt import docopt from docopt import docopt
from tqdm import tqdm from configparser import ConfigParser
from guerrillamail import GuerrillaMailSession
from free_bandcamp_downloader import __version__, logger from free_bandcamp_downloader import __version__
from free_bandcamp_downloader.bandcamp_http_adapter import BandcampHTTPAdapter from free_bandcamp_downloader.bc_free_downloader import (
AlbumInfo,
BCFreeDownloader,
BCFreeDownloaderOptions,
)
from free_bandcamp_downloader import logger
@dataclass class Config:
class BCFreeDownloaderOptions: def __init__(self):
country: str = "United States" config_dir = get_config_dir()
zipcode: str = "00000" # initialize with default config first
email: str = "auto" # this is combined from the dataclass and CLI params
format: str = "FLAC" self.parser = ConfigParser(allow_no_value=True)
dir: str = "." self.parser["free-bandcamp-downloader"] = {}
for field in dataclasses.fields(BCFreeDownloaderOptions):
self.parser["free-bandcamp-downloader"][field.name] = field.default
@dataclass self.parser["free-bandcamp-downloader"]["force"] = "false"
class BCFreeDownloaderAlbumData: self.parser["free-bandcamp-downloader"]["no-unzip"] = "false"
about: Optional[str] self.parser["free-bandcamp-downloader"]["download-history-file"] = (
credits: Optional[str] get_data_dir() + "/downloaded.txt"
tags: Optional[str]
id: str
title: Optional[str]
class BCFreeDownloadError(Exception):
pass
class BCFreeDownloader:
CHUNK_SIZE = 1024 * 1024
LINK_REGEX = re.compile(r'<a href="(?P<url>[^"]*)">')
RETRY_URL_REGEX = re.compile(r'"retry_url":"(?P<retry_url>[^"]*)"')
FORMATS = {
"FLAC": "flac",
"V0MP3": "mp3-v0",
"320MP3": "mp3-320",
"AAC": "aac-hi",
"Ogg": "vorbis",
"ALAC": "alac",
"WAV": "wav",
"AIFF": "aiff-lossless",
}
def __init__(
self,
options: BCFreeDownloaderOptions,
config_dir: str,
download_history_file: str,
unzip: bool = True,
cookies_file: Optional[str] = None,
identity: Optional[str] = None,
):
self.options = options
self.config_dir = config_dir
self.download_history_file = download_history_file
self.downloaded: Set[str] = set() # can be URL or ID
self.mail_session = None
self.mail_album_data: Dict[str, BCFreeDownloaderAlbumData] = {}
self.unzip = unzip
self._init_downloaded()
self._init_session(cookies_file, identity)
def _init_email(self):
self.mail_session = GuerrillaMailSession()
self.options.email = self.mail_session.get_session_state()["email_address"]
def _init_downloaded(self):
if self.download_history_file:
with open(self.download_history_file, "r") as f:
for line in f:
self.downloaded.add(line.strip())
def _init_session(self, cookies_file: Optional[str], identity: Optional[str]):
self.session = requests.Session()
self.session.mount("https://", BandcampHTTPAdapter())
if cookies_file:
cj = MozillaCookieJar(cookies_file)
cj.load()
self.session.cookies = cj
if identity:
self.session.cookies.set("identity", identity)
def _download_file(
self,
download_page_url: str,
format: str,
album_data: Optional[BCFreeDownloaderAlbumData] = None,
) -> str:
r = self.session.get(download_page_url)
r.raise_for_status()
soup = BeautifulSoup(r.text, "html.parser")
album_url = soup.find("div", class_="download-artwork").find("a").attrs["href"]
if album_data is None:
r = self.session.get(album_url)
r.raise_for_status()
id = self._get_album_data_from_soup(BeautifulSoup(r.text, "html.parser")).id
album_data = self.mail_album_data[id]
data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"])
download_url = data["digital_items"][0]["downloads"][self.FORMATS[format]][
"url"
]
def download(download_url: str) -> str:
with self.session.get(download_url, stream=True) as r:
r.raise_for_status()
size = int(r.headers["content-length"])
name = pyrfc6266.requests_response_to_filename(r)
file_name = os.path.join(self.options.dir, name)
with tqdm(total=size, unit="iB", unit_scale=True) as pbar:
with open(file_name, "wb") as f:
for chunk in r.iter_content(chunk_size=self.CHUNK_SIZE):
f.write(chunk)
pbar.update(len(chunk))
return file_name
try:
file_name = download(download_url)
except Exception:
statdownload_url = download_url.replace("/download/", "/statdownload/")
with self.session.get(statdownload_url) as r:
r.raise_for_status()
download_url = self.RETRY_URL_REGEX.search(r.text).group("retry_url")
if download_url:
file_name = download(download_url)
else:
# retry requires email address
raise BCFreeDownloadError(
"Download expired. Make sure your payment email is linked "
"to your fan account (Settings > Fan > Payment email addresses)"
) )
logger.info(f"Downloaded {file_name}") # read config file
self.config_path = os.path.join(config_dir, "free-bandcamp-downloader.cfg")
if file_name.endswith("zip") and self.unzip: if not os.path.exists(self.config_path):
# Unzip archive with open(self.config_path, "w") as f:
dir_name = file_name[:-4] self.parser.write(f)
with zipfile.ZipFile(file_name, "r") as f: return
f.extractall(dir_name) self.parser.read(self.config_path)
logger.info(f"Unzipped to {dir_name}. Use --no-unzip to prevent this")
os.remove(file_name)
files = glob.glob(os.path.join(dir_name, "*"))
else:
files = [file_name]
# Tag downloaded audio files with url & comment
logger.info("Setting tags...")
for file in files:
f = mutagen.File(file)
if f is None:
continue
f["website"] = album_url
if album_data.tags:
f["genre"] = album_data.tags
comment = ""
if album_data.about:
comment += album_data.about
if album_data.about and album_data.credits:
comment += "\n\n"
if album_data.credits:
comment += album_data.credits
f["comment"] = comment
f.save()
# successfully downloaded file, add to download history
self.downloaded.add(album_data.id)
if self.download_history_file:
with open(self.download_history_file, "a") as f:
f.write(f"{album_data.id}\n")
return album_data.id
@staticmethod
def _get_album_data_from_soup(soup: BeautifulSoup) -> BCFreeDownloaderAlbumData:
about = soup.find("div", class_="tralbum-about")
credits = soup.find("div", class_="tralbum-credits")
tags = [tag.get_text() for tag in soup.find_all("a", class_="tag")]
properties = json.loads(
soup.find("meta", attrs={"name": "bc-page-properties"})["content"]
)
id = f"{properties['item_type']}:{properties['item_id']}"
return BCFreeDownloaderAlbumData(
about=about.get_text("\n") if about else None,
credits=credits.get_text("\n") if credits else None,
tags=",".join(sorted(tags)),
id=id,
title=None,
)
def _download_purchased_album(
self, user_id: int, album_data: BCFreeDownloaderAlbumData
):
logger.info("Downloading album from collection...")
logger.debug(f"Searching for album: '{album_data.title}'")
data = {
"fan_id": user_id,
"search_key": album_data.title,
"search_type": "collection",
}
r = self.session.post(
"https://bandcamp.com/api/fancollection/1/search_items", json=data
)
r.raise_for_status()
results = r.json()
tralbums = results["tralbums"]
redownload_urls = results["redownload_urls"]
try:
tralbum = next(
filter(
lambda tralbum: f"{tralbum['tralbum_type']}:{tralbum['tralbum_id']}"
== album_data.id,
tralbums,
)
)
except StopIteration:
raise BCFreeDownloadError("Could not find album in collection")
sale_id = f"{tralbum['sale_item_type']}{tralbum['sale_item_id']}"
if sale_id not in redownload_urls:
raise BCFreeDownloadError("Could not find album download URL in collection")
download_url = redownload_urls[sale_id]
logger.debug(f"Got download URL: {download_url}")
self._download_file(download_url, self.options.format, album_data)
# check if the album was already downloaded given an id and url
# format of id is [at]:<id>
def is_downloaded(self, id: str, url: str = ""):
return url in self.downloaded or id in self.downloaded
def _download_album(self, soup: BeautifulSoup, force: bool = False):
album_data = self._get_album_data_from_soup(soup)
url = soup.head.find("meta", attrs={"property": "og:url"})["content"]
if not force and self.is_downloaded(album_data.id, url):
raise BCFreeDownloadError(
f"{url} already downloaded. To download anyways, use option --force"
)
logger.debug(f"Album data: {album_data}")
tralbum_data = soup.find("script", {"data-tralbum": True}).attrs["data-tralbum"]
tralbum_data = json.loads(tralbum_data)
if not tralbum_data["hasAudio"]:
raise BCFreeDownloadError(f"{url} has no audio. Skipping...")
head_data = soup.head.find(
"script", {"type": "application/ld+json"}, recursive=False
).string
head_data = json.loads(head_data)
head_id = head_data.get("@id")
# fallback if a track link was provided
# track releases have this inAlbum key even if they're standalone
head_data = head_data.get("inAlbum", head_data)["albumRelease"]
# find the albumRelease object that matches the overall album @id link
# this will ensure that strictly the page release is downloaded
head_data = next(obj for obj in head_data if obj["@id"] == head_id)
if "offers" not in head_data:
raise BCFreeDownloadError(f"{url} has no digital download. Skipping...")
album_data.title = head_data["name"]
if head_data["offers"]["price"] == 0.0:
if tralbum_data["current"]["require_email"]:
if not self.options.email or self.options.email == "auto":
self._init_email()
logger.info(f"{url} requires email")
email_post_url = urljoin(url, "/email_download")
r = self.session.post(
email_post_url,
data={
"encoding_name": "none",
"item_id": tralbum_data["current"]["id"],
"item_type": tralbum_data["current"]["type"],
"address": self.options.email,
"country": self.options.country,
"postcode": self.options.zipcode,
},
)
r.raise_for_status()
r = r.json()
if not r["ok"]:
raise ValueError(f"Bad response when sending email address: {r}")
self.mail_album_data[album_data.id] = album_data
else:
logger.info(f"{url} does not require email")
self._download_file(
tralbum_data["freeDownloadPage"], self.options.format, album_data
)
else:
if tralbum_data["is_purchased"]:
collection_info = soup.find(
"script", {"data-tralbum-collect-info": True}
).attrs["data-tralbum-collect-info"]
collection_info = json.loads(collection_info)
self._download_purchased_album(collection_info["fan_id"], album_data)
else:
raise BCFreeDownloadError(
f"{url} is not free. If you have purchased this album, "
"use the --cookies flag or --identity flag to pass your login cookie."
)
def _download_label(self, soup: BeautifulSoup, force: bool = False):
albums = []
baseurl = soup.head.find("meta", attrs={"property": "og:url"})["content"]
# bandcamp splits the album between this music-grid html and some json blob
grid = soup.find("ol", id="music-grid")
for li in grid.find_all("li"):
if "display:none" in li.get("style", ""):
continue
data = li["data-item-id"].split("-")
albums += [{"url": li.a["href"], "id": f"{data[0][0]}:{data[1]}"}]
for obj in json.loads(html.unescape(grid.get("data-client-items", {}))):
if obj.get("filtered"):
continue
albums += [{"url": obj["page_url"], "id": f"{obj['type'][0]}:{obj['id']}"}]
for album in albums:
if album["url"][0] == "/":
album["url"] = urljoin(baseurl, album["url"])
# perform a check here for already-downloaded album to prevent mass requests
# to bandcamp for large labels and getting rate limited as a result
if not force and self.is_downloaded(album["id"], album["url"]):
logger.info(
f"{album['url']} already downloaded. To download anyways, use option --force"
)
continue
logger.info(f"Downloading {album['url']}")
r = self.session.get(album["url"])
r.raise_for_status()
soup = BeautifulSoup(r.text, "html.parser")
try:
self._download_album(soup, force)
except BCFreeDownloadError as ex:
logger.info(ex)
def download_url(self, url: str, force: bool = False):
# detect whether it's a release or label page
r = self.session.get(url)
r.raise_for_status()
soup = BeautifulSoup(r.text, "html.parser")
try:
url_type = soup.head.find("meta", attrs={"property": "og:type"})["content"]
except Exception:
raise BCFreeDownloadError(f"{url} does not have an og:type property.")
if url_type == "album" or url_type == "song":
self._download_album(soup, force)
elif url_type == "band":
self._download_label(soup, force)
else:
raise BCFreeDownloadError(f"{url} does not have a valid og:type value")
def wait_for_email_downloads(self):
checked_ids = set()
while (expected_emails := len(self.mail_album_data)) > 0:
logger.info(f"Waiting for {expected_emails} emails from Bandcamp...")
time.sleep(5)
for email in self.mail_session.get_email_list():
email_id = email.guid
if email_id not in checked_ids:
checked_ids.add(email_id)
if (
email.sender == "noreply@bandcamp.com"
and "download" in email.subject
):
logger.info(f'Received email "{email.subject}"')
content = self.mail_session.get_email(email_id).body
match = self.LINK_REGEX.search(content)
if match:
download_url = match.group("url")
album_id = self._download_file(
download_url, self.options.format
)
self.mail_album_data.pop(album_id)
else:
logger.error(
f"Could not find download URL in body: {content}"
)
class BCFreeDownloaderConfig:
def __init__(self, config_path: str):
self.config_path = config_path
self.parser = ConfigParser()
self.parser.read(config_path)
atexit.register(self.save)
def get(self, key): def get(self, key):
return self.parser["free-bandcamp-downloader"].get(key, None) return self.parser["free-bandcamp-downloader"].get(key)
def set(self, key, value): def set(self, key, value):
if value is not None:
value = str(value)
self.parser["free-bandcamp-downloader"][key] = value self.parser["free-bandcamp-downloader"][key] = value
def save(self): def save(self):
@@ -471,7 +106,16 @@ class BCFreeDownloaderConfig:
return pprint.pformat(dict(self.parser["free-bandcamp-downloader"]), indent=2) return pprint.pformat(dict(self.parser["free-bandcamp-downloader"]), indent=2)
def get_config_dir(): def options_from_config(config: Config):
options = BCFreeDownloaderOptions()
for field in dataclasses.fields(options):
setattr(
options, field.name, config.parser["free-bandcamp-downloader"][field.name]
)
return options
def get_config_dir() -> str:
if "XDG_CONFIG_HOME" in os.environ: if "XDG_CONFIG_HOME" in os.environ:
config_dir = os.path.join( config_dir = os.path.join(
os.environ["XDG_CONFIG_HOME"], "free-bandcamp-downloader" os.environ["XDG_CONFIG_HOME"], "free-bandcamp-downloader"
@@ -480,84 +124,154 @@ def get_config_dir():
config_dir = os.path.join( config_dir = os.path.join(
os.path.expanduser("~"), ".config", "free-bandcamp-downloader" os.path.expanduser("~"), ".config", "free-bandcamp-downloader"
) )
if not os.path.exists(config_dir):
os.makedirs(config_dir)
return config_dir return config_dir
def get_data_dir(): def get_data_dir() -> str:
if "XDG_DATA_HOME" in os.environ: if "XDG_DATA_HOME" in os.environ:
data_dir = os.path.join(os.environ["XDG_DATA_HOME"], "free-bandcamp-downloader") data_dir = os.path.join(os.environ["XDG_DATA_HOME"], "free-bandcamp-downloader")
else: else:
data_dir = os.path.join( data_dir = os.path.join(
os.path.expanduser("~"), ".local", "share", "free-bandcamp-downloader" os.path.expanduser("~"), ".local", "share", "free-bandcamp-downloader"
) )
if not os.path.exists(data_dir):
os.makedirs(data_dir)
return data_dir return data_dir
def get_config(data_dir: str, config_dir: str): def is_downloaded(downloaded_set, id: Tuple[str, int], url: str = None) -> bool:
download_history_file = os.path.join(data_dir, "downloaded.txt") return id in downloaded_set or url in downloaded_set
default_config = f"""[free-bandcamp-downloader]
country = United States
zipcode = 00000 def add_to_dl_file(config: Config, id: Tuple[str, int]):
email = auto history_file = config.parser["free-bandcamp-downloader"]["download-history-file"]
format = FLAC with open(history_file, "a") as f:
dir = . f.write(f"{id[0][0]}:{id[1]}\n")
download_history_file = {download_history_file}"""
config_file = os.path.join(config_dir, "free-bandcamp-downloader.cfg")
if not os.path.exists(config_file): def get_downloaded(config: Config) -> Set[Tuple[str, int | str]]:
if not os.path.exists(config_dir): history_file = config.parser["free-bandcamp-downloader"]["download-history-file"]
os.makedirs(config_dir) if not os.path.exists(history_file):
with open(config_file, "w") as f: with open(history_file, "w") as f:
f.write(default_config)
if not os.path.exists(download_history_file):
if not os.path.exists(data_dir):
os.makedirs(data_dir)
with open(download_history_file, "w") as f:
pass pass
config = BCFreeDownloaderConfig(config_file)
return config downloaded = set()
with open(history_file, "r") as f:
for line in f:
type = line.strip()[:2]
if type == "a:":
type = "album"
data = int(line[2:])
elif type == "t:":
type = "track"
data = int(line[2:])
else:
type = "url"
data = line.strip()
downloaded.add((type, data))
return downloaded
def post_download(album_info: AlbumInfo, config: Config):
file_name = album_info["file_name"]
# file list for setting tags
files = [file_name]
unzip = not config.parser.getboolean("free-bandcamp-downloader", "no-unzip")
# unzip if needed
if unzip and file_name.endswith(".zip"):
files = BCFreeDownloader.unzip_album(file_name)
logger.info("Setting tags...")
for file in files:
BCFreeDownloader.tag_file(file, album_info["head_data"])
def download_urls(urls: List[str], config: Config):
downloader = BCFreeDownloader(options_from_config(config))
downloaded = get_downloaded(config)
force = config.parser.getboolean("free-bandcamp-downloader", "force")
for url in urls:
soup = downloader.get_url_soup(url)
url_info = downloader.get_page_info(soup)
urltype = url_info and url_info.get("type")
if urltype == "album" or urltype == "song":
tralbum = url_info["info"]["tralbum_data"]
type = tralbum["current"]["type"]
id = tralbum["current"]["id"]
url = tralbum["url"]
if not force and is_downloaded(downloaded, (type, id), url):
logger.error(
f"{url} already downloaded. To download anyways, use --force."
)
continue
ret = downloader.download_album(soup)
if ret["is_downloaded"]:
add_to_dl_file(config, (type, id))
downloaded.add((type, id))
post_download(ret, config)
elif urltype == "band":
for rel in url_info["info"]["releases"]:
type = rel["type"]
id = rel["id"]
url = rel["url"]
if not force and is_downloaded(downloaded, (type, id), url):
logger.error(
f"{url} already downloaded. To download anyways, use --force."
)
continue
soup = downloader.get_url_soup(url)
ret = downloader.download_album(soup)
if ret["is_downloaded"]:
add_to_dl_file(config, (type, id))
downloaded.add((type, id))
post_download(ret, config)
else:
continue
# finish up downloading
ret = downloader.flush_email_downloads()
for album_info in ret:
type = album_info["tralbum_data"]["current"]["type"]
id = album_info["tralbum_data"]["current"]["id"]
add_to_dl_file(config, (type, id))
downloaded.add((type, id))
post_download(album_info, config)
def main(): def main():
data_dir = get_data_dir() config = Config()
config_dir = get_config_dir()
config = get_config(data_dir, config_dir)
options = BCFreeDownloaderOptions()
arguments = docopt(__doc__, version=__version__) arguments = docopt(__doc__, version=__version__)
if arguments["--debug"]: if arguments["--debug"]:
logger.setLevel(logging.DEBUG) logger.setLevel(logging.DEBUG)
# set options if needed # set config if needed
if arguments["URL"] or arguments["setdefault"]: if arguments["URL"] or arguments["setdefault"]:
for field in dataclasses.fields(options): for option in config.parser["free-bandcamp-downloader"].keys():
option = field.name
arg = f"--{option}" arg = f"--{option}"
if arguments[arg]: if arguments.get(arg):
setattr(options, option, arguments[arg]) config.set(option, arguments[arg])
else: if (
setattr(options, option, config.get(option)) config.parser["free-bandcamp-downloader"]["format"]
if not getattr(options, option): not in BCFreeDownloader.FORMATS
):
logger.error( logger.error(
f'{option} is not set, use "bcdl-free setdefault {arg} <{option}>"' f'{config.parser.get("format")} is not a valid format. See "bcdl-free -h" for valid formats'
)
sys.exit(1)
if options.format not in BCFreeDownloader.FORMATS:
logger.error(
f'{options["format"]} is not a valid format. See "bcdl-free -h" for valid formats'
) )
sys.exit(1) sys.exit(1)
# write to config file
if arguments["setdefault"]: if arguments["setdefault"]:
# write arguments to config config.save()
for field in dataclasses.fields(options):
option = field.name
arg = f"--{option}"
if arguments[arg]:
config.set(option, arguments[arg])
sys.exit(0) sys.exit(0)
if arguments["clear"]: if arguments["clear"]:
with open(config.get("download_history_file"), "w"): with open(config.config_path, "w"):
pass pass
sys.exit(0) sys.exit(0)
@@ -566,24 +280,7 @@ def main():
sys.exit(0) sys.exit(0)
if arguments["URL"]: if arguments["URL"]:
# init downloader download_urls(arguments["URL"], config)
downloader = BCFreeDownloader(
options,
config_dir,
config.get("download_history_file"),
not arguments["--no-unzip"],
arguments["--cookies"],
arguments["--identity"],
)
for url in arguments["URL"]:
try:
downloader.download_url(url)
except BCFreeDownloadError as ex:
logger.info(ex)
# finish up downloading
downloader.wait_for_email_downloads()
if __name__ == "__main__": if __name__ == "__main__":
@@ -0,0 +1,448 @@
import glob
import html
import json
import os
import re
import time
import zipfile
import mutagen
import pyrfc6266
import requests
from bs4 import BeautifulSoup
from tqdm import tqdm
from dataclasses import dataclass
from http.cookiejar import MozillaCookieJar
from typing import Dict, List, Optional, Tuple, TypedDict
from urllib.parse import urljoin
from guerrillamail import GuerrillaMailSession
from urllib3 import Retry
from free_bandcamp_downloader import logger
from free_bandcamp_downloader.bandcamp_http_adapter import BandcampHTTPAdapter
class DownloadRet(TypedDict):
id: Tuple[str, int]
file_name: str
class AlbumInfo(TypedDict):
tralbum_data: Dict
head_data: Dict
is_downloaded: Optional[bool]
email_queued: Optional[bool]
file_name: Optional[str]
class LabelReleaseInfo(TypedDict):
type: str
id: Tuple[str, int]
band_id: int
url: str
release_info: Optional[AlbumInfo]
class LabelInfo(TypedDict):
label_info: Dict
releases: List[LabelReleaseInfo]
class PageInfo(TypedDict):
type: str
info: LabelInfo | AlbumInfo
@dataclass
class BCFreeDownloaderOptions:
country: str = "United States"
zipcode: str = "00000"
email: str = "auto"
format: str = "FLAC"
dir: str = "."
cookies: Optional[str] = None
identity: Optional[str] = None
class BCFreeDownloadError(Exception):
pass
class BCFreeDownloader:
CHUNK_SIZE = 1024 * 1024
LINK_REGEX = re.compile(r'<a href="(?P<url>[^"]*)">')
RETRY_URL_REGEX = re.compile(r'"retry_url":"(?P<retry_url>[^"]*)"')
FORMATS = {
"FLAC": "flac",
"V0MP3": "mp3-v0",
"320MP3": "mp3-320",
"AAC": "aac-hi",
"Ogg": "vorbis",
"ALAC": "alac",
"WAV": "wav",
"AIFF": "aiff-lossless",
}
def __init__(self, options: BCFreeDownloaderOptions):
self.options = options
self.mail_session = None
self.queued_emails = {} # { ("album"|"track", id): {info} }
self.session = None
self.email = None
self._init_session()
def _init_email(self):
logger.info("Starting mail session...")
if not self.options.email:
self.options.email = "auto"
if self.options.email == "auto":
self.mail_session = GuerrillaMailSession()
self.options.email = self.mail_session.get_session_state()["email_address"]
def _init_session(self):
self.session = requests.Session()
retries = Retry(
total=10, backoff_factor=10, backoff_max=60, allowed_methods={"POST", "GET"}
)
self.session.mount("https://", BandcampHTTPAdapter(max_retries=retries))
if self.options.cookies:
cj = MozillaCookieJar(self.options.cookies)
cj.load()
self.session.cookies = cj
if self.options.identity:
self.session.cookies.set("identity", self.options.identity)
def _download_file(self, download_page_url: str, format: str) -> DownloadRet:
soup = self.get_url_soup(download_page_url)
data = json.loads(soup.find("div", {"id": "pagedata"}).attrs["data-blob"])[
"digital_items"
][0]
download_url = data["downloads"][self.FORMATS[format]]["url"]
id = (data["type"], int(data["item_id"]))
def download(download_url: str) -> str:
with self.get_url(download_url, stream=True) as r:
size = int(r.headers["content-length"])
name = pyrfc6266.requests_response_to_filename(r)
file_name = os.path.join(self.options.dir, name)
with tqdm(total=size, unit="iB", unit_scale=True) as pbar:
with open(file_name, "wb") as f:
for chunk in r.iter_content(chunk_size=self.CHUNK_SIZE):
f.write(chunk)
pbar.update(len(chunk))
return file_name
try:
file_name = download(download_url)
except Exception:
statdownload_url = download_url.replace("/download/", "/statdownload/")
with self.get_url(statdownload_url) as r:
download_url = self.RETRY_URL_REGEX.search(r.text).group("retry_url")
if download_url:
file_name = download(download_url)
else:
# retry requires email address
raise BCFreeDownloadError(
"Download expired. Make sure your payment email is linked "
"to your fan account (Settings > Fan > Payment email addresses)"
)
logger.info(f"Downloaded {file_name}")
return {"id": id, "file_name": file_name}
# unzip the provided file and return all file paths
@staticmethod
def unzip_album(file_name: str) -> List[str]:
dir_name = file_name[:-4]
with zipfile.ZipFile(file_name, "r") as f:
f.extractall(dir_name)
logger.info(f"Unzipped {file_name}.")
os.remove(file_name)
return glob.glob(os.path.join(dir_name, "*"))
# Tag downloaded audio file with url & comment
@staticmethod
def tag_file(file_name: str, head_data: Dict):
try:
f = mutagen.File(file_name)
if f is None:
return
f["website"] = head_data["@id"]
if head_data.get("keywords"):
f["genre"] = head_data["keywords"]
comment = ""
comment += head_data.get("description", "").strip()
comment += "\n\n" + head_data.get("creditText", "")
f["comment"] = comment.strip()
f.save()
except Exception:
# only should happen if the file doesn't support tags
pass
def _download_purchased_album(
self, user_id: int, tralbum_data: Dict
) -> DownloadRet:
logger.info("Downloading album from collection...")
logger.debug(f"Searching for album: '{tralbum_data['current']['title']}'")
data = {
"fan_id": user_id,
"search_key": tralbum_data["current"]["title"],
"search_type": "collection",
}
results = self.post_url_json(
"https://bandcamp.com/api/fancollection/1/search_items", json=data
)
tralbums = results["tralbums"]
redownload_urls = results["redownload_urls"]
wanted_id = f"{tralbum_data['item_type'][0]}:{tralbum_data['id']}"
try:
tralbum = next(
filter(
lambda tralbum: f"{tralbum['tralbum_type']}:{tralbum['tralbum_id']}"
== wanted_id,
tralbums,
)
)
except StopIteration:
raise BCFreeDownloadError("Could not find album in collection")
sale_id = f"{tralbum['sale_item_type']}{tralbum['sale_item_id']}"
if sale_id not in redownload_urls:
raise BCFreeDownloadError("Could not find album download URL in collection")
download_url = redownload_urls[sale_id]
logger.debug(f"Got download URL: {download_url}")
return self._download_file(download_url, self.options.format)
# download from release page
def download_album(self, soup: BeautifulSoup) -> AlbumInfo:
album_data = BCFreeDownloader.get_album_info(soup)
tralbum_data = album_data["tralbum_data"]
head_data = album_data["head_data"]
album_data["is_downloaded"] = False
album_data["email_queued"] = False
url = tralbum_data["url"]
logger.debug(f"tralbum data: {tralbum_data}")
logger.debug(f"album head data: {head_data}")
if not tralbum_data["hasAudio"]:
logger.error(f"{url} has no audio.")
return album_data
head_id = head_data.get("@id")
# fallback if a track link was provided
# track releases have this inAlbum key even if they're standalone
album_release = head_data.get("inAlbum", head_data)["albumRelease"]
# find the albumRelease object that matches the overall album @id link
# this will ensure that strictly the page release is downloaded
album_release = next(obj for obj in album_release if obj["@id"] == head_id)
if "offers" not in album_release:
logger.error(f"{url} has no available offers.")
if tralbum_data["freeDownloadPage"]:
logger.info(f"{url} does not require email")
dlret = self._download_file(
tralbum_data["freeDownloadPage"], self.options.format
)
elif album_release["offers"]["price"] == 0.0:
logger.info(f"{url} requires email")
if self.mail_session is None:
self._init_email()
email_post_url = urljoin(url, "/email_download")
r = self.post_url_json(
email_post_url,
data={
"encoding_name": "none",
"item_id": tralbum_data["current"]["id"],
"item_type": tralbum_data["current"]["type"],
"address": self.options.email,
"country": self.options.country,
"postcode": self.options.zipcode,
},
)
if not r["ok"]:
raise ValueError(f"Bad response when sending email address: {r}")
type = tralbum_data["current"]["type"]
id = tralbum_data["current"]["id"]
album_data["email_queued"] = True
self.queued_emails[(type, id)] = album_data
return album_data
elif tralbum_data["is_purchased"]:
collection_info = soup.find(
"script", {"data-tralbum-collect-info": True}
).attrs["data-tralbum-collect-info"]
collection_info = json.loads(collection_info)
dlret = self._download_purchased_album(
collection_info["fan_id"], tralbum_data
)
else:
logger.error(
f"{url} is not free. If you have purchased this album, "
"use the --cookies flag or --identity flag to pass your login cookie."
)
return album_data
album_data["is_downloaded"] = True
album_data["file_name"] = dlret["file_name"]
return album_data
# unconditionally download from release page
def download_label(self, soup: BeautifulSoup) -> LabelInfo:
info = BCFreeDownloader.get_label_info(soup)
for release in info["releases"]:
logger.info(f"Downloading {release['url']}")
soup = self.get_url_soup(release["url"])
try:
ret = self.download_album(soup)
release["release_info"] = ret
except BCFreeDownloadError as ex:
logger.info(ex)
return info
# unconditionally downloads the provided url
# returns either the result of download_album or download_label
# with the `page_type` set to album|song|band
# exception if download error
def download_url(self, url: str, force: bool = False):
soup = self.get_url_soup(url)
page_info = self.get_page_info(soup)
page_type = page_info.get("type")
if page_type == "album" or page_type == "song":
ret = self.download_album(soup, force)
else:
ret = self.download_label(soup, force)
ret["page_type"] = page_type
return ret
def flush_email_downloads(self) -> List[AlbumInfo]:
checked_ids = set()
downloaded = []
while len(self.queued_emails) > 0:
logger.info(
f"Waiting for {len(self.queued_emails)} emails from Bandcamp..."
)
time.sleep(5)
for email in self.mail_session.get_email_list():
email_id = email.guid
if email_id in checked_ids:
continue
checked_ids.add(email_id)
if (
email.sender == "noreply@bandcamp.com"
and "download" in email.subject
):
logger.info(f'Received email "{email.subject}"')
content = self.mail_session.get_email(email_id).body
match = self.LINK_REGEX.search(content)
if match:
download_url = match.group("url")
dlret = self._download_file(download_url, self.options.format)
self.queued_emails[dlret["id"]]["file_name"] = dlret[
"file_name"
]
downloaded.append(self.queued_emails[dlret["id"]])
self.queued_emails.pop(dlret["id"])
else:
logger.error(f"Could not find download URL in body: {content}")
return downloaded
# get_url_x can't be staticmethods because of special session context
def get_url(self, url: str, **kwargs) -> requests.Response:
r = self.session.get(url, **kwargs)
r.raise_for_status()
return r
def get_url_soup(self, url: str, **kwargs) -> BeautifulSoup:
return BeautifulSoup(self.get_url(url, **kwargs).text, "html.parser")
def get_url_info(self, url: str, **kwargs) -> PageInfo:
soup = self.get_url_soup(url, **kwargs)
try:
return self.get_page_info(soup)
except Exception:
raise BCFreeDownloadError(f"Could not get page info for {url}")
def post_url(self, url: str, **kwargs) -> requests.Response:
r = self.session.post(url, **kwargs)
r.raise_for_status()
return r
def post_url_json(self, url: str, **kwargs) -> Dict:
return self.post_url(url, **kwargs).json()
# get the album/label info of a bandcamp page
@staticmethod
def get_page_info(soup: BeautifulSoup) -> PageInfo:
page_type = soup.head.find("meta", attrs={"property": "og:type"}).get("content")
if page_type == "album" or page_type == "song":
return {"type": page_type, "info": BCFreeDownloader.get_album_info(soup)}
if page_type == "band":
return {"type": page_type, "info": BCFreeDownloader.get_label_info(soup)}
else:
# only bandcamp pages are supported
raise BCFreeDownloadError("Page does not have a valid og:type value")
@staticmethod
def get_label_info(soup: BeautifulSoup) -> LabelInfo:
label_info = soup.find("script", attrs={"data-band": True})
if label_info is None:
raise BCFreeDownloadError("Page has no data-band script.")
label_info = json.loads(label_info["data-band"])
releases = []
# needed for releases
local_url = label_info["local_url"]
# bandcamp splits the release between this music-grid html and some json blob
grid = soup.find("ol", id="music-grid")
for li in grid.find_all("li"):
if "display:none" in li.get("style", ""):
continue
data = li["data-item-id"].split("-")
# most important fields
releases.append(
{
"type": data[0],
"id": int(data[1]),
"url": li.a["href"],
"band_id": int(li["data-band-id"]),
}
)
for obj in json.loads(html.unescape(grid.get("data-client-items", {}))):
if obj.get("filtered"):
continue
# normalize to fit the other half
obj["url"] = obj.pop("page_url")
releases.append(obj)
# fixup local urls into global ones
for release in releases:
if release["url"][0] == "/":
release["url"] = urljoin(local_url, release["url"])
return {"label_info": label_info, "releases": releases}
@staticmethod
def get_album_info(soup: BeautifulSoup) -> AlbumInfo:
tralbum_data = soup.find("script", {"data-tralbum": True}).attrs["data-tralbum"]
tralbum_data = json.loads(tralbum_data)
head_data = soup.head.find(
"script", {"type": "application/ld+json"}, recursive=False
).string
head_data = json.loads(head_data)
return {"tralbum_data": tralbum_data, "head_data": head_data}
+1 -1
View File
@@ -1,6 +1,6 @@
[tool.poetry] [tool.poetry]
name = "free-bandcamp-downloader" name = "free-bandcamp-downloader"
version = "0.4.4" version = "0.5.0"
description = "Download free and name-your-price albums from Bandcamp in lossless quality" description = "Download free and name-your-price albums from Bandcamp in lossless quality"
authors = ["7x11x13 <x7x11x13@gmail.com>"] authors = ["7x11x13 <x7x11x13@gmail.com>"]
license = "MIT" license = "MIT"