Files
BookDeDRM/book-download.py
2026-07-31 17:07:55 -07:00

261 lines
8.5 KiB
Python

import argparse
import random
import re
import subprocess
from pathlib import Path
from playwright.sync_api import sync_playwright
DOWNLOAD_DIR = Path(__file__).parent / "downloads"
PROFILE_DIR = Path(__file__).parent / "chrome-profile"
SCRIPT_DIR = Path(__file__).parent
MIN_DELAY = 2.0
MAX_DELAY = 5.0
def normalize_title(title: str) -> str:
"""Normalize a title for comparison by keeping only lowercase alphanumeric."""
title = re.sub(r"-(epub|pdf)$", "", title, flags=re.IGNORECASE)
return re.sub(r"[^a-z0-9]", "", title.lower())
def title_exists(title: str, existing_books: set[str]) -> bool:
"""Check if title matches any existing book (handles truncated filenames)."""
normalized = normalize_title(title)
for existing in existing_books:
# Check if either starts with the other (handles truncation)
if normalized.startswith(existing) or existing.startswith(normalized):
return True
return False
def auto_scroll(page):
page.evaluate("""
async () => {
await new Promise((resolve) => {
let totalHeight = 0;
const distance = 500;
const timer = setInterval(() => {
window.scrollBy(0, distance);
totalHeight += distance;
if (totalHeight >= document.body.scrollHeight) {
clearInterval(timer);
resolve();
}
}, 300);
setTimeout(() => {
clearInterval(timer);
resolve();
}, 60000);
});
}
""")
page.wait_for_timeout(2000)
def get_existing_books():
"""Get set of normalized book names already downloaded."""
if not DOWNLOAD_DIR.exists():
return set()
return {normalize_title(f.stem) for f in DOWNLOAD_DIR.iterdir() if f.is_file()}
def run_knock(acsm_path: Path):
"""Run knock on an ACSM file via WSL."""
print(f" Running knock on {acsm_path.name}...")
try:
# Convert to WSL relative path from script directory
relative_acsm = acsm_path.relative_to(SCRIPT_DIR)
wsl_relative_path = str(relative_acsm).replace("\\", "/")
result = subprocess.run(
["wsl", "--", "./knock", wsl_relative_path],
capture_output=True,
text=True,
timeout=120,
cwd=SCRIPT_DIR,
)
if result.returncode == 0:
print(f" ✓ knock completed successfully")
else:
print(f" ✗ knock failed: {result.stderr}")
except subprocess.TimeoutExpired:
print(" ✗ knock timed out")
except Exception as e:
print(f" ✗ knock error: {e}")
def run_knock_all():
"""Run knock on all .acsm files in downloads folder."""
if not DOWNLOAD_DIR.exists():
print("Downloads folder doesn't exist")
return
acsm_files = list(DOWNLOAD_DIR.glob("*.acsm"))
if not acsm_files:
print("No .acsm files found in downloads folder")
return
print(f"Found {len(acsm_files)} .acsm files")
for i, acsm_path in enumerate(acsm_files, 1):
print(f"\n[{i}/{len(acsm_files)}] {acsm_path.name}")
run_knock(acsm_path)
print("\nDone!")
def get_book_title(page, menu_button) -> str | None:
"""Extract book title from the DOM near the menu button."""
return page.evaluate("""
(btn) => {
let parent = btn;
for (let i = 0; i < 10; i++) {
if (!parent.parentElement) break;
parent = parent.parentElement;
const title = parent.querySelector('[class*="title"]')
|| parent.querySelector('h3')
|| parent.querySelector('h4')
|| parent.querySelector('[role="heading"]');
if (title && title.textContent.trim()) {
return title.textContent.trim();
}
}
return null;
}
""", menu_button)
def download_books():
"""Main download logic."""
DOWNLOAD_DIR.mkdir(exist_ok=True)
existing_books = get_existing_books()
print(f"Found {len(existing_books)} existing books in downloads folder")
with sync_playwright() as p:
browser = p.chromium.launch_persistent_context(
user_data_dir=str(PROFILE_DIR),
channel="chrome",
headless=False,
accept_downloads=True,
viewport={"width": 1280, "height": 800},
args=["--disable-blink-features=AutomationControlled"],
ignore_default_args=["--enable-automation"],
)
page = browser.new_page()
page.goto("https://play.google.com/books")
input("Press Enter once your library is visible...")
print("Scrolling to load all books...")
auto_scroll(page)
more_options = page.query_selector_all('button[aria-label="More Options"]')
total_books = len(more_options)
print(f"Found {total_books} books in library")
downloaded = 0
skipped = 0
for i in range(total_books):
try:
more_options = page.query_selector_all('button[aria-label="More Options"]')
if i >= len(more_options):
print(f"\n[{i + 1}/{total_books}] ✗ Could not find menu button")
continue
menu_button = more_options[i]
book_title = get_book_title(page, menu_button)
if book_title:
print(f"\n[{i + 1}/{total_books}] {book_title}")
if title_exists(book_title, existing_books):
print(f" ⏭ Skipping (already exists)")
skipped += 1
continue
else:
print(f"\n[{i + 1}/{total_books}] (Unknown title)")
menu_button.click()
page.wait_for_timeout(1000)
export_button = page.get_by_text("Export", exact=True)
if export_button:
export_button.click()
page.wait_for_timeout(1000)
export_options = [
("Export ACSM for EPUB", True),
("Export as EPUB", False),
("Export as PDF", False),
]
clicked = False
for option_text, is_acsm in export_options:
option_button = page.get_by_text(option_text, exact=True)
if option_button and option_button.is_visible():
with page.expect_download(timeout=30000) as download_info:
option_button.click()
download = download_info.value
filename = download.suggested_filename
save_path = DOWNLOAD_DIR / filename
download.save_as(save_path)
print(f" ✓ Downloaded: {filename}")
existing_books.add(normalize_title(Path(filename).stem))
downloaded += 1
if is_acsm:
run_knock(save_path)
clicked = True
break
if not clicked:
print(" ✗ No export option available")
page.keyboard.press("Escape")
else:
print(" ✗ No Export option in menu")
page.keyboard.press("Escape")
except Exception as e:
print(f" ✗ Error: {e}")
page.keyboard.press("Escape")
page.wait_for_timeout(500)
page.keyboard.press("Escape")
delay = random.uniform(MIN_DELAY, MAX_DELAY)
page.wait_for_timeout(int(delay * 1000))
print(f"\n{'=' * 40}")
print(f"Done!")
print(f" Downloaded: {downloaded}")
print(f" Skipped (already existed): {skipped}")
print(f" Files saved to: {DOWNLOAD_DIR}")
browser.close()
def main():
parser = argparse.ArgumentParser(description="Google Play Books downloader")
parser.add_argument(
"--knock",
action="store_true",
help="Run knock on all .acsm files in downloads folder",
)
args = parser.parse_args()
if args.knock:
run_knock_all()
else:
download_books()
if __name__ == "__main__":
main()