#! /usr/bin/env python

# bunkr_dl - download Bunkr albums and search results
# Copyright (C) 2025-2026 Danilo M. <danix@danix.xyz>
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License version 2 as
# published by the Free Software Foundation.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
# GNU General Public License for more details.
#
# You should have received a copy of the GNU General Public License along
# with this program; if not, write to the Free Software Foundation, Inc.,
# 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.

import requests
from bs4 import BeautifulSoup
import re
import os
import argparse
import time
from urllib.parse import urljoin, urlparse, quote

DEFAULT_OUTPUT = "/data/bunkr"  # Base download folder, override with -o/--output

# Define headers to mimic a browser request
headers = {
    'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/85.0.4183.121 Safari/537.36'
}

def get_soup(url):
    """Helper function to get parsed HTML"""
    response = requests.get(url, headers=headers)
    if response.status_code == 200:
        return BeautifulSoup(response.content, 'lxml')
    else:
        print(f"Failed to fetch {url}, status code: {response.status_code}")
        return None

def is_search_page(url):
    """Check if the provided URL is a search page (contains '?search=')"""
    return '?search=' in url

def extract_search_query(url):
    """Extract and format the search query from the URL"""
    query = re.search(r'\?search=([^&]+)', url)
    if query:
        search_text = query.group(1).replace('%20', ' ')  # Replace encoded spaces with actual spaces
        # Capitalize each word in the search text
        return ' '.join(word.capitalize() for word in search_text.split())
    return "Unknown_Search"

def extract_album_title(soup):
    """Extract album title from <h1> tag if available"""
    title_tag = soup.find('h1') if soup else None
    if title_tag:
        return title_tag.text.strip()
    return "Unnamed_Album"

def find_album_links_from_search(base_url):
    """Function to find all album links from the search results page"""
    page = 1
    album_links = []

    while True:
        # Modify the base_url to add page parameter if necessary
        search_url = f"{base_url}&page={page}" if page > 1 else base_url
        soup = get_soup(search_url)

        if not soup:
            print(f"Failed to fetch page {page}")
            break
        
        # Find album links on the current search page
        found = False
        for link in soup.find_all('a', href=True):
            if re.search(r"/a/\w+", link['href']):
                album_links.append(urljoin(search_url, link['href']))
                found = True
        
        # If no more albums found on this page, stop
        if not found:
            break
        
        page += 1
    
    return album_links

def find_media_links(album_url, soup):
    """Function to find all file page links (/f/<slug>) on the album page"""
    media_links = []
    for link in soup.find_all('a', href=True):
        if re.match(r"(https://[^/]+)?/f/\w+", link['href']):
            url = urljoin(album_url, link['href'])
            if url not in media_links:
                media_links.append(url)
    return media_links

def find_download_link(media_page_url):
    """Function to extract the download page link (https://dl.<domain>/file/<id>)"""
    soup = get_soup(media_page_url)
    if not soup:
        return None

    for link in soup.find_all('a', href=True):
        if re.search(r"https://[^/]+/file/\d+", link['href']):
            return link['href']

    return None

def get_final_media_link(download_page_url):
    """Resolve the download page to a signed media URL and the original file name.

    Mirrors the page's JS: POST the file id to /api/_001_v2 for the storage
    path, then ask the sign service for a token.
    """
    file_id = download_page_url.rstrip('/').split('/')[-1]
    host = urlparse(download_page_url)
    try:
        meta = requests.post(f"{host.scheme}://{host.netloc}/api/_001_v2", json={'id': file_id},
                             headers={**headers, 'Referer': download_page_url}, timeout=30)
        meta.raise_for_status()
        meta = meta.json()
        sign = requests.get("https://glb-apisign.cdn.cr/sign", params={'path': meta['path']},
                            headers=headers, timeout=30)
        sign.raise_for_status()
        sign = sign.json()
    except (requests.exceptions.RequestException, ValueError, KeyError) as e:
        print(f"Failed to resolve {download_page_url}: {e}")
        return None

    url = f"{meta['mediafiles']}{quote(meta['path'])}?token={sign['token']}&ex={sign['ex']}"
    name = (meta.get('original') or meta['path'].split('/')[-1]).replace('/', '_')
    return url, name

def download_file(file_url, file_name, save_dir, retries=5):
    """Function to download the file with retries and resume functionality"""
    local_filename = os.path.join(save_dir, file_name)

    attempt = 0
    while attempt < retries:
        # Recompute each attempt so a retry resumes from what was written
        req_headers = dict(headers)
        if os.path.exists(local_filename):
            req_headers["Range"] = f"bytes={os.path.getsize(local_filename)}-"
        try:
            with requests.get(file_url, headers=req_headers, stream=True, timeout=60) as response:
                if response.status_code == 416:  # Range past end: already complete
                    print(f"Already downloaded: {local_filename}")
                    return local_filename
                response.raise_for_status()
                # Check if it's a resumable download
                if response.status_code == 206:
                    print(f"Resuming download for: {local_filename}")
                    mode = 'ab'
                else:  # Full body: start over instead of appending to a partial file
                    print(f"Downloading: {local_filename}")
                    mode = 'wb'

                with open(local_filename, mode) as f:
                    for chunk in response.iter_content(chunk_size=8192):
                        if chunk:  # Filter out keep-alive chunks
                            f.write(chunk)
            print(f"Download complete: {local_filename}")
            return local_filename
        except (requests.exceptions.ConnectionError, requests.exceptions.ChunkedEncodingError, requests.exceptions.Timeout) as e:
            attempt += 1
            print(f"Error during download: {e}. Retrying {attempt}/{retries}...")
            time.sleep(5)  # Wait before retrying
        except Exception as e:
            print(f"Failed to download {file_url}: {e}")
            break

    print(f"Failed to download {file_url} after {retries} attempts.")
    return None

def ensure_unique_folder(base_folder, folder_name):
    """Ensure the folder name is unique by adding suffixes like ' - 2', ' - 3' if necessary"""
    folder_path = os.path.join(base_folder, folder_name)
    if not os.path.exists(folder_path):
        return folder_path

    # If folder exists, add a numerical suffix
    suffix = 2
    while os.path.exists(f"{folder_path} - {suffix}"):
        suffix += 1

    return f"{folder_path} - {suffix}"

def process_album(album_url, base_save_directory):
    """Process a single album to download all media"""
    soup = get_soup(album_url)
    if not soup:
        return
    album_title = extract_album_title(soup)

    # Create a unique subfolder for the album title inside the base directory
    album_save_directory = ensure_unique_folder(base_save_directory, album_title)
    os.makedirs(album_save_directory, exist_ok=True)

    # Step 1: Get all media links (images/videos) from the album page
    media_tile_links = find_media_links(album_url, soup)

    # Large albums are split into ?page=N (100 files each); out-of-range pages
    # repeat the last one, so stop when a page adds nothing new
    page = 2
    while soup.find('a', href=f"?page={page}"):
        soup = get_soup(f"{album_url.split('?')[0]}?page={page}")
        new = [l for l in find_media_links(album_url, soup) if l not in media_tile_links] if soup else []
        if not new:
            break
        media_tile_links += new
        page += 1
    
    if not media_tile_links:
        print(f"No media links found in album {album_url}")
        return

    # Step 2: Follow each media tile link and extract the download page URL
    for media_link in media_tile_links:
        download_link = find_download_link(media_link)
        if download_link:
            # Step 3: Get the final media URL from the download page
            final_media = get_final_media_link(download_link)
            if final_media:
                # Step 4: Download the file
                download_file(*final_media, album_save_directory)
            else:
                print(f"Failed to get final media from {download_link}")
        else:
            print(f"Failed to find download link on {media_link}")

def main():
    parser = argparse.ArgumentParser(
        description="Download every file of a Bunkr album, or of every album found by a\n"
                    "Bunkr search. Paginated albums are followed, files keep their original\n"
                    "names, and partially downloaded files are resumed when the same folder\n"
                    "is used again.",
        epilog="Album files go to OUTPUT/<album title>/ (a ' - 2', ' - 3' suffix is added\n"
               "if that folder exists). Search results go to\n"
               "OUTPUT/<Search Terms>/<album title>/.\n\n"
               "Quote search URLs so the shell leaves '?' and '&' alone, e.g.:\n"
               "  %(prog)s https://bunkr.cr/a/XXXXXXXX\n"
               "  %(prog)s -o ~/Downloads 'https://bunkr.cr/?search=some%%20name'",
        formatter_class=argparse.RawDescriptionHelpFormatter)
    parser.add_argument('url', help="album URL (https://<domain>/a/<id>) or search URL (containing ?search=)")
    parser.add_argument('-o', '--output', default=DEFAULT_OUTPUT, metavar='DIR',
                        help=f"base download folder (default: {DEFAULT_OUTPUT})")
    args = parser.parse_args()

    url = args.url
    base_save_directory = os.path.expanduser(args.output)

    # Determine the folder name based on search query
    if is_search_page(url):
        search_query = extract_search_query(url)
        save_directory = os.path.join(base_save_directory, search_query)
    else:
        # If it's a direct album link, name the folder based on the album title
        save_directory = base_save_directory

    # Create the directory if it doesn't exist
    os.makedirs(save_directory, exist_ok=True)

    # Check if the URL is a search result or a direct album link
    if is_search_page(url):
        print(f"Processing search results from {url}")
        # Get all album links from the search page
        album_links = find_album_links_from_search(url)
        
        if not album_links:
            print("No albums found in the search results.")
            return
        
        # Process each album found in the search results
        for album_link in album_links:
            print(f"Processing album {album_link}")
            process_album(album_link, save_directory)

    else:
        # Direct album link, process it
        print(f"Processing album {url}")
        process_album(url, save_directory)

if __name__ == "__main__":
    main()
