import argparse import logging import os from pathlib import Path from pathvalidate import sanitize_filename import random import requests from selenium import webdriver from selenium.webdriver.common.action_chains import ActionChains from selenium.webdriver.common.by import By from selenium.common.exceptions import NoSuchElementException from selenium.webdriver.chrome.service import Service from selenium.webdriver.chrome.options import Options from selenium.webdriver.support.ui import WebDriverWait, Select from selenium.webdriver.support import expected_conditions as EC import sys import time import traceback __version__ = "2026.4.6.0" # Helper for logging def setup_logging(folder, mode="syslog", logfile=None): logger = logging.getLogger() logger.setLevel(logging.INFO) # remove default handlers for handler in logger.handlers[:]: logger.removeHandler(handler) if mode == "none": logging.disable(logging.CRITICAL) return formatter = logging.Formatter( "%(asctime)s - %(levelname)s - %(message)s" ) if mode in ("syslog", "all"): console = logging.StreamHandler(sys.stdout) console.setFormatter(formatter) logger.addHandler(console) if mode in ("file", "all"): if not logfile: raise ValueError("File logging requires a logfile path") file_handler = logging.FileHandler(os.path.join(folder, logfile)) file_handler.setFormatter(formatter) logger.addHandler(file_handler) # Create the parser and add a description def args(): parser = argparse.ArgumentParser( description="V-dlp options and variables...", epilog="End of help documentation..." ) parser.add_argument("-l", type=str, help="Choose logging option", default="syslog", choices=["syslog","file","all","none"]) parser.add_argument("-lf", help="If log is file/all need to specify file to log to") parser.add_argument("-d", type=str, help="Download directory to use", default=f"{Path.home() / 'Downloads'}") parser.add_argument("-u", type=str, help="File with links inside", default="urls.txt") parser.add_argument("-uh", type=bool, help="Use headless Chrome", default=False) parser.add_argument("-tl", type=float, help="How long to wait for webpage to load before timeout", default=8) parser.add_argument("-r", type=float, help="How often to refresh statistics on screen", default=2) parser.add_argument("-tw", type=float, help="Number of seconds to pause between downloads", default=4) parser.add_argument("-nm", type=bool, help="Do not monitor download statistics", default=False) parser.add_argument("-gc", type=int, help="Download cover image (0: don't download, 1: small (avif), 2: large (webp), 3: both)", default=0, choices=[0, 1, 2, 3]) parser.add_argument("-v", action='store_true', help="Display version information") return parser # URLs to process def open_urls(file): try: with open(file) as f: urls = [line.strip() for line in f] urls = list(dict.fromkeys(urls)) url_length = len(urls) ess = "s" if url_length == 0: logging.warning("There are no URLs to process!") sys.exit("There are no URLs to process!") if url_length == 1: ess = "" logging.info(f"Will process {url_length} URL{ess}...") return urls, url_length except Exception as e: exc_type, exc_obj, exc_tb = sys.exc_info() emessage = f"Error {exc_tb.tb_lineno}: Could not find url file: {e}!" logging.error(emessage) sys.exit(emessage) # Launch Chrome def open_chrome(headless, download_dir): # Generate list of random screen resolutions display_resolutions = ["2560,1440","1920,1080","1600,1200"] # Create and add Chrome options chrome_options = Options() if headless == True: chrome_options.add_argument("--headless=new") chrome_options.add_argument("--disable-gpu") chrome_options.add_argument(f"--window-size=2560,1440") chrome_options.add_argument("--no-sandbox") chrome_options.add_argument("--disable-dev-shm-usage") chrome_options.add_argument("--simulate-outdated-no-au='Tue, 31 Dec 2099 23:59:59 GMT'") chrome_options.add_argument("--disable-background-networking") chrome_options.add_argument("--disable-component-update") prefs = { "download.default_directory": download_dir, "download.prompt_for_download": False, "download.directory_upgrade": True } chrome_options.add_experimental_option("prefs", prefs) driver = webdriver.Chrome(options=chrome_options) # Chrome sometimes blocks downloads in headless mode if headless == True: driver.execute_cdp_cmd( "Page.setDownloadBehavior", { "behavior": "allow", "downloadPath": download_dir } ) return driver # Helper to calculate percentages def percentage_of_total(part, whole): if whole == 0: return 0 # Handle division by zero case return round(((part / whole) * 100), 1) # Check for temp files limiting amount of time spent before failing def wait_for_file(download_dir, check_interval_seconds=0.25, max_wait_seconds=10): start_time = time.time() while True: # Check if the file exists files = os.listdir(download_dir) partial = [f for f in files if f.endswith(".crdownload")] if partial: elapsed_time = time.time() - start_time return True # Calculate elapsed time and check if timeout is reached elapsed_time = time.time() - start_time if elapsed_time >= max_wait_seconds: print(f"Timed out after {max_wait_seconds} seconds. Temp file not found at {download_dir}.") return False # Wait for the specified interval before the next check remaining_time = max_wait_seconds - elapsed_time if remaining_time < check_interval_seconds: time_to_sleep = remaining_time # Time to sleep time.sleep(check_interval_seconds) # Display useful stats about ongoing downloads def monitor_download(folder, fsize, refresh_rate, tstart=time.time()): downloading = True last_size = 0 while downloading: files = os.listdir(folder) partial = [f for f in files if f.endswith(".crdownload")] if partial: file_path = os.path.join(folder, partial[0]) size = os.path.getsize(file_path) tot_perc = percentage_of_total(size, fsize) if size != last_size: ct = time.time() dt = ct - tstart if dt == 0: dt = 1 speed = round(((size - last_size)/1024/1024/refresh_rate)*8, 2) etas = ((fsize - size) / (size / dt)) etam, etams = divmod(etas, 60) if args.l in ("syslog", "all"): print(F"Downloading at {speed} Mbps... {tot_perc}%. ETA: {int(etam)}m {int(etams)}s ", end="\r") last_size = size time.sleep(refresh_rate) else: downloading = False tsec = time.time() - tstart if tsec == 0: tsec = 1 tmin, tmsec = divmod(tsec, 60) avg_speed = round((fsize/tsec/1024/1024)*8, 1) if avg_speed > 175: logging.warning("Download did not start (probably)") return -999, -999 else: logging.info(f"Downloaded {round((fsize/1024/1024), 2)}MB in {int(tmin)}m {int(tmsec)}s ({avg_speed} Mbps).") return last_size, tsec # Get cover art def download_img(driver, download_dir, title, itype, baseurl): try: img_element = WebDriverWait(driver, 5).until( EC.presence_of_element_located((By.XPATH, '//img[@alt="Box"]')) ) if img_element: if itype in [1,3]: img_url = img_element.get_attribute('src') if img_url: img_saved = save_img(download_dir, title, itype, img_url, baseurl, 1) if not img_saved: logging.warning("See above WARNING.") else: logging.warning("No box image url found.") if itype in [2,3]: body_element = driver.find_element(By.ID, "main") img_element.click() try: dialog_element = WebDriverWait(driver, 5).until( EC.presence_of_element_located((By.ID, "imageDialog")) ) if dialog_element: img_element2 = dialog_element.find_element(By.TAG_NAME, "img") img_url2 = img_element2.get_attribute('src') if img_url2: save_img(download_dir, title, itype, img_url2, baseurl, 2) else: logging.warning("No large box image found.") actions = ActionChains(driver) actions.move_to_element_with_offset(body_element, random.randint(1, 100), random.randint(1, 100)).click().perform() time.sleep(1) except Exception as e: exc_type, exc_obj, exc_tb = sys.exc_info() logging.warning(f"Error {exc_tb.tb_lineno}: No large box image found!") except Exception as e: exc_type, exc_obj, exc_tb = sys.exc_info() logging.warning(f"Error {exc_tb.tb_lineno}: No box image exists. Skipping...") # Save cover art def save_img(download_dir, title, itype, iurl, baseurl, iform): headers: dict[str, str] = { 'Accept': 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8', 'Accept-Encoding': 'gzip, deflate, br, zstd', 'Connection': 'keep-alive', 'User-Agent': 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:149.0) Gecko/20100101 Firefox/149.0', 'Referer': f'{iurl}' } response = requests.get(iurl, headers=headers, allow_redirects=True, stream=True) response.raise_for_status() if response.status_code == 200: fext = 'avif' ftitle = sanitize_filename(title) if iform == 2: fext = 'webp' save_path = os.path.join(f"{download_dir}", f"{ftitle}.{fext}") with open(save_path, 'wb') as file: for chunk in response.iter_content(1024): file.write(chunk) return True else: logging.warning(f"Could not download box image: {response.status_code} - {response.reason}") return False # Download logic def download_it(driver, wait, monitor, wait_time, download_dir, refresh_rate, total_size, total_time, cur_url, url_length, dtitle, durl, cover, failed_urls, ddisc=0): # Get file size size = 0 try: size_element = wait.until( EC.presence_of_element_located((By.ID, "dl_size")) ) size_raw = size_element.text except: exc_type, exc_obj, exc_tb = sys.exc_info() logging.warning(f"Error {exc_tb.tb_lineno}: Could not determine download size from webpage") size_raw = '1 GB' pass logging.info(f"{dtitle} {size_raw}") # Download box cover if cover > 0: raw_title = driver.title.replace("The Vault: ", "") download_img(driver, download_dir, raw_title, cover, durl) # Click Download button dl_start = time.time() try: download_form = wait.until( EC.presence_of_element_located((By.ID, "dl_form")) ) download_button = wait.until( EC.element_to_be_clickable((By.XPATH, "//button[text()='Download']")) ) actions = ActionChains(driver) if download_form: actions.move_to_element_with_offset(download_form, random.randint(1, 50), random.randint(1, 11)).perform() time.sleep(1) actions.move_to_element_with_offset(download_button, random.randint(1, 25), random.randint(1, 6)).perform() time.sleep(0.5) #time.sleep(3) download_form = wait.until( EC.presence_of_element_located((By.ID, "dl_form")) ) #logging.info("Clicked form submit") if download_form: download_form.submit() raw_title = driver.title if raw_title == "Vimm's Lair: Error 400": return -998, -998 else: raise ValueError("Download form could not be submitted") else: actions.move_to_element_with_offset(download_button, random.randint(1, 13), random.randint(1, 3)).perform() time.sleep(1) #logging.info("Clicked download button") download_button = wait.until( EC.element_to_be_clickable((By.XPATH, "//button[text()='Download']")) ) if download_button: download_button.click() #driver.execute_script("arguments[0].click();", download_button) raw_title = driver.title if raw_title == "Vimm's Lair: Error 400": return -998, -998 else: raise ValueError("Download button could not be submitted") dl_start = time.time() except Exception as e: exc_type, exc_obj, exc_tb = sys.exc_info() logging.warning(f"Error {exc_tb.tb_lineno}: Download button not found for {dtitle}!") logging.error(f"{e}") failed_urls.append(dtitle) failed_urls.append(durl) return 0, 0 # Convert human readable size to bytes if " KB" in size_raw: size = float(size_raw.replace(" KB", "")) * 1024 elif " MB" in size_raw: size = float(size_raw.replace(" MB", "")) * 1024 ** 2 elif " GB" in size_raw: size = float(size_raw.replace(" GB", "")) * 1024 ** 3 elif " TB" in size_raw: size = float(size_raw.replace(" TB", "")) * 1024 ** 4 else: size = 500 * 1024 ** 2 # Monitor current download argmonitor = monitor argwait = wait_time ctime = 0 csize = 0 # If file size is >32MB, go ahead and monitor anyway if size > 33554431 and monitor: monitor = False # If file size <32MB, force no monitor if size < 33554432: monitor = True wait_time = 15 size_limit = 2097152 size_factor = size // size_limit if size_factor < wait_time: wait_time = size_factor if size_factor == 0: wait_time = 1 # If we are monitoring downloads if monitor == False: if wait_for_file(download_dir, 0.25, 8): csize, ctime = monitor_download(download_dir, size*1.01, refresh_rate, dl_start) if csize == -999 and ctime == -999: failed_urls.append(dtitle) failed_urls.append(durl) elif csize == -998 and ctime == -998: logging.warning("Vimm's Lair Error 400: An unexpected browser error has occurred (we think you might be a bot)") failed_urls.append(dtitle) failed_urls.append(durl) else: total_size += csize total_time += ctime else: # Try to click Continue if it appears try: raw_title = driver.title if raw_title != "Vimm's Lair: Error 400": continue_button = WebDriverWait(driver, 4).until( EC.element_to_be_clickable((By.XPATH, "//input[@value='Continue']")) ) continue_button.click() dl_start = time.time() logging.info(f"DL Click: {dl_start} Check: {round((time.time() - dl_start),1)}") if wait_for_file(files, 0.25, 8): csize, ctime = monitor_download(download_dir, files, size*1.01, refresh_rate, dl_start) total_size += csize total_time += ctime else: logging.warning("Never found temp file for download...") failed_urls.append(dtitle) failed_urls.append(durl) pass else: logging.warning("Vimm's Lair Error 400: An unexpected browser error has occurred (we think you might be a bot)") failed_urls.append(dtitle) failed_urls.append(durl) except: exc_type, exc_obj, exc_tb = sys.exc_info() logging.warning(f"Error {exc_tb.tb_lineno}: Could not click on Continue button...") failed_urls.append(dtitle) failed_urls.append(durl) pass # Wait to call next URL if cur_url < url_length: if size < 33554432: logging.warning(f"File size was too small to monitor! (Less than 32MB). Waiting {round(wait_time, 1)} seconds...") time.sleep(wait_time) else: if size < 33554432: logging.warning(f"File size was too small to monitor! (Less than 32MB).") cur_url += 1 monitor = argmonitor wait_time = argwait return csize, ctime # Main program def main(args): download_dir = args.d url_file = args.u headless = args.uh page_load_time = args.tl refresh_rate = args.r wait_time = args.tw monitor = args.nm cover = args.gc logging.info(f"Download Directory: {download_dir}") logging.info(f"Reading URLs from: {url_file}") urls, url_length = open_urls(url_file) # For each URL in the file total_size = 0 total_time = 0 cur_url = 1 failed_urls = [] for url in urls: # Launch chrome and open URL driver = open_chrome(headless, download_dir) driver.get(url) # Wait to open URL wait = WebDriverWait(driver, page_load_time) # Check for multiple discs try: disc_element = driver.find_element(By.ID, "disc_number") disc_select = Select(disc_element) url_length += len(disc_select.options) - 1 for option in disc_select.options: disc_value = option.get_attribute("value") disc_text = option.get_attribute("text") disc_replace = f"{cur_url}/{url_length}: ({disc_text}) " if len(disc_select.options) == 1: disc_replace = f"{cur_url}/{url_length}: " disc_select.select_by_value(disc_value) title = driver.title.replace("The Vault: ", disc_replace) osize, otime = download_it(driver, wait, monitor, wait_time, download_dir, refresh_rate, total_size, total_time, cur_url, url_length, title, url, cover, failed_urls, disc_value) total_size += osize total_time += otime cur_url += 1 except KeyboardInterrupt: logging.info("\nSignal received. Shutting down gracefully...") driver.close() driver.quit() sys.exit(67) except Exception as e: logging.warning(f"Multiple discs error? {e}") title = driver.title.replace("The Vault: ", f"{cur_url}/{url_length}: ") osize, otime = download_it(driver, wait, monitor, wait_time, download_dir, refresh_rate, total_size, total_time, cur_url, url_length, title, url, cover, failed_urls) total_size += osize total_time += otime cur_url += 1 finally: # Close and end Chrome session driver.close() driver.quit() # If we were monitoring, display total statistics if total_size > 0: if total_time == 0: total_time = 1 ttmin, ttmsec = divmod(total_time, 60) tavg_speed = round((total_size/total_time/1024/1024)*8, 1) logging.info(f"Downloaded {round((total_size/1024/1024), 2)}MB in {int(ttmin)}m {int(ttmsec)}s ({tavg_speed} Mbps).") # If anything failed, write to file if failed_urls: fn = os.path.join(f"{download_dir}", "failed.txt") logging.warning(f"Appending failed downloads to {fn}") fnt = "a" with open(f"{fn}", fnt) as f: for item in failed_urls: f.write(item + '\n') if __name__ == "__main__": parser = args() args = parser.parse_args() if args.v: sys.exit(f"v{__version__}\n") # If directory does not exist, create it os.makedirs(args.d, exist_ok=True) setup_logging(args.d, args.l, args.lf) main(args)