#!/usr/bin/env python3 """Download public Rule34Vault bookmarks for a username. Usage: python r34vault_downloader.py krosh --start-page 1 --end-page 3 --out downloads The downloader uses public pages/API only; no credentials are required. """ from __future__ import annotations import argparse import json import mimetypes import re import sys import time from dataclasses import dataclass from pathlib import Path from typing import Any, Iterable, Iterator from urllib.parse import urljoin, urlparse import requests from bs4 import BeautifulSoup from requests import Response, Session from tqdm import tqdm SITE = "https://rule34vault.com" CDN = "https://r34xyz.b-cdn.net" RETRY_STATUSES = {429, 500, 502, 503, 504} IMAGE_EXTS = ("jpg", "png", "webp", "gif", "jpeg") VIDEO_EXTS = ("mp4", "webm", "mov", "m4v") DERIVATIVE_SUFFIXES = ("small", "preview", "thumbnail", "thumb", "medium", "large") HEADERS = { "User-Agent": ( "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 " "(KHTML, like Gecko) Chrome/126.0 Safari/537.36" ), "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8", "Accept-Language": "en-US,en;q=0.9", "Connection": "keep-alive", "Referer": f"{SITE}/", } API_HEADERS = { **HEADERS, "Accept": "application/json,text/plain,*/*", "Origin": SITE, "Content-Type": "application/json", } @dataclass(frozen=True) class MediaSources: images: list[str] image_fallbacks: list[str] videos: list[str] def make_session() -> Session: session = requests.Session() session.headers.update(HEADERS) return session def request_with_retries( session: Session, method: str, url: str, *, tries: int = 4, base_retry_delay: float = 1.5, retry_statuses: set[int] = RETRY_STATUSES, **kwargs: Any, ) -> Response: """Run an HTTP request with exponential backoff for transient failures.""" last_exc: Exception | None = None for attempt in range(tries): try: response = session.request(method, url, **kwargs) if response.status_code not in retry_statuses: return response response.close() last_exc = RuntimeError(f"HTTP {response.status_code} for {url}") except requests.RequestException as exc: last_exc = exc if attempt < tries - 1: time.sleep(base_retry_delay * (2**attempt)) if last_exc: raise last_exc raise RuntimeError(f"request failed without exception: {method} {url}") def fetch_user_page(session: Session, username: str, page: int, **retry_kwargs: Any) -> str: url = f"{SITE}/u/{username}?page={page}" response = request_with_retries(session, "GET", url, headers={**HEADERS, "Referer": f"{SITE}/"}, timeout=30, **retry_kwargs) response.raise_for_status() return response.text def parse_ng_state(html: str) -> dict[str, Any]: soup = BeautifulSoup(html, "html.parser") script = soup.find("script", id="ng-state", type="application/json") or soup.find("script", id="ng-state") if not script: raise ValueError("script#ng-state was not found") text = script.string or script.get_text() or "" if not text.strip(): raise ValueError("script#ng-state is empty") return json.loads(text) def extract_user_id(state: dict[str, Any], username: str) -> int: key = f"get:/api/v2/account/user/{username}" value = state.get(key) if isinstance(value, dict): for field in ("id", "userId"): if isinstance(value.get(field), int): return value[field] data = value.get("data") if isinstance(data, dict): for field in ("userId", "id"): if isinstance(data.get(field), int): return data[field] # Fallback for case variations or encoded usernames. suffix = f"/api/v2/account/user/{username}".lower() for candidate_key, candidate_value in state.items(): if candidate_key.lower().endswith(suffix) and isinstance(candidate_value, dict): data = candidate_value.get("data") if isinstance(candidate_value.get("data"), dict) else candidate_value for field in ("userId", "id"): if isinstance(data.get(field), int): return data[field] raise ValueError(f"could not find public user id for {username!r} in ng-state") def extract_bookmark_items_from_state(state: dict[str, Any], user_id: int) -> list[dict[str, Any]]: exact = f"post:/api/v2/post/search/bookmarked/{user_id}" value = state.get(exact) if isinstance(value, dict) and isinstance(value.get("items"), list): return list(value["items"]) prefix = "post:/api/v2/post/search/bookmarked/" for key, candidate in state.items(): if key.startswith(prefix) and isinstance(candidate, dict) and isinstance(candidate.get("items"), list): return list(candidate["items"]) return [] def fetch_bookmark_items_api( session: Session, user_id: int, *, page: int | None = None, skip: int | None = None, take: int = 20, **retry_kwargs: Any, ) -> list[dict[str, Any]]: body: dict[str, int] = {"take": take} if page is not None: body["page"] = page if skip is not None: body["skip"] = skip url = f"{SITE}/api/v2/post/search/bookmarked/{user_id}" response = request_with_retries(session, "POST", url, headers=API_HEADERS, json=body, timeout=30, **retry_kwargs) response.raise_for_status() payload = response.json() items = payload.get("items", []) if not isinstance(items, list): raise ValueError(f"unexpected API response shape from {url}") return items def post_prefix(post_id: str) -> str: """Return Rule34Vault media path prefix: 1263322 -> 1263, 738794 -> 738.""" return post_id[:-3] if len(post_id) > 3 else post_id def candidate_image_urls(post_id: str) -> Iterator[str]: prefix = post_prefix(post_id) for ext in IMAGE_EXTS: yield f"{SITE}/posts/{prefix}/{post_id}/{post_id}.{ext}" yield f"{CDN}/posts/{prefix}/{post_id}/{post_id}.{ext}" def candidate_video_urls(post_id: str) -> Iterator[str]: prefix = post_prefix(post_id) # Observed videos use 480/720 variants; the path prefix is post_id without the last 3 digits. names = [ f"{post_id}.480.mp4", f"{post_id}.720.mp4", f"{post_id}.720.hevc.mp4", f"{post_id}.mp4", f"{post_id}.webm", ] for name in names: yield f"{CDN}/posts/{prefix}/{post_id}/{name}" yield f"{SITE}/posts/{prefix}/{post_id}/{name}" def original_from_derivative_url(url: str, post_id: str) -> str: """Remove display-size suffix from Rule34Vault media filename if present.""" parsed = urlparse(url) path = parsed.path suffix_re = "|".join(map(re.escape, DERIVATIVE_SUFFIXES)) new_path = re.sub(rf"/{re.escape(post_id)}\.({suffix_re})(\.[A-Za-z0-9]+)$", rf"/{post_id}\2", path) if new_path == path: return url if parsed.scheme and parsed.netloc: return parsed._replace(path=new_path).geturl() return new_path def _dedupe(items: Iterable[str]) -> list[str]: seen: set[str] = set() result: list[str] = [] for item in items: if item and item not in seen: seen.add(item) result.append(item) return result def extract_media_sources_from_html(html: str, post_id: str, page_url: str) -> MediaSources: soup = BeautifulSoup(html, "html.parser") images: list[str] = [] image_fallbacks: list[str] = [] videos: list[str] = [] for video in soup.find_all("video"): src = video.get("src") if src: videos.append(urljoin(page_url, src)) for source in video.find_all("source"): src = source.get("src") if src: videos.append(urljoin(page_url, src)) for source in soup.find_all("source"): src = source.get("src") media_type = source.get("type", "") if src and (media_type.startswith("video/") or Path(urlparse(src).path).suffix.lower().lstrip(".") in VIDEO_EXTS): videos.append(urljoin(page_url, src)) for img in soup.find_all("img"): src = img.get("src") if not src: continue abs_url = urljoin(page_url, src) path = urlparse(abs_url).path if f"/{post_id}/" not in path or not path.lower().endswith(tuple(f".{ext}" for ext in IMAGE_EXTS)): continue original = urljoin(page_url, original_from_derivative_url(abs_url, post_id)) if original != abs_url: images.append(original) image_fallbacks.append(abs_url) else: images.append(abs_url) for meta_name in ("og:image", "twitter:image"): meta = soup.find("meta", attrs={"property": meta_name}) or soup.find("meta", attrs={"name": meta_name}) content = meta.get("content") if meta else None if content: abs_url = urljoin(page_url, content) original = urljoin(page_url, original_from_derivative_url(abs_url, post_id)) if original != abs_url: images.append(original) image_fallbacks.append(abs_url) else: images.append(abs_url) return MediaSources(images=_dedupe(images), image_fallbacks=_dedupe(image_fallbacks), videos=_dedupe(videos)) def fetch_post_html(session: Session, post_id: str, **retry_kwargs: Any) -> str: url = f"{SITE}/post/{post_id}" response = request_with_retries(session, "GET", url, headers={**HEADERS, "Referer": SITE + "/"}, timeout=30, **retry_kwargs) response.raise_for_status() return response.text def choose_video_source(sources: list[str]) -> str | None: if not sources: return None def score(url: str) -> tuple[int, int, int]: lower = url.lower() hevc_penalty = 1 if "hevc" in lower or "hvc1" in lower else 0 ext_rank = 0 if lower.endswith(".mp4") else 1 # Prefer compatible/non-HEVC; for compatibility, 480 mp4 before heavier variants. resolution_rank = 0 if ".480." in lower else 1 if ".720." in lower else 2 return hevc_penalty, ext_rank, resolution_rank return sorted(sources, key=score)[0] def response_extension(url: str, response: Response) -> str: content_type = response.headers.get("content-type", "").split(";", 1)[0].strip().lower() guessed = mimetypes.guess_extension(content_type) if content_type else None if guessed == ".jpe": guessed = ".jpg" if guessed: return guessed suffix = Path(urlparse(url).path).suffix return suffix or ".bin" def is_media_response(response: Response, media: str) -> bool: if response.status_code != 200: return False content_type = response.headers.get("content-type", "").lower() if media == "image": return content_type.startswith("image/") if media == "video": return content_type.startswith("video/") or "mp4" in content_type or "webm" in content_type return content_type.startswith(("image/", "video/")) def probe_media_url( session: Session, url: str, *, media: str, post_id: str, tries: int, base_retry_delay: float, ) -> Response | None: headers = {**HEADERS, "Accept": "*/*", "Referer": f"{SITE}/post/{post_id}"} try: response = request_with_retries( session, "GET", url, headers=headers, stream=True, timeout=45, tries=tries, base_retry_delay=base_retry_delay, ) except Exception: return None if is_media_response(response, media): return response response.close() return None def destination_for(out_dir: Path, post_id: str, url: str, response: Response | None = None) -> Path: ext = response_extension(url, response) if response is not None else Path(urlparse(url).path).suffix or ".bin" return out_dir / post_prefix(post_id) / f"{post_id}{ext}" def stream_download_response(response: Response, dest: Path, chunk_size: int = 1024 * 256) -> int: dest.parent.mkdir(parents=True, exist_ok=True) tmp = dest.with_suffix(dest.suffix + ".part") total = 0 with tmp.open("wb") as fh: for chunk in response.iter_content(chunk_size=chunk_size): if not chunk: continue fh.write(chunk) total += len(chunk) tmp.replace(dest) return total def already_downloaded(post_id: str, out_dir: Path) -> Path | None: prefix_dir = out_dir / post_prefix(post_id) if not prefix_dir.exists(): return None matches = sorted(path for path in prefix_dir.glob(f"{post_id}.*") if path.suffix != ".part") return matches[0] if matches else None def append_line(path: Path, line: str) -> None: path.parent.mkdir(parents=True, exist_ok=True) with path.open("a", encoding="utf-8") as fh: fh.write(line.rstrip("\n") + "\n") def append_download_order(manifest: Path, index: int, post_id: str, path: Path, source_url: str) -> None: append_line(manifest, f"{index:06d} {post_id} {path.as_posix()} {source_url}") def read_seen(path: Path) -> set[str]: if not path.exists(): return set() return {line.strip().split()[0] for line in path.read_text(encoding="utf-8").splitlines() if line.strip()} def resolve_media( session: Session, item: dict[str, Any], *, tries: int, base_retry_delay: float, ) -> tuple[str, Response] | None: post_id = str(item["id"]) post_type = int(item.get("type", 0)) if post_type == 0: for url in candidate_image_urls(post_id): response = probe_media_url(session, url, media="image", post_id=post_id, tries=tries, base_retry_delay=base_retry_delay) if response: return url, response html = fetch_post_html(session, post_id, tries=tries, base_retry_delay=base_retry_delay) sources = extract_media_sources_from_html(html, post_id, f"{SITE}/post/{post_id}") if post_type == 1: ordered_video_sources = [] chosen = choose_video_source(sources.videos) if chosen: ordered_video_sources.append(chosen) ordered_video_sources.extend(url for url in sources.videos if url != chosen) ordered_video_sources.extend(candidate_video_urls(post_id)) for url in _dedupe(ordered_video_sources): response = probe_media_url(session, url, media="video", post_id=post_id, tries=tries, base_retry_delay=base_retry_delay) if response: return url, response else: for url in _dedupe([*sources.images, *sources.image_fallbacks]): response = probe_media_url(session, url, media="image", post_id=post_id, tries=tries, base_retry_delay=base_retry_delay) if response: return url, response return None def bookmark_items_for_pages( session: Session, username: str, *, start_page: int, end_page: int | None, take: int, tries: int, base_retry_delay: float, ) -> Iterator[dict[str, Any]]: """Yield bookmark items from public user pages, preserving page/DOM order. The direct POST API exists, but observed `page` semantics did not exactly match `/u/?page=N`, so the downloader treats the public page as the source of truth and parses its `script#ng-state` payload for each page. """ del take # kept as CLI/API option for future direct-API use; page HTML is authoritative now. user_id: int | None = None page = start_page while True: if end_page is not None and page > end_page: break html = fetch_user_page(session, username, page, tries=tries, base_retry_delay=base_retry_delay) state = parse_ng_state(html) if user_id is None: user_id = extract_user_id(state, username) items = extract_bookmark_items_from_state(state, user_id) if not items: break for item in items: yield item page += 1 def parse_args(argv: list[str] | None = None) -> argparse.Namespace: parser = argparse.ArgumentParser(description="Download public Rule34Vault bookmarks for a username") parser.add_argument("username", help="Rule34Vault username, e.g. krosh") parser.add_argument("--start-page", type=int, default=1, help="first bookmark page to fetch") parser.add_argument("--end-page", type=int, help="last bookmark page to fetch; omitted means until empty") parser.add_argument("--out", type=Path, default=Path("downloads"), help="output directory") parser.add_argument("--tries", type=int, default=4, help="request attempts for transient failures") parser.add_argument("--delay", type=float, default=0.5, help="sleep between posts") parser.add_argument("--base-retry-delay", type=float, default=1.5, help="initial exponential retry sleep") parser.add_argument("--dry-run", action="store_true", help="list bookmark ids that would be processed; do not download files") parser.add_argument("--limit", type=int, help="maximum number of bookmark items to process after pagination") return parser.parse_args(argv) def main(argv: list[str] | None = None) -> int: args = parse_args(argv) out_dir: Path = args.out out_dir.mkdir(parents=True, exist_ok=True) manifest = out_dir / "download_order.txt" seen_path = out_dir / "seen.txt" failed_path = out_dir / "failed.txt" skipped_path = out_dir / "skipped.txt" seen = read_seen(seen_path) session = make_session() success_index = sum(1 for line in manifest.read_text(encoding="utf-8").splitlines()) if manifest.exists() else 0 items = bookmark_items_for_pages( session, args.username, start_page=args.start_page, end_page=args.end_page, take=20, tries=args.tries, base_retry_delay=args.base_retry_delay, ) processed_count = 0 for item in tqdm(items, desc="bookmarks", unit="post"): if args.limit is not None and processed_count >= args.limit: break processed_count += 1 post_id = str(item.get("id")) if not post_id or post_id == "None": append_line(failed_path, f"unknown invalid bookmark item {item!r}") continue existing = already_downloaded(post_id, out_dir) if post_id in seen or existing: append_line(skipped_path, f"{post_id} already downloaded {existing or ''}".rstrip()) continue try: if args.dry_run: append_line(skipped_path, f"{post_id} dry-run") continue resolved = resolve_media( session, item, tries=args.tries, base_retry_delay=args.base_retry_delay, ) if not resolved: append_line(failed_path, f"{post_id} no media found") continue source_url, response = resolved try: dest = destination_for(out_dir, post_id, source_url, response) size = stream_download_response(response, dest) finally: response.close() success_index += 1 append_download_order(manifest, success_index, post_id, dest, source_url) append_line(seen_path, f"{post_id} {dest.as_posix()} {size}") except Exception as exc: # keep batch running and log plain text failure append_line(failed_path, f"{post_id} {type(exc).__name__}: {exc}") time.sleep(args.delay) return 0 if __name__ == "__main__": raise SystemExit(main())