#!/usr/bin/env python3 """Fetch all works from api.asmr-200.com and save them as a dataset.""" from __future__ import annotations import argparse import json import math import sys import time from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path from typing import Any from urllib.error import HTTPError, URLError from urllib.parse import urlencode from urllib.request import Request, urlopen API_URL = "https://api.asmr-200.com/api/works" def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="Fetch all ASMR works with paginated concurrent requests." ) parser.add_argument("--output", default="asmr_works.jsonl", help="Output file path.") parser.add_argument( "--format", choices=("jsonl", "json"), default="jsonl", help="Output format. JSONL is better for large datasets.", ) parser.add_argument("--order", default="create_date", help="API order parameter.") parser.add_argument("--sort", default="desc", help="API sort parameter.") parser.add_argument("--subtitle", default="0", help="API subtitle parameter.") parser.add_argument( "--page-size", type=int, default=100, help="API pageSize parameter.", ) parser.add_argument( "--concurrency", type=int, default=10, help="Number of pages to fetch in parallel.", ) parser.add_argument("--timeout", type=float, default=30.0, help="Request timeout seconds.") parser.add_argument("--retries", type=int, default=3, help="Retries per failed page.") parser.add_argument( "--max-pages", type=int, default=None, help="Optional cap for testing. Omit this to fetch every page.", ) return parser.parse_args() def build_url(args: argparse.Namespace, page: int) -> str: query = urlencode( { "order": args.order, "sort": args.sort, "page": page, "pageSize": args.page_size, "subtitle": args.subtitle, } ) return f"{API_URL}?{query}" def fetch_page(args: argparse.Namespace, page: int) -> dict[str, Any]: url = build_url(args, page) last_error: Exception | None = None for attempt in range(1, args.retries + 2): try: request = Request(url, headers={"User-Agent": "dlsite-vector-dataset/1.0"}) with urlopen(request, timeout=args.timeout) as response: if response.status != 200: raise RuntimeError(f"HTTP {response.status}") return json.load(response) except (HTTPError, URLError, TimeoutError, json.JSONDecodeError, RuntimeError) as exc: last_error = exc if attempt > args.retries: break time.sleep(min(2**attempt, 10)) raise RuntimeError(f"failed to fetch page {page}: {last_error}") def iter_pages(args: argparse.Namespace, total_pages: int): pages = range(2, total_pages + 1) with ThreadPoolExecutor(max_workers=args.concurrency) as executor: futures = {executor.submit(fetch_page, args, page): page for page in pages} completed = 1 for future in as_completed(futures): page = futures[future] data = future.result() completed += 1 print( f"Fetched page {page}/{total_pages} " f"({completed}/{total_pages}, {len(data.get('works', []))} works)", file=sys.stderr, ) yield page, data def write_jsonl(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int: count = 0 with output_path.open("w", encoding="utf-8") as file: for page in sorted(page_data): for work in page_data[page]: file.write(json.dumps(work, ensure_ascii=False, separators=(",", ":"))) file.write("\n") count += 1 return count def write_json(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int: works: list[dict[str, Any]] = [] for page in sorted(page_data): works.extend(page_data[page]) with output_path.open("w", encoding="utf-8") as file: json.dump(works, file, ensure_ascii=False, indent=2) file.write("\n") return len(works) def main() -> int: args = parse_args() if args.page_size <= 0: raise SystemExit("--page-size must be greater than 0") if args.concurrency <= 0: raise SystemExit("--concurrency must be greater than 0") output_path = Path(args.output) output_path.parent.mkdir(parents=True, exist_ok=True) first_page = fetch_page(args, 1) pagination = first_page.get("pagination", {}) total_count = int(pagination.get("totalCount", 0)) page_size = int(pagination.get("pageSize", args.page_size)) total_pages = max(1, math.ceil(total_count / page_size)) if args.max_pages is not None: if args.max_pages <= 0: raise SystemExit("--max-pages must be greater than 0") total_pages = min(total_pages, args.max_pages) print( f"Total works: {total_count}, page size: {page_size}, pages: {total_pages}", file=sys.stderr, ) page_data: dict[int, list[dict[str, Any]]] = {1: first_page.get("works", [])} for page, data in iter_pages(args, total_pages): page_data[page] = data.get("works", []) if args.format == "jsonl": written = write_jsonl(output_path, page_data) else: written = write_json(output_path, page_data) print(f"Wrote {written} works to {output_path}", file=sys.stderr) return 0 if __name__ == "__main__": raise SystemExit(main())