Files
2026-06-11 03:43:59 +09:00

173 lines
5.6 KiB
Python

#!/usr/bin/env python3
"""Fetch all works from api.asmr-200.com and save them as a dataset."""
from __future__ import annotations
import argparse
import json
import math
import sys
import time
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from typing import Any
from urllib.error import HTTPError, URLError
from urllib.parse import urlencode
from urllib.request import Request, urlopen
API_URL = "https://api.asmr-200.com/api/works"
def parse_args() -> argparse.Namespace:
parser = argparse.ArgumentParser(
description="Fetch all ASMR works with paginated concurrent requests."
)
parser.add_argument("--output", default="asmr_works.jsonl", help="Output file path.")
parser.add_argument(
"--format",
choices=("jsonl", "json"),
default="jsonl",
help="Output format. JSONL is better for large datasets.",
)
parser.add_argument("--order", default="create_date", help="API order parameter.")
parser.add_argument("--sort", default="desc", help="API sort parameter.")
parser.add_argument("--subtitle", default="0", help="API subtitle parameter.")
parser.add_argument(
"--page-size",
type=int,
default=100,
help="API pageSize parameter.",
)
parser.add_argument(
"--concurrency",
type=int,
default=10,
help="Number of pages to fetch in parallel.",
)
parser.add_argument("--timeout", type=float, default=30.0, help="Request timeout seconds.")
parser.add_argument("--retries", type=int, default=3, help="Retries per failed page.")
parser.add_argument(
"--max-pages",
type=int,
default=None,
help="Optional cap for testing. Omit this to fetch every page.",
)
return parser.parse_args()
def build_url(args: argparse.Namespace, page: int) -> str:
query = urlencode(
{
"order": args.order,
"sort": args.sort,
"page": page,
"pageSize": args.page_size,
"subtitle": args.subtitle,
}
)
return f"{API_URL}?{query}"
def fetch_page(args: argparse.Namespace, page: int) -> dict[str, Any]:
url = build_url(args, page)
last_error: Exception | None = None
for attempt in range(1, args.retries + 2):
try:
request = Request(url, headers={"User-Agent": "dlsite-vector-dataset/1.0"})
with urlopen(request, timeout=args.timeout) as response:
if response.status != 200:
raise RuntimeError(f"HTTP {response.status}")
return json.load(response)
except (HTTPError, URLError, TimeoutError, json.JSONDecodeError, RuntimeError) as exc:
last_error = exc
if attempt > args.retries:
break
time.sleep(min(2**attempt, 10))
raise RuntimeError(f"failed to fetch page {page}: {last_error}")
def iter_pages(args: argparse.Namespace, total_pages: int):
pages = range(2, total_pages + 1)
with ThreadPoolExecutor(max_workers=args.concurrency) as executor:
futures = {executor.submit(fetch_page, args, page): page for page in pages}
completed = 1
for future in as_completed(futures):
page = futures[future]
data = future.result()
completed += 1
print(
f"Fetched page {page}/{total_pages} "
f"({completed}/{total_pages}, {len(data.get('works', []))} works)",
file=sys.stderr,
)
yield page, data
def write_jsonl(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int:
count = 0
with output_path.open("w", encoding="utf-8") as file:
for page in sorted(page_data):
for work in page_data[page]:
file.write(json.dumps(work, ensure_ascii=False, separators=(",", ":")))
file.write("\n")
count += 1
return count
def write_json(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int:
works: list[dict[str, Any]] = []
for page in sorted(page_data):
works.extend(page_data[page])
with output_path.open("w", encoding="utf-8") as file:
json.dump(works, file, ensure_ascii=False, indent=2)
file.write("\n")
return len(works)
def main() -> int:
args = parse_args()
if args.page_size <= 0:
raise SystemExit("--page-size must be greater than 0")
if args.concurrency <= 0:
raise SystemExit("--concurrency must be greater than 0")
output_path = Path(args.output)
output_path.parent.mkdir(parents=True, exist_ok=True)
first_page = fetch_page(args, 1)
pagination = first_page.get("pagination", {})
total_count = int(pagination.get("totalCount", 0))
page_size = int(pagination.get("pageSize", args.page_size))
total_pages = max(1, math.ceil(total_count / page_size))
if args.max_pages is not None:
if args.max_pages <= 0:
raise SystemExit("--max-pages must be greater than 0")
total_pages = min(total_pages, args.max_pages)
print(
f"Total works: {total_count}, page size: {page_size}, pages: {total_pages}",
file=sys.stderr,
)
page_data: dict[int, list[dict[str, Any]]] = {1: first_page.get("works", [])}
for page, data in iter_pages(args, total_pages):
page_data[page] = data.get("works", [])
if args.format == "jsonl":
written = write_jsonl(output_path, page_data)
else:
written = write_json(output_path, page_data)
print(f"Wrote {written} works to {output_path}", file=sys.stderr)
return 0
if __name__ == "__main__":
raise SystemExit(main())