173 lines
5.6 KiB
Python
173 lines
5.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Fetch all works from api.asmr-200.com and save them as a dataset."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import json
|
|
import math
|
|
import sys
|
|
import time
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
from typing import Any
|
|
from urllib.error import HTTPError, URLError
|
|
from urllib.parse import urlencode
|
|
from urllib.request import Request, urlopen
|
|
|
|
|
|
API_URL = "https://api.asmr-200.com/api/works"
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Fetch all ASMR works with paginated concurrent requests."
|
|
)
|
|
parser.add_argument("--output", default="asmr_works.jsonl", help="Output file path.")
|
|
parser.add_argument(
|
|
"--format",
|
|
choices=("jsonl", "json"),
|
|
default="jsonl",
|
|
help="Output format. JSONL is better for large datasets.",
|
|
)
|
|
parser.add_argument("--order", default="create_date", help="API order parameter.")
|
|
parser.add_argument("--sort", default="desc", help="API sort parameter.")
|
|
parser.add_argument("--subtitle", default="0", help="API subtitle parameter.")
|
|
parser.add_argument(
|
|
"--page-size",
|
|
type=int,
|
|
default=100,
|
|
help="API pageSize parameter.",
|
|
)
|
|
parser.add_argument(
|
|
"--concurrency",
|
|
type=int,
|
|
default=10,
|
|
help="Number of pages to fetch in parallel.",
|
|
)
|
|
parser.add_argument("--timeout", type=float, default=30.0, help="Request timeout seconds.")
|
|
parser.add_argument("--retries", type=int, default=3, help="Retries per failed page.")
|
|
parser.add_argument(
|
|
"--max-pages",
|
|
type=int,
|
|
default=None,
|
|
help="Optional cap for testing. Omit this to fetch every page.",
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def build_url(args: argparse.Namespace, page: int) -> str:
|
|
query = urlencode(
|
|
{
|
|
"order": args.order,
|
|
"sort": args.sort,
|
|
"page": page,
|
|
"pageSize": args.page_size,
|
|
"subtitle": args.subtitle,
|
|
}
|
|
)
|
|
return f"{API_URL}?{query}"
|
|
|
|
|
|
def fetch_page(args: argparse.Namespace, page: int) -> dict[str, Any]:
|
|
url = build_url(args, page)
|
|
last_error: Exception | None = None
|
|
|
|
for attempt in range(1, args.retries + 2):
|
|
try:
|
|
request = Request(url, headers={"User-Agent": "dlsite-vector-dataset/1.0"})
|
|
with urlopen(request, timeout=args.timeout) as response:
|
|
if response.status != 200:
|
|
raise RuntimeError(f"HTTP {response.status}")
|
|
return json.load(response)
|
|
except (HTTPError, URLError, TimeoutError, json.JSONDecodeError, RuntimeError) as exc:
|
|
last_error = exc
|
|
if attempt > args.retries:
|
|
break
|
|
time.sleep(min(2**attempt, 10))
|
|
|
|
raise RuntimeError(f"failed to fetch page {page}: {last_error}")
|
|
|
|
|
|
def iter_pages(args: argparse.Namespace, total_pages: int):
|
|
pages = range(2, total_pages + 1)
|
|
with ThreadPoolExecutor(max_workers=args.concurrency) as executor:
|
|
futures = {executor.submit(fetch_page, args, page): page for page in pages}
|
|
completed = 1
|
|
|
|
for future in as_completed(futures):
|
|
page = futures[future]
|
|
data = future.result()
|
|
completed += 1
|
|
print(
|
|
f"Fetched page {page}/{total_pages} "
|
|
f"({completed}/{total_pages}, {len(data.get('works', []))} works)",
|
|
file=sys.stderr,
|
|
)
|
|
yield page, data
|
|
|
|
|
|
def write_jsonl(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int:
|
|
count = 0
|
|
with output_path.open("w", encoding="utf-8") as file:
|
|
for page in sorted(page_data):
|
|
for work in page_data[page]:
|
|
file.write(json.dumps(work, ensure_ascii=False, separators=(",", ":")))
|
|
file.write("\n")
|
|
count += 1
|
|
return count
|
|
|
|
|
|
def write_json(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int:
|
|
works: list[dict[str, Any]] = []
|
|
for page in sorted(page_data):
|
|
works.extend(page_data[page])
|
|
|
|
with output_path.open("w", encoding="utf-8") as file:
|
|
json.dump(works, file, ensure_ascii=False, indent=2)
|
|
file.write("\n")
|
|
|
|
return len(works)
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
if args.page_size <= 0:
|
|
raise SystemExit("--page-size must be greater than 0")
|
|
if args.concurrency <= 0:
|
|
raise SystemExit("--concurrency must be greater than 0")
|
|
|
|
output_path = Path(args.output)
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
|
|
first_page = fetch_page(args, 1)
|
|
pagination = first_page.get("pagination", {})
|
|
total_count = int(pagination.get("totalCount", 0))
|
|
page_size = int(pagination.get("pageSize", args.page_size))
|
|
total_pages = max(1, math.ceil(total_count / page_size))
|
|
if args.max_pages is not None:
|
|
if args.max_pages <= 0:
|
|
raise SystemExit("--max-pages must be greater than 0")
|
|
total_pages = min(total_pages, args.max_pages)
|
|
|
|
print(
|
|
f"Total works: {total_count}, page size: {page_size}, pages: {total_pages}",
|
|
file=sys.stderr,
|
|
)
|
|
|
|
page_data: dict[int, list[dict[str, Any]]] = {1: first_page.get("works", [])}
|
|
for page, data in iter_pages(args, total_pages):
|
|
page_data[page] = data.get("works", [])
|
|
|
|
if args.format == "jsonl":
|
|
written = write_jsonl(output_path, page_data)
|
|
else:
|
|
written = write_json(output_path, page_data)
|
|
|
|
print(f"Wrote {written} works to {output_path}", file=sys.stderr)
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|