init
This commit is contained in:
@@ -0,0 +1,172 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Fetch all works from api.asmr-200.com and save them as a dataset."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import math
|
||||
import sys
|
||||
import time
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
from urllib.error import HTTPError, URLError
|
||||
from urllib.parse import urlencode
|
||||
from urllib.request import Request, urlopen
|
||||
|
||||
|
||||
API_URL = "https://api.asmr-200.com/api/works"
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Fetch all ASMR works with paginated concurrent requests."
|
||||
)
|
||||
parser.add_argument("--output", default="asmr_works.jsonl", help="Output file path.")
|
||||
parser.add_argument(
|
||||
"--format",
|
||||
choices=("jsonl", "json"),
|
||||
default="jsonl",
|
||||
help="Output format. JSONL is better for large datasets.",
|
||||
)
|
||||
parser.add_argument("--order", default="create_date", help="API order parameter.")
|
||||
parser.add_argument("--sort", default="desc", help="API sort parameter.")
|
||||
parser.add_argument("--subtitle", default="0", help="API subtitle parameter.")
|
||||
parser.add_argument(
|
||||
"--page-size",
|
||||
type=int,
|
||||
default=100,
|
||||
help="API pageSize parameter.",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--concurrency",
|
||||
type=int,
|
||||
default=10,
|
||||
help="Number of pages to fetch in parallel.",
|
||||
)
|
||||
parser.add_argument("--timeout", type=float, default=30.0, help="Request timeout seconds.")
|
||||
parser.add_argument("--retries", type=int, default=3, help="Retries per failed page.")
|
||||
parser.add_argument(
|
||||
"--max-pages",
|
||||
type=int,
|
||||
default=None,
|
||||
help="Optional cap for testing. Omit this to fetch every page.",
|
||||
)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def build_url(args: argparse.Namespace, page: int) -> str:
|
||||
query = urlencode(
|
||||
{
|
||||
"order": args.order,
|
||||
"sort": args.sort,
|
||||
"page": page,
|
||||
"pageSize": args.page_size,
|
||||
"subtitle": args.subtitle,
|
||||
}
|
||||
)
|
||||
return f"{API_URL}?{query}"
|
||||
|
||||
|
||||
def fetch_page(args: argparse.Namespace, page: int) -> dict[str, Any]:
|
||||
url = build_url(args, page)
|
||||
last_error: Exception | None = None
|
||||
|
||||
for attempt in range(1, args.retries + 2):
|
||||
try:
|
||||
request = Request(url, headers={"User-Agent": "dlsite-vector-dataset/1.0"})
|
||||
with urlopen(request, timeout=args.timeout) as response:
|
||||
if response.status != 200:
|
||||
raise RuntimeError(f"HTTP {response.status}")
|
||||
return json.load(response)
|
||||
except (HTTPError, URLError, TimeoutError, json.JSONDecodeError, RuntimeError) as exc:
|
||||
last_error = exc
|
||||
if attempt > args.retries:
|
||||
break
|
||||
time.sleep(min(2**attempt, 10))
|
||||
|
||||
raise RuntimeError(f"failed to fetch page {page}: {last_error}")
|
||||
|
||||
|
||||
def iter_pages(args: argparse.Namespace, total_pages: int):
|
||||
pages = range(2, total_pages + 1)
|
||||
with ThreadPoolExecutor(max_workers=args.concurrency) as executor:
|
||||
futures = {executor.submit(fetch_page, args, page): page for page in pages}
|
||||
completed = 1
|
||||
|
||||
for future in as_completed(futures):
|
||||
page = futures[future]
|
||||
data = future.result()
|
||||
completed += 1
|
||||
print(
|
||||
f"Fetched page {page}/{total_pages} "
|
||||
f"({completed}/{total_pages}, {len(data.get('works', []))} works)",
|
||||
file=sys.stderr,
|
||||
)
|
||||
yield page, data
|
||||
|
||||
|
||||
def write_jsonl(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int:
|
||||
count = 0
|
||||
with output_path.open("w", encoding="utf-8") as file:
|
||||
for page in sorted(page_data):
|
||||
for work in page_data[page]:
|
||||
file.write(json.dumps(work, ensure_ascii=False, separators=(",", ":")))
|
||||
file.write("\n")
|
||||
count += 1
|
||||
return count
|
||||
|
||||
|
||||
def write_json(output_path: Path, page_data: dict[int, list[dict[str, Any]]]) -> int:
|
||||
works: list[dict[str, Any]] = []
|
||||
for page in sorted(page_data):
|
||||
works.extend(page_data[page])
|
||||
|
||||
with output_path.open("w", encoding="utf-8") as file:
|
||||
json.dump(works, file, ensure_ascii=False, indent=2)
|
||||
file.write("\n")
|
||||
|
||||
return len(works)
|
||||
|
||||
|
||||
def main() -> int:
|
||||
args = parse_args()
|
||||
if args.page_size <= 0:
|
||||
raise SystemExit("--page-size must be greater than 0")
|
||||
if args.concurrency <= 0:
|
||||
raise SystemExit("--concurrency must be greater than 0")
|
||||
|
||||
output_path = Path(args.output)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
first_page = fetch_page(args, 1)
|
||||
pagination = first_page.get("pagination", {})
|
||||
total_count = int(pagination.get("totalCount", 0))
|
||||
page_size = int(pagination.get("pageSize", args.page_size))
|
||||
total_pages = max(1, math.ceil(total_count / page_size))
|
||||
if args.max_pages is not None:
|
||||
if args.max_pages <= 0:
|
||||
raise SystemExit("--max-pages must be greater than 0")
|
||||
total_pages = min(total_pages, args.max_pages)
|
||||
|
||||
print(
|
||||
f"Total works: {total_count}, page size: {page_size}, pages: {total_pages}",
|
||||
file=sys.stderr,
|
||||
)
|
||||
|
||||
page_data: dict[int, list[dict[str, Any]]] = {1: first_page.get("works", [])}
|
||||
for page, data in iter_pages(args, total_pages):
|
||||
page_data[page] = data.get("works", [])
|
||||
|
||||
if args.format == "jsonl":
|
||||
written = write_jsonl(output_path, page_data)
|
||||
else:
|
||||
written = write_json(output_path, page_data)
|
||||
|
||||
print(f"Wrote {written} works to {output_path}", file=sys.stderr)
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user