#!/usr/bin/env python3 """Build static vector-search data with an OpenAI-compatible embeddings API.""" from __future__ import annotations import argparse import http.client import json import math import os import struct import time from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait from datetime import datetime, timezone from pathlib import Path from typing import Any from urllib.error import HTTPError, URLError from urllib.request import Request, urlopen from build_vector_data import embedding_text, iter_jsonl, normalize_work DEFAULT_BASE_URL = "http://192.168.0.35:1234/" DEFAULT_MODEL = "text-embedding-qwen3-embedding-8b" def parse_args() -> argparse.Namespace: parser = argparse.ArgumentParser( description="Generate web/data assets using a remote OpenAI-compatible embeddings API." ) parser.add_argument("--input", default="asmr_works.jsonl", help="Input JSONL file.") parser.add_argument("--output-dir", default="web/data", help="Output data directory.") parser.add_argument("--base-url", default=DEFAULT_BASE_URL, help="API base URL.") parser.add_argument("--model", default=DEFAULT_MODEL, help="Embedding model name.") parser.add_argument( "--api-key-env", default="EMBEDDING_API_KEY", help="Environment variable that contains the API key.", ) parser.add_argument("--batch-size", type=int, default=32, help="Texts per API request.") parser.add_argument("--concurrency", type=int, default=10, help="Concurrent API requests.") parser.add_argument("--timeout", type=float, default=120.0, help="Request timeout seconds.") parser.add_argument("--retries", type=int, default=3, help="Retries per failed batch.") parser.add_argument("--retry-wait", type=float, default=2.0, help="Initial retry wait seconds.") parser.add_argument("--max-retry-wait", type=float, default=60.0, help="Maximum retry wait seconds.") parser.add_argument( "--retry-forever", dest="retry_forever", action="store_true", default=True, help="Retry failed batches until they succeed. This is enabled by default.", ) parser.add_argument( "--no-retry-forever", dest="retry_forever", action="store_false", help="Stop after --retries attempts instead of retrying forever.", ) parser.add_argument("--limit", type=int, default=None, help="Optional cap for development.") parser.add_argument( "--no-normalize", action="store_true", help="Do not L2-normalize vectors before writing embeddings.f32.", ) return parser.parse_args() def embeddings_url(base_url: str) -> str: return f"{base_url.rstrip('/')}/v1/embeddings" def normalize_vector(vector: list[float]) -> list[float]: norm = math.sqrt(sum(value * value for value in vector)) if norm == 0: return vector return [value / norm for value in vector] def request_embeddings( *, url: str, model: str, inputs: list[str], api_key: str, timeout: float, ) -> list[list[float]]: body = json.dumps({"model": model, "input": inputs}).encode("utf-8") headers = {"Content-Type": "application/json"} if api_key: headers["Authorization"] = f"Bearer {api_key}" request = Request(url, data=body, headers=headers, method="POST") try: with urlopen(request, timeout=timeout) as response: payload = json.load(response) except HTTPError as exc: detail = exc.read(1000).decode("utf-8", errors="replace") raise RuntimeError(f"HTTP {exc.code}: {detail}") from exc except ( URLError, TimeoutError, json.JSONDecodeError, http.client.IncompleteRead, http.client.RemoteDisconnected, ConnectionResetError, OSError, ) as exc: raise RuntimeError(str(exc)) from exc data = payload.get("data") if isinstance(payload, dict) else None if not isinstance(data, list): raise RuntimeError("response does not contain data[]") ordered: list[list[float] | None] = [None] * len(inputs) for fallback_index, item in enumerate(data): if not isinstance(item, dict): raise RuntimeError("response data item is not an object") index = item.get("index", fallback_index) embedding = item.get("embedding") if not isinstance(index, int) or not 0 <= index < len(inputs): raise RuntimeError(f"invalid embedding index: {index}") if not isinstance(embedding, list) or not embedding: raise RuntimeError(f"missing embedding for index {index}") try: ordered[index] = [float(value) for value in embedding] except (TypeError, ValueError) as exc: raise RuntimeError(f"embedding for index {index} contains non-numeric values") from exc missing = [index for index, vector in enumerate(ordered) if vector is None] if missing: raise RuntimeError(f"missing embeddings for indexes: {missing[:10]}") return [vector for vector in ordered if vector is not None] def request_embeddings_with_retries( *, url: str, model: str, inputs: list[str], api_key: str, timeout: float, retries: int, retry_wait: float, max_retry_wait: float, retry_forever: bool, batch_start: int, ) -> list[list[float]]: last_error: Exception | None = None attempt = 0 while True: try: return request_embeddings( url=url, model=model, inputs=inputs, api_key=api_key, timeout=timeout, ) except RuntimeError as exc: last_error = exc if not retry_forever and attempt >= retries: break wait_seconds = min(max_retry_wait, retry_wait * (2 ** min(attempt, 10))) print( f"Batch {batch_start}: request failed ({exc}); retrying in {wait_seconds:.1f}s", flush=True, ) time.sleep(wait_seconds) attempt += 1 raise RuntimeError(str(last_error)) def batched(values: list[str], size: int): for index in range(0, len(values), size): yield index, values[index : index + size] def write_remote_embeddings( *, path: Path, texts: list[str], url: str, model: str, api_key: str, batch_size: int, timeout: float, retries: int, retry_wait: float, max_retry_wait: float, retry_forever: bool, concurrency: int, should_normalize: bool, ) -> int: dimensions: int | None = None temp_path = path.with_suffix(path.suffix + ".tmp") try: with temp_path.open("wb") as file: pending = {} completed: dict[int, list[list[float]]] = {} next_submit = 0 next_write = 0 def submit_available(executor: ThreadPoolExecutor) -> None: nonlocal next_submit while next_submit < len(texts) and len(pending) + len(completed) < concurrency: start = next_submit batch = texts[start : start + batch_size] future = executor.submit( request_embeddings_with_retries, url=url, model=model, inputs=batch, api_key=api_key, timeout=timeout, retries=retries, retry_wait=retry_wait, max_retry_wait=max_retry_wait, retry_forever=retry_forever, batch_start=start, ) pending[future] = start next_submit += len(batch) def write_vectors(vectors: list[list[float]]) -> None: nonlocal dimensions for vector in vectors: if should_normalize: vector = normalize_vector(vector) if dimensions is None: dimensions = len(vector) elif len(vector) != dimensions: raise RuntimeError( f"embedding dimensions changed: expected {dimensions}, got {len(vector)}" ) file.write(struct.pack(f"<{dimensions}f", *vector)) with ThreadPoolExecutor(max_workers=concurrency) as executor: submit_available(executor) while pending: done, _ = wait(pending, return_when=FIRST_COMPLETED) for future in done: start = pending.pop(future) vectors = future.result() expected = min(batch_size, len(texts) - start) if len(vectors) != expected: raise RuntimeError( f"batch at {start} returned {len(vectors)} embeddings; expected {expected}" ) completed[start] = vectors while next_write in completed: vectors = completed.pop(next_write) write_vectors(vectors) next_write += len(vectors) print(f"Embedded {next_write}/{len(texts)} works", flush=True) submit_available(executor) except Exception: temp_path.unlink(missing_ok=True) raise if dimensions is None: raise RuntimeError("no embeddings were written") temp_path.replace(path) return dimensions def main() -> int: args = parse_args() input_path = Path(args.input) output_dir = Path(args.output_dir) if args.batch_size <= 0: raise SystemExit("--batch-size must be greater than 0") if args.concurrency <= 0: raise SystemExit("--concurrency must be greater than 0") if args.timeout <= 0: raise SystemExit("--timeout must be greater than 0") if args.retries < 0: raise SystemExit("--retries must be greater than or equal to 0") if args.retry_wait < 0: raise SystemExit("--retry-wait must be greater than or equal to 0") if args.max_retry_wait < 0: raise SystemExit("--max-retry-wait must be greater than or equal to 0") if not input_path.exists(): raise SystemExit(f"input file not found: {input_path}") output_dir.mkdir(parents=True, exist_ok=True) works: list[dict[str, Any]] = [] texts: list[str] = [] for raw_work in iter_jsonl(input_path, args.limit): work = normalize_work(raw_work, len(works)) works.append(work) texts.append(embedding_text(work)) if not works: raise SystemExit("no works found") api_key = os.environ.get(args.api_key_env, "") url = embeddings_url(args.base_url) embeddings_path = output_dir / "embeddings.f32" try: dimensions = write_remote_embeddings( path=embeddings_path, texts=texts, url=url, model=args.model, api_key=api_key, batch_size=args.batch_size, timeout=args.timeout, retries=args.retries, retry_wait=args.retry_wait, max_retry_wait=args.max_retry_wait, retry_forever=args.retry_forever, concurrency=args.concurrency, should_normalize=not args.no_normalize, ) except RuntimeError as exc: raise SystemExit(f"embedding request failed: {exc}") from exc manifest = { "count": len(works), "dimensions": dimensions, "embeddingFile": "embeddings.f32", "worksFile": "works.json", "method": "remote-openai-compatible", "model": args.model, "baseUrl": args.base_url.rstrip("/"), "generatedAt": datetime.now(timezone.utc).isoformat(), "normalized": not args.no_normalize, "batchSize": args.batch_size, "concurrency": args.concurrency, "retryForever": args.retry_forever, "maxRetryWait": args.max_retry_wait, "score": { "vectorWeight": 0.8, "tagWeight": 0.2, }, } with (output_dir / "works.json").open("w", encoding="utf-8") as file: json.dump(works, file, ensure_ascii=False, separators=(",", ":")) with (output_dir / "manifest.json").open("w", encoding="utf-8") as file: json.dump(manifest, file, ensure_ascii=False, indent=2) file.write("\n") print(f"Wrote {len(works)} works to {output_dir}") print(f"Embedding method: remote-openai-compatible ({dimensions} dimensions)") return 0 if __name__ == "__main__": raise SystemExit(main())