358 lines
13 KiB
Python
358 lines
13 KiB
Python
#!/usr/bin/env python3
|
|
"""Build static vector-search data with an OpenAI-compatible embeddings API."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import http.client
|
|
import json
|
|
import math
|
|
import os
|
|
import struct
|
|
import time
|
|
from concurrent.futures import FIRST_COMPLETED, ThreadPoolExecutor, wait
|
|
from datetime import datetime, timezone
|
|
from pathlib import Path
|
|
from typing import Any
|
|
from urllib.error import HTTPError, URLError
|
|
from urllib.request import Request, urlopen
|
|
|
|
from build_vector_data import embedding_text, iter_jsonl, normalize_work
|
|
|
|
|
|
DEFAULT_BASE_URL = "http://192.168.0.35:1234/"
|
|
DEFAULT_MODEL = "text-embedding-qwen3-embedding-8b"
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
parser = argparse.ArgumentParser(
|
|
description="Generate web/data assets using a remote OpenAI-compatible embeddings API."
|
|
)
|
|
parser.add_argument("--input", default="asmr_works.jsonl", help="Input JSONL file.")
|
|
parser.add_argument("--output-dir", default="web/data", help="Output data directory.")
|
|
parser.add_argument("--base-url", default=DEFAULT_BASE_URL, help="API base URL.")
|
|
parser.add_argument("--model", default=DEFAULT_MODEL, help="Embedding model name.")
|
|
parser.add_argument(
|
|
"--api-key-env",
|
|
default="EMBEDDING_API_KEY",
|
|
help="Environment variable that contains the API key.",
|
|
)
|
|
parser.add_argument("--batch-size", type=int, default=32, help="Texts per API request.")
|
|
parser.add_argument("--concurrency", type=int, default=10, help="Concurrent API requests.")
|
|
parser.add_argument("--timeout", type=float, default=120.0, help="Request timeout seconds.")
|
|
parser.add_argument("--retries", type=int, default=3, help="Retries per failed batch.")
|
|
parser.add_argument("--retry-wait", type=float, default=2.0, help="Initial retry wait seconds.")
|
|
parser.add_argument("--max-retry-wait", type=float, default=60.0, help="Maximum retry wait seconds.")
|
|
parser.add_argument(
|
|
"--retry-forever",
|
|
dest="retry_forever",
|
|
action="store_true",
|
|
default=True,
|
|
help="Retry failed batches until they succeed. This is enabled by default.",
|
|
)
|
|
parser.add_argument(
|
|
"--no-retry-forever",
|
|
dest="retry_forever",
|
|
action="store_false",
|
|
help="Stop after --retries attempts instead of retrying forever.",
|
|
)
|
|
parser.add_argument("--limit", type=int, default=None, help="Optional cap for development.")
|
|
parser.add_argument(
|
|
"--no-normalize",
|
|
action="store_true",
|
|
help="Do not L2-normalize vectors before writing embeddings.f32.",
|
|
)
|
|
return parser.parse_args()
|
|
|
|
|
|
def embeddings_url(base_url: str) -> str:
|
|
return f"{base_url.rstrip('/')}/v1/embeddings"
|
|
|
|
|
|
def normalize_vector(vector: list[float]) -> list[float]:
|
|
norm = math.sqrt(sum(value * value for value in vector))
|
|
if norm == 0:
|
|
return vector
|
|
return [value / norm for value in vector]
|
|
|
|
|
|
def request_embeddings(
|
|
*,
|
|
url: str,
|
|
model: str,
|
|
inputs: list[str],
|
|
api_key: str,
|
|
timeout: float,
|
|
) -> list[list[float]]:
|
|
body = json.dumps({"model": model, "input": inputs}).encode("utf-8")
|
|
headers = {"Content-Type": "application/json"}
|
|
if api_key:
|
|
headers["Authorization"] = f"Bearer {api_key}"
|
|
|
|
request = Request(url, data=body, headers=headers, method="POST")
|
|
try:
|
|
with urlopen(request, timeout=timeout) as response:
|
|
payload = json.load(response)
|
|
except HTTPError as exc:
|
|
detail = exc.read(1000).decode("utf-8", errors="replace")
|
|
raise RuntimeError(f"HTTP {exc.code}: {detail}") from exc
|
|
except (
|
|
URLError,
|
|
TimeoutError,
|
|
json.JSONDecodeError,
|
|
http.client.IncompleteRead,
|
|
http.client.RemoteDisconnected,
|
|
ConnectionResetError,
|
|
OSError,
|
|
) as exc:
|
|
raise RuntimeError(str(exc)) from exc
|
|
|
|
data = payload.get("data") if isinstance(payload, dict) else None
|
|
if not isinstance(data, list):
|
|
raise RuntimeError("response does not contain data[]")
|
|
|
|
ordered: list[list[float] | None] = [None] * len(inputs)
|
|
for fallback_index, item in enumerate(data):
|
|
if not isinstance(item, dict):
|
|
raise RuntimeError("response data item is not an object")
|
|
index = item.get("index", fallback_index)
|
|
embedding = item.get("embedding")
|
|
if not isinstance(index, int) or not 0 <= index < len(inputs):
|
|
raise RuntimeError(f"invalid embedding index: {index}")
|
|
if not isinstance(embedding, list) or not embedding:
|
|
raise RuntimeError(f"missing embedding for index {index}")
|
|
try:
|
|
ordered[index] = [float(value) for value in embedding]
|
|
except (TypeError, ValueError) as exc:
|
|
raise RuntimeError(f"embedding for index {index} contains non-numeric values") from exc
|
|
|
|
missing = [index for index, vector in enumerate(ordered) if vector is None]
|
|
if missing:
|
|
raise RuntimeError(f"missing embeddings for indexes: {missing[:10]}")
|
|
|
|
return [vector for vector in ordered if vector is not None]
|
|
|
|
|
|
def request_embeddings_with_retries(
|
|
*,
|
|
url: str,
|
|
model: str,
|
|
inputs: list[str],
|
|
api_key: str,
|
|
timeout: float,
|
|
retries: int,
|
|
retry_wait: float,
|
|
max_retry_wait: float,
|
|
retry_forever: bool,
|
|
batch_start: int,
|
|
) -> list[list[float]]:
|
|
last_error: Exception | None = None
|
|
attempt = 0
|
|
while True:
|
|
try:
|
|
return request_embeddings(
|
|
url=url,
|
|
model=model,
|
|
inputs=inputs,
|
|
api_key=api_key,
|
|
timeout=timeout,
|
|
)
|
|
except RuntimeError as exc:
|
|
last_error = exc
|
|
if not retry_forever and attempt >= retries:
|
|
break
|
|
wait_seconds = min(max_retry_wait, retry_wait * (2 ** min(attempt, 10)))
|
|
print(
|
|
f"Batch {batch_start}: request failed ({exc}); retrying in {wait_seconds:.1f}s",
|
|
flush=True,
|
|
)
|
|
time.sleep(wait_seconds)
|
|
attempt += 1
|
|
raise RuntimeError(str(last_error))
|
|
|
|
|
|
def batched(values: list[str], size: int):
|
|
for index in range(0, len(values), size):
|
|
yield index, values[index : index + size]
|
|
|
|
|
|
def write_remote_embeddings(
|
|
*,
|
|
path: Path,
|
|
texts: list[str],
|
|
url: str,
|
|
model: str,
|
|
api_key: str,
|
|
batch_size: int,
|
|
timeout: float,
|
|
retries: int,
|
|
retry_wait: float,
|
|
max_retry_wait: float,
|
|
retry_forever: bool,
|
|
concurrency: int,
|
|
should_normalize: bool,
|
|
) -> int:
|
|
dimensions: int | None = None
|
|
temp_path = path.with_suffix(path.suffix + ".tmp")
|
|
|
|
try:
|
|
with temp_path.open("wb") as file:
|
|
pending = {}
|
|
completed: dict[int, list[list[float]]] = {}
|
|
next_submit = 0
|
|
next_write = 0
|
|
|
|
def submit_available(executor: ThreadPoolExecutor) -> None:
|
|
nonlocal next_submit
|
|
while next_submit < len(texts) and len(pending) + len(completed) < concurrency:
|
|
start = next_submit
|
|
batch = texts[start : start + batch_size]
|
|
future = executor.submit(
|
|
request_embeddings_with_retries,
|
|
url=url,
|
|
model=model,
|
|
inputs=batch,
|
|
api_key=api_key,
|
|
timeout=timeout,
|
|
retries=retries,
|
|
retry_wait=retry_wait,
|
|
max_retry_wait=max_retry_wait,
|
|
retry_forever=retry_forever,
|
|
batch_start=start,
|
|
)
|
|
pending[future] = start
|
|
next_submit += len(batch)
|
|
|
|
def write_vectors(vectors: list[list[float]]) -> None:
|
|
nonlocal dimensions
|
|
for vector in vectors:
|
|
if should_normalize:
|
|
vector = normalize_vector(vector)
|
|
if dimensions is None:
|
|
dimensions = len(vector)
|
|
elif len(vector) != dimensions:
|
|
raise RuntimeError(
|
|
f"embedding dimensions changed: expected {dimensions}, got {len(vector)}"
|
|
)
|
|
file.write(struct.pack(f"<{dimensions}f", *vector))
|
|
|
|
with ThreadPoolExecutor(max_workers=concurrency) as executor:
|
|
submit_available(executor)
|
|
while pending:
|
|
done, _ = wait(pending, return_when=FIRST_COMPLETED)
|
|
for future in done:
|
|
start = pending.pop(future)
|
|
vectors = future.result()
|
|
expected = min(batch_size, len(texts) - start)
|
|
if len(vectors) != expected:
|
|
raise RuntimeError(
|
|
f"batch at {start} returned {len(vectors)} embeddings; expected {expected}"
|
|
)
|
|
completed[start] = vectors
|
|
|
|
while next_write in completed:
|
|
vectors = completed.pop(next_write)
|
|
write_vectors(vectors)
|
|
next_write += len(vectors)
|
|
print(f"Embedded {next_write}/{len(texts)} works", flush=True)
|
|
|
|
submit_available(executor)
|
|
except Exception:
|
|
temp_path.unlink(missing_ok=True)
|
|
raise
|
|
|
|
if dimensions is None:
|
|
raise RuntimeError("no embeddings were written")
|
|
|
|
temp_path.replace(path)
|
|
return dimensions
|
|
|
|
|
|
def main() -> int:
|
|
args = parse_args()
|
|
input_path = Path(args.input)
|
|
output_dir = Path(args.output_dir)
|
|
|
|
if args.batch_size <= 0:
|
|
raise SystemExit("--batch-size must be greater than 0")
|
|
if args.concurrency <= 0:
|
|
raise SystemExit("--concurrency must be greater than 0")
|
|
if args.timeout <= 0:
|
|
raise SystemExit("--timeout must be greater than 0")
|
|
if args.retries < 0:
|
|
raise SystemExit("--retries must be greater than or equal to 0")
|
|
if args.retry_wait < 0:
|
|
raise SystemExit("--retry-wait must be greater than or equal to 0")
|
|
if args.max_retry_wait < 0:
|
|
raise SystemExit("--max-retry-wait must be greater than or equal to 0")
|
|
if not input_path.exists():
|
|
raise SystemExit(f"input file not found: {input_path}")
|
|
|
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
|
|
works: list[dict[str, Any]] = []
|
|
texts: list[str] = []
|
|
for raw_work in iter_jsonl(input_path, args.limit):
|
|
work = normalize_work(raw_work, len(works))
|
|
works.append(work)
|
|
texts.append(embedding_text(work))
|
|
|
|
if not works:
|
|
raise SystemExit("no works found")
|
|
|
|
api_key = os.environ.get(args.api_key_env, "")
|
|
url = embeddings_url(args.base_url)
|
|
embeddings_path = output_dir / "embeddings.f32"
|
|
try:
|
|
dimensions = write_remote_embeddings(
|
|
path=embeddings_path,
|
|
texts=texts,
|
|
url=url,
|
|
model=args.model,
|
|
api_key=api_key,
|
|
batch_size=args.batch_size,
|
|
timeout=args.timeout,
|
|
retries=args.retries,
|
|
retry_wait=args.retry_wait,
|
|
max_retry_wait=args.max_retry_wait,
|
|
retry_forever=args.retry_forever,
|
|
concurrency=args.concurrency,
|
|
should_normalize=not args.no_normalize,
|
|
)
|
|
except RuntimeError as exc:
|
|
raise SystemExit(f"embedding request failed: {exc}") from exc
|
|
|
|
manifest = {
|
|
"count": len(works),
|
|
"dimensions": dimensions,
|
|
"embeddingFile": "embeddings.f32",
|
|
"worksFile": "works.json",
|
|
"method": "remote-openai-compatible",
|
|
"model": args.model,
|
|
"baseUrl": args.base_url.rstrip("/"),
|
|
"generatedAt": datetime.now(timezone.utc).isoformat(),
|
|
"normalized": not args.no_normalize,
|
|
"batchSize": args.batch_size,
|
|
"concurrency": args.concurrency,
|
|
"retryForever": args.retry_forever,
|
|
"maxRetryWait": args.max_retry_wait,
|
|
"score": {
|
|
"vectorWeight": 0.8,
|
|
"tagWeight": 0.2,
|
|
},
|
|
}
|
|
|
|
with (output_dir / "works.json").open("w", encoding="utf-8") as file:
|
|
json.dump(works, file, ensure_ascii=False, separators=(",", ":"))
|
|
with (output_dir / "manifest.json").open("w", encoding="utf-8") as file:
|
|
json.dump(manifest, file, ensure_ascii=False, indent=2)
|
|
file.write("\n")
|
|
|
|
print(f"Wrote {len(works)} works to {output_dir}")
|
|
print(f"Embedding method: remote-openai-compatible ({dimensions} dimensions)")
|
|
return 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|