daily data update and add ci
Update ASMR data / update (push) Has been cancelled

This commit is contained in:
tokuzou0829
2026-06-17 03:06:11 +09:00
parent eb63500b63
commit c4487a0ff4
4 changed files with 224 additions and 120 deletions
+62
View File
@@ -0,0 +1,62 @@
name: Update ASMR data
on:
schedule:
# 00:30 JST. Scheduled actions may run a little later depending on runner load.
- cron: "30 15 * * *"
push:
branches:
- main
workflow_dispatch:
jobs:
update:
runs-on: ubuntu-latest
steps:
- name: Checkout repository
uses: actions/checkout@v4
with:
ref: main
fetch-depth: 0
- name: Set up Python
uses: actions/setup-python@v5
with:
python-version: "3.11"
- name: Update ASMR JSONL data
run: |
if [ "${GITHUB_EVENT_NAME:-}" = "push" ]; then
last_commit_message="$(git log -1 --pretty=%B)"
case "$last_commit_message" in
*"[skip ci]"*)
echo "Skipping data update for CI-generated commit."
touch .skip-asmr-data-update
exit 0
;;
esac
python scripts/update_daily_asmr_data.py --skip-web-data --force
else
python scripts/update_daily_asmr_data.py --skip-web-data
fi
- name: Commit and push changes
run: |
if [ -f .skip-asmr-data-update ]; then
echo "Data update skipped."
exit 0
fi
if git diff --quiet -- asmr_works.jsonl asmr_works.filtered.jsonl; then
echo "No ASMR data changes to commit."
exit 0
fi
git config user.name "gitea-actions[bot]"
git config user.email "gitea-actions[bot]@users.noreply.local"
git add asmr_works.jsonl asmr_works.filtered.jsonl
git commit -m "daily data update [skip ci]"
git pull --rebase --autostash origin main
git push origin HEAD:main
File diff suppressed because one or more lines are too long
+84 -70
View File
File diff suppressed because one or more lines are too long
+20 -5
View File
@@ -45,6 +45,11 @@ def parse_args() -> argparse.Namespace:
help="Filtered ASMR JSONL output file.",
)
parser.add_argument("--data-dir", default="web/data", help="Directory containing web data assets.")
parser.add_argument(
"--skip-web-data",
action="store_true",
help="Update raw and filtered JSONL files without reading or writing web/data assets.",
)
parser.add_argument("--order", default="create_date", help="ASMR API order parameter.")
parser.add_argument("--sort", default="desc", help="ASMR API sort parameter.")
parser.add_argument("--subtitle", default="0", help="ASMR API subtitle parameter.")
@@ -712,21 +717,27 @@ def main() -> int:
existing_keys = {key for work in existing_works for key in identity_keys(work)}
new_works, fetched_existing_by_key, scanned_pages = fetch_latest_works(args, existing_keys)
if not new_works and not args.force and not web_data_out_of_sync(filtered_path, Path(args.data_dir)):
needs_web_data_update = False if args.skip_web_data else web_data_out_of_sync(filtered_path, Path(args.data_dir))
if not new_works and not args.force and not needs_web_data_update:
print(f"Scanned pages: {scanned_pages}")
print("No new works found. web/data update skipped.")
print("No new works found. update skipped.")
return 0
merged_works = merge_raw_works(existing_works, new_works, fetched_existing_by_key)
filtered_works, filter_stats = filter_raw_works(args, merged_works)
web_stats = dry_run_web_stats(args, filtered_works) if args.dry_run else None
web_stats = None
if args.dry_run:
web_stats = {"embeddingDiffs": []} if args.skip_web_data else dry_run_web_stats(args, filtered_works)
print(f"Scanned pages: {scanned_pages}")
print(f"New raw works: {len(new_works)}")
print(f"Raw works after merge: {len(merged_works)}")
print(f"Filtered works: {filter_stats['output']} ({filter_stats['total_removed']} removed)")
if web_stats:
print(f"web/data would reuse {web_stats['reused']} embeddings and request {web_stats['embedded']} embeddings")
if args.skip_web_data:
print("web/data would be skipped")
else:
print(f"web/data would reuse {web_stats['reused']} embeddings and request {web_stats['embedded']} embeddings")
print_dry_run_diff(
args=args,
new_works=new_works,
@@ -743,9 +754,13 @@ def main() -> int:
write_jsonl_atomic(input_path, merged_works)
write_jsonl_atomic(filtered_path, filtered_works)
web_stats = build_incremental_web_data(args, filtered_works)
print(f"Updated {input_path}")
print(f"Updated {filtered_path}")
if args.skip_web_data:
print("Skipped web/data update")
return 0
web_stats = build_incremental_web_data(args, filtered_works)
print(
f"Updated {args.data_dir}: {web_stats['count']} works, "
f"reused {web_stats['reused']} embeddings, requested {web_stats['embedded']} embeddings"