From e1753043f976eda6df6f886bd43b0515c832d81a Mon Sep 17 00:00:00 2001 From: Nick Sweeting Date: Thu, 28 May 2026 07:18:02 -0700 Subject: [PATCH] release: archivebox 0.9.33rc18 --- archivebox/cli/archivebox_update.py | 20 +++++++++++++-- archivebox/services/runner.py | 38 ++++++++++++++++++++--------- etc/package.json | 2 +- pyproject.toml | 8 +++--- 4 files changed, 50 insertions(+), 18 deletions(-) diff --git a/archivebox/cli/archivebox_update.py b/archivebox/cli/archivebox_update.py index 5ae26c33..0e1b79a8 100644 --- a/archivebox/cli/archivebox_update.py +++ b/archivebox/cli/archivebox_update.py @@ -304,7 +304,16 @@ def update( print_combined_stats(stats_combined) if do_run_until_idle: - print("[*] Phase 3: Running queued/interrupted crawl work until idle...") + # Filesystem migration is maintenance on existing + # Snapshot rows: Snapshot.save() moves archive/ to + # the current output_dir and preserves the lifecycle + # status. Drain those retry_at ticks before queuing + # search backfill below. Otherwise the sealed/paused + # runner branch correctly sees queued ArchiveResult + # rows first, runs the targeted plugins, and may leave + # the fs_version maintenance tick hidden behind that + # plugin work until another update pass. + print("[*] Phase 3: Running filesystem maintenance until idle...") from archivebox.cli.archivebox_run import run_runner, run_snapshot_worker if is_filtered_update: @@ -315,7 +324,7 @@ def update( if exit_code != 0: raise SystemExit(exit_code) else: - exit_code = run_runner(daemon=False, maintenance_only=migrate_only) + exit_code = run_runner(daemon=False, maintenance_only=True) if exit_code != 0: raise SystemExit(exit_code) ran_post_migrate_runner = True @@ -352,6 +361,13 @@ def update( touched_snapshot_ids.update(stats.get("snapshot_ids", [])) if do_run_until_idle and (do_index or not ran_post_migrate_runner): + # Search/index backfill intentionally queues targeted + # ArchiveResult rows without reopening sealed/paused + # snapshots. This second runner pass drains those plugin + # rows after filesystem maintenance has had its own turn. + # For a normal unfiltered `archivebox update`, keep the + # historical final pass broad enough to resume genuinely + # queued/interrupted crawl work after maintenance is done. print("[*] Phase 3: Running queued/interrupted crawl work until idle...") from archivebox.cli.archivebox_run import run_runner, run_snapshot_worker diff --git a/archivebox/services/runner.py b/archivebox/services/runner.py index bd2f2a2c..ebcfac3a 100644 --- a/archivebox/services/runner.py +++ b/archivebox/services/runner.py @@ -1192,16 +1192,18 @@ def run_snapshot_maintenance(snapshot_id: str) -> bool: snapshot = Snapshot.objects.select_related("crawl", "crawl__created_by").filter(id=snapshot_id).first() if snapshot is None: return False - if snapshot.archiveresult_set.filter(status=ArchiveResult.StatusChoices.QUEUED).exists(): - return False - # retry_at is the universal "tick me" signal. For already-sealed snapshots, - # a tick with no queued ArchiveResults is maintenance-only: run normal - # save/write side effects like lazy fs migration/json rewriting, then clear - # retry_at. Paused snapshots do not reach this helper while search/index - # plugin rows are queued; run_due_snapshot restores their paused scheduler - # marker after the targeted plugin rows finish. - snapshot.retry_at = None + has_queued_results = snapshot.archiveresult_set.filter(status=ArchiveResult.StatusChoices.QUEUED).exists() + # retry_at is the scheduler signal for both lifecycle work and targeted + # maintenance. Filesystem migration/json rewriting is independent from + # queued ArchiveResult rows, so run it whenever this helper is called. + # The only thing queued rows change is the next scheduler value: + # - no queued rows left: clear retry_at because maintenance is done + # - queued rows remain: leave the Snapshot due so the sealed/paused runner + # branch can process those targeted plugin rows on the next tick + # This avoids reopening final/paused snapshots while also avoiding stranded + # queued ArchiveResults that have no independent scheduler. + snapshot.retry_at = timezone.now() if has_queued_results else None snapshot.save(update_fields=["retry_at", "modified_at"]) snapshot.write_index_jsonl() snapshot.write_json_details() @@ -1284,9 +1286,17 @@ def run_due_snapshot(snapshot, *, lock_seconds: int) -> bool: if snapshot.is_paused: selected_plugins = queued_plugins_for_snapshot(str(snapshot.id)) + if snapshot.fs_migration_needed and Snapshot.claim_for_worker(snapshot, lock_seconds=lock_seconds): + run_snapshot_maintenance(str(snapshot.id)) + if not selected_plugins: + # No targeted plugin rows remain, so put paused snapshots back + # behind the indefinite retry_at marker. If queued plugin rows + # do remain, run_snapshot_maintenance kept retry_at due so the + # next tick can process them and the finally block below will + # restore the paused marker after that targeted work completes. + snapshot.restore_paused_scheduler_marker() + return True if not selected_plugins: - if snapshot.fs_migration_needed and Snapshot.claim_for_worker(snapshot, lock_seconds=lock_seconds): - run_snapshot_maintenance(str(snapshot.id)) # Paused is a real lifecycle state; retry_at=MAX is only the # orchestrator selection marker. If a direct maintenance/update # command bumps retry_at on a paused snapshot but there are no @@ -1318,6 +1328,12 @@ def run_due_snapshot(snapshot, *, lock_seconds: int) -> bool: if not Snapshot.claim_for_worker(snapshot, lock_seconds=lock_seconds): return False snapshot.refresh_from_db() + if snapshot.fs_migration_needed: + # Final snapshots can still need maintenance after an old data-dir + # migration. Run the filesystem/json save path before queued search + # backfill rows so both maintenance streams stay ordered without + # changing Snapshot.status away from SEALED. + return run_snapshot_maintenance(str(snapshot.id)) selected_plugins = queued_plugins_for_snapshot(str(snapshot.id)) if selected_plugins: run_crawl( diff --git a/etc/package.json b/etc/package.json index 5f7f3851..86e105f5 100644 --- a/etc/package.json +++ b/etc/package.json @@ -1,6 +1,6 @@ { "name": "archivebox", - "version": "0.9.33rc17", + "version": "0.9.33rc18", "repository": "github:ArchiveBox/ArchiveBox", "license": "MIT", "dependencies": { diff --git a/pyproject.toml b/pyproject.toml index 71b2de77..f7de0c88 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "archivebox" -version = "0.9.33rc17" +version = "0.9.33rc18" requires-python = ">=3.13" description = "Self-hosted internet archiving solution." authors = [{name = "Nick Sweeting", email = "pyproject.toml@archivebox.io"}] @@ -79,9 +79,9 @@ dependencies = [ ### Extractor dependencies (optional binaries detected at runtime via shutil.which) ### Binary/Package Management "abxbus==2.5.7", # EventBus API - "abxpkg>=1.11.42", # for: detecting, versioning, and installing binaries via apt/brew/pip/npm - "abx-plugins>=1.11.48", # shared ArchiveBox plugin package with Chrome/Puppeteer dependency wiring - "abx-dl>=1.11.48", # shared ArchiveBox downloader package with blocking install preflight + "abxpkg>=1.11.44", # for: detecting, versioning, and installing binaries via apt/brew/pip/npm + "abx-plugins>=1.11.50", # shared ArchiveBox plugin package with Chrome/Puppeteer dependency wiring + "abx-dl>=1.11.50", # shared ArchiveBox downloader package with blocking install preflight ### UUID7 backport for Python <3.14 "uuid7>=0.1.0; python_version < '3.14'", # provides the uuid_extensions module on Python 3.13 ]