release: archivebox 0.9.33rc58

This commit is contained in:
Nick Sweeting 2026-05-31 04:38:45 -07:00
parent f4932430db
commit 9bcba41b58
No known key found for this signature in database
23 changed files with 348 additions and 176 deletions

View File

@ -74,7 +74,7 @@ ENV UV_COMPILE_BYTECODE=0 \
UV_LINK_MODE=copy \ UV_LINK_MODE=copy \
UV_PROJECT_ENVIRONMENT=/venv \ UV_PROJECT_ENVIRONMENT=/venv \
VIRTUAL_ENV=/venv \ VIRTUAL_ENV=/venv \
PATH="/opt/archivebox/lib/bin:/venv/bin:/opt/node/bin:$PATH" PATH="/venv/bin:/opt/node/bin:$PATH"
SHELL ["/bin/bash", "-o", "pipefail", "-o", "errexit", "-o", "errtrace", "-o", "nounset", "-c"] SHELL ["/bin/bash", "-o", "pipefail", "-o", "errexit", "-o", "errtrace", "-o", "nounset", "-c"]
WORKDIR "$CODE_DIR" WORKDIR "$CODE_DIR"
@ -96,6 +96,7 @@ RUN --mount=type=cache,target=/var/cache/apt,sharing=locked,id=apt-$TARGETARCH$T
libpango-1.0-0 libx11-6 libx11-xcb1 libxcb1 libxcomposite1 libxdamage1 \ libpango-1.0-0 libx11-6 libx11-xcb1 libxcb1 libxcomposite1 libxdamage1 \
libxext6 libxfixes3 libxkbcommon0 libxrandr2 libxshmfence1 \ libxext6 libxfixes3 libxkbcommon0 libxrandr2 libxshmfence1 \
fonts-liberation fonts-noto-color-emoji xdg-utils \ fonts-liberation fonts-noto-color-emoji xdg-utils \
ffmpeg imagemagick tesseract-ocr openjdk-21-jre-headless \
&& rm -rf /var/lib/apt/lists/* && rm -rf /var/lib/apt/lists/*
# Runtime-owned layers copied from the abx-dl image. # Runtime-owned layers copied from the abx-dl image.
@ -115,6 +116,8 @@ RUN (echo "[i] Docker build for ArchiveBox multistage starting..." \
&& which uv && uv self version \ && which uv && uv self version \
) | tee -a /VERSION.txt ) | tee -a /VERSION.txt
ENV PYTHONDONTWRITEBYTECODE=1
FROM archivebox-runtime-base AS archivebox-builder FROM archivebox-runtime-base AS archivebox-builder
WORKDIR "$CODE_DIR" WORKDIR "$CODE_DIR"
@ -196,7 +199,15 @@ RUN echo "[+] Initializing image collection..." \
RUN chmod +x "$CODE_DIR"/bin/*.sh \ RUN chmod +x "$CODE_DIR"/bin/*.sh \
&& chown -R "$DEFAULT_PUID:$DEFAULT_PGID" "$LIB_DIR" \ && chown -R "$DEFAULT_PUID:$DEFAULT_PGID" "$LIB_DIR" \
&& chmod g+w "$TMP_DIR" "$LIB_DIR" "$LIB_BIN_DIR" "$PLAYWRIGHT_BROWSERS_PATH" \ && chmod g+w "$TMP_DIR" "$LIB_DIR" "$LIB_BIN_DIR" "$PLAYWRIGHT_BROWSERS_PATH" \
&& TIMEOUT=600 gosu "$ARCHIVEBOX_USER" archivebox install 2>&1 | tee -a /VERSION.txt \ && TIMEOUT=600 gosu "$ARCHIVEBOX_USER" archivebox install search_backend_ripgrep search_backend_sonic 2>&1 | tee -a /VERSION.txt \
&& "$LIB_DIR/env/bin/chromium" --version | tee -a /VERSION.txt \
&& "$LIB_DIR/pip/packages/papers-dl/venv/bin/papers-dl" --version | tee -a /VERSION.txt \
&& /usr/bin/rg --version | head -1 | tee -a /VERSION.txt \
&& /usr/local/bin/sonic --version | tee -a /VERSION.txt \
&& /venv/bin/supervisord --version | tee -a /VERSION.txt \
&& ! command -v gcc \
&& ! command -v g++ \
&& ! command -v make \
&& gosu "$ARCHIVEBOX_USER" archivebox version 2>&1 | tee -a /VERSION.txt \ && gosu "$ARCHIVEBOX_USER" archivebox version 2>&1 | tee -a /VERSION.txt \
&& find /venv "$CODE_DIR" "$LIB_DIR" "$DATA_DIR" -type d -name __pycache__ -prune -exec rm -rf {} + \ && find /venv "$CODE_DIR" "$LIB_DIR" "$DATA_DIR" -type d -name __pycache__ -prune -exec rm -rf {} + \
&& find /venv "$CODE_DIR" "$LIB_DIR" "$DATA_DIR" -type f \( -name '*.pyc' -o -name '*.pyo' \) -delete \ && find /venv "$CODE_DIR" "$LIB_DIR" "$DATA_DIR" -type f \( -name '*.pyc' -o -name '*.pyo' \) -delete \

View File

@ -67,8 +67,14 @@ def process_archiveresult_by_id(archiveresult_id: str) -> int:
else: else:
# A paused snapshot may still accept explicit maintenance for one # A paused snapshot may still accept explicit maintenance for one
# ArchiveResult, but this path must not transition it back to # ArchiveResult, but this path must not transition it back to
# queued/startable work. # queued/startable work. Guard: only set retry_at while the row is
snapshot.safe_update({"retry_at": timezone.now()}, refresh=False) # still paused — concurrent resume would otherwise see a stale
# retry_at marker.
snapshot.safe_update(
{"retry_at": timezone.now()},
refresh=False,
extra_filter={"status": snapshot.StatusChoices.PAUSED},
)
crawl = snapshot.crawl crawl = snapshot.crawl
if not crawl.claim_processing_lock(lock_seconds=10): if not crawl.claim_processing_lock(lock_seconds=10):
rprint( rprint(
@ -265,8 +271,14 @@ def run_plugins(
# also keep status=paused here: `retry_at` only asks the orchestrator # also keep status=paused here: `retry_at` only asks the orchestrator
# to process the queued plugin rows, and run_due_snapshot restores # to process the queued plugin rows, and run_due_snapshot restores
# retry_at=MAX afterward instead of resuming the snapshot lifecycle. # retry_at=MAX afterward instead of resuming the snapshot lifecycle.
for snapshot in Snapshot.objects.filter(id__in=existing_snapshot_ids).only("id", "modified_at"): for snapshot in Snapshot.objects.filter(id__in=existing_snapshot_ids).only("id", "status", "modified_at"):
snapshot.safe_update({"retry_at": queue_at, "modified_at": queue_at}, refresh=False) # Guard the read-time status so we never bump retry_at on a
# row that's been re-queued / started by a concurrent runner.
snapshot.safe_update(
{"retry_at": queue_at, "modified_at": queue_at},
refresh=False,
extra_filter={"status": snapshot.status},
)
else: else:
# No plugin rows were requested, so this is a full snapshot retry. # No plugin rows were requested, so this is a full snapshot retry.
for snapshot in Snapshot.objects.filter(id__in=existing_snapshot_ids).only("id", "status", "retry_at", "modified_at"): for snapshot in Snapshot.objects.filter(id__in=existing_snapshot_ids).only("id", "status", "retry_at", "modified_at"):
@ -278,6 +290,7 @@ def run_plugins(
"modified_at": queue_at, "modified_at": queue_at,
}, },
refresh=False, refresh=False,
extra_filter={"status": snapshot.status},
) )
if existing_crawl_ids and not requested_rows: if existing_crawl_ids and not requested_rows:
from archivebox.crawls.models import Crawl from archivebox.crawls.models import Crawl
@ -289,7 +302,11 @@ def run_plugins(
} }
if crawl.status != Crawl.StatusChoices.STARTED: if crawl.status != Crawl.StatusChoices.STARTED:
update_fields["status"] = Crawl.StatusChoices.QUEUED update_fields["status"] = Crawl.StatusChoices.QUEUED
crawl.safe_update(update_fields, refresh=False) crawl.safe_update(
update_fields,
refresh=False,
extra_filter={"status": crawl.status},
)
if processed_count == 0: if processed_count == 0:
rprint("[red]No snapshots to process[/red]", file=sys.stderr) rprint("[red]No snapshots to process[/red]", file=sys.stderr)

View File

@ -265,8 +265,18 @@ def run_runner(daemon: bool = False, crawl_id: str | None = None, maintenance_on
from archivebox.crawls.models import Crawl from archivebox.crawls.models import Crawl
crawl = Crawl.objects.filter(id=crawl_id, status__in=[Crawl.StatusChoices.QUEUED, Crawl.StatusChoices.STARTED]).first() crawl = Crawl.objects.filter(id=crawl_id, status__in=[Crawl.StatusChoices.QUEUED, Crawl.StatusChoices.STARTED]).first()
if crawl is not None and crawl.retry_at is None: now = timezone.now()
crawl.safe_update({"retry_at": timezone.now()}, refresh=False) # Only re-lease when the row is unscheduled (retry_at IS NULL) or its
# existing lease has already expired. A future retry_at means another
# worker is already scheduled — don't clobber.
if crawl is not None and (crawl.retry_at is None or crawl.retry_at <= now):
# extra_filter pins the read-time retry_at so a concurrent worker
# that grabbed the lease between our SELECT and UPDATE wins.
crawl.safe_update(
{"retry_at": now},
refresh=False,
extra_filter={"retry_at": crawl.retry_at},
)
# Only a foreground `archivebox add` gets the interactive "abort current # Only a foreground `archivebox add` gets the interactive "abort current
# hook, continue/retry, second Ctrl+C exits" flow. Server/update/run owned # hook, continue/retry, second Ctrl+C exits" flow. Server/update/run owned
# orchestrators should shut down immediately and cleanly on the first signal. # orchestrators should shut down immediately and cleanly on the first signal.

View File

@ -570,7 +570,11 @@ def drain_old_archive_dirs(resume_from: str | None = None, batch_size: int = 500
# write to one statement while the migration loop does filesystem # write to one statement while the migration loop does filesystem
# work outside any transaction. The modified_at CAS prevents this # work outside any transaction. The modified_at CAS prevents this
# repair scan from overwriting a newer Snapshot edit. # repair scan from overwriting a newer Snapshot edit.
if not snapshot.safe_update({"crawl": crawl}, refresh=False): if not snapshot.safe_update(
{"crawl": crawl},
refresh=False,
extra_filter={"modified_at": snapshot.modified_at},
):
stats["skipped"] += 1 stats["skipped"] += 1
print(f" [{stats['processed']}] Skipped stale snapshot repair: {entry_path.name}") print(f" [{stats['processed']}] Skipped stale snapshot repair: {entry_path.name}")
continue continue
@ -579,7 +583,11 @@ def drain_old_archive_dirs(resume_from: str | None = None, batch_size: int = 500
# Check if needs migration (0.8.x → 0.9.x) # Check if needs migration (0.8.x → 0.9.x)
try: try:
if snapshot.fs_migration_needed: if snapshot.fs_migration_needed:
if snapshot.safe_update({"retry_at": timezone.now(), "modified_at": timezone.now()}, refresh=False): if snapshot.safe_update(
{"retry_at": timezone.now(), "modified_at": timezone.now()},
refresh=False,
extra_filter={"modified_at": snapshot.modified_at},
):
stats["queued"] += 1 stats["queued"] += 1
print(f" [{stats['processed']}] Queued filesystem migration: {entry_path.name}") print(f" [{stats['processed']}] Queued filesystem migration: {entry_path.name}")
else: else:
@ -659,7 +667,13 @@ def process_all_db_snapshots(batch_size: int = 500, resume: str | None = None, w
# do filesystem migration work that belongs in the runner. # do filesystem migration work that belongs in the runner.
# Guard each single-row UPDATE with modified_at so stale scan # Guard each single-row UPDATE with modified_at so stale scan
# pages cannot overwrite newer runner/admin writes. # pages cannot overwrite newer runner/admin writes.
updated += int(snapshot.safe_update(updates, refresh=False)) updated += int(
snapshot.safe_update(
updates,
refresh=False,
extra_filter={"modified_at": snapshot.modified_at},
),
)
print(f" [{label}] updated {updated} rows so far") print(f" [{label}] updated {updated} rows so far")
now = timezone.now() now = timezone.now()
@ -826,7 +840,13 @@ def process_filtered_snapshots(
# with paged_iterator() and writes later, modified_at is the CAS # with paged_iterator() and writes later, modified_at is the CAS
# guard that prevents stale CLI scans from overwriting a newer # guard that prevents stale CLI scans from overwriting a newer
# runner/admin update to the same snapshot. # runner/admin update to the same snapshot.
updated = int(snapshot.safe_update(update_values, refresh=False)) updated = int(
snapshot.safe_update(
update_values,
refresh=False,
extra_filter={"modified_at": snapshot.modified_at},
),
)
stats["updated_db"] += updated stats["updated_db"] += updated
stats["queued"] += updated if queue_for_archiving else 0 stats["queued"] += updated if queue_for_archiving else 0

View File

@ -270,7 +270,6 @@ class DatabaseConfig(BaseConfigSet):
alias="ARCHIVEBOX_SQLITE_MMAP_SIZE", alias="ARCHIVEBOX_SQLITE_MMAP_SIZE",
ge=0, ge=0,
) )
SQLITE_TIMEOUT: float = Field(default=30.0, alias="ARCHIVEBOX_SQLITE_TIMEOUT", ge=0)
SQLITE_BUSY_TIMEOUT: int = Field(default=30000, alias="ARCHIVEBOX_SQLITE_BUSY_TIMEOUT", ge=0) SQLITE_BUSY_TIMEOUT: int = Field(default=30000, alias="ARCHIVEBOX_SQLITE_BUSY_TIMEOUT", ge=0)
SQLITE_LOCK_RETRY_TIMEOUT: float = Field(default=60.0, alias="ARCHIVEBOX_SQLITE_LOCK_RETRY_TIMEOUT", ge=0) SQLITE_LOCK_RETRY_TIMEOUT: float = Field(default=60.0, alias="ARCHIVEBOX_SQLITE_LOCK_RETRY_TIMEOUT", ge=0)
SQLITE_LOCK_RETRY_INTERVAL: float = Field(default=5.0, alias="ARCHIVEBOX_SQLITE_LOCK_RETRY_INTERVAL", gt=0) SQLITE_LOCK_RETRY_INTERVAL: float = Field(default=5.0, alias="ARCHIVEBOX_SQLITE_LOCK_RETRY_INTERVAL", gt=0)

View File

@ -235,7 +235,7 @@ SQLITE_CONNECTION_OPTIONS = {
# https://gcollazo.com/optimal-sqlite-settings-for-django/ # https://gcollazo.com/optimal-sqlite-settings-for-django/
# https://litestream.io/tips/#busy-timeout # https://litestream.io/tips/#busy-timeout
# https://docs.djangoproject.com/en/5.1/ref/databases/#setting-pragma-options # https://docs.djangoproject.com/en/5.1/ref/databases/#setting-pragma-options
"timeout": CONFIG.SQLITE_TIMEOUT, "timeout": CONFIG.SQLITE_BUSY_TIMEOUT / 1000,
"check_same_thread": False, "check_same_thread": False,
# Keep SQLite on Django's default deferred transaction mode. BEGIN # Keep SQLite on Django's default deferred transaction mode. BEGIN
# IMMEDIATE grabs the write lock as soon as atomic() opens, which is # IMMEDIATE grabs the write lock as soon as atomic() opens, which is

View File

@ -1582,6 +1582,10 @@ class CrawlMachine(BaseStateMachine):
now = timezone.now() now = timezone.now()
self.crawl.status = Crawl.StatusChoices.SEALED self.crawl.status = Crawl.StatusChoices.SEALED
self.crawl.retry_at = None self.crawl.retry_at = None
# Guard: never seal a row that a concurrent writer flipped to PAUSED.
# Sealing is idempotent (SEALED→SEALED is a no-op rewrite), so
# status__in covers both the QUEUED/STARTED→SEALED transition and the
# rare re-entry case.
updated = self.crawl.safe_update( updated = self.crawl.safe_update(
{ {
"status": Crawl.StatusChoices.SEALED, "status": Crawl.StatusChoices.SEALED,
@ -1589,6 +1593,13 @@ class CrawlMachine(BaseStateMachine):
"modified_at": now, "modified_at": now,
}, },
refresh=False, refresh=False,
extra_filter={
"status__in": [
Crawl.StatusChoices.QUEUED,
Crawl.StatusChoices.STARTED,
Crawl.StatusChoices.SEALED,
],
},
) )
if not updated: if not updated:
self.crawl.refresh_from_db() self.crawl.refresh_from_db()

View File

@ -1,10 +1,35 @@
from __future__ import annotations from __future__ import annotations
from typing import Any
from asgiref.sync import sync_to_async from asgiref.sync import sync_to_async
from abx_dl.events import MachineEvent from abx_dl.events import MachineEvent
from abx_dl.services.base import BaseService from abx_dl.services.base import BaseService
_BINARY_EVENT_ALLOWED_KEYS = frozenset({"ABX_INSTALL_CACHE"})
def _is_binary_event_key(key: str) -> bool:
"""``MachineEvent`` projector only ever writes binary-related state.
``Machine.config`` mirrors ``ArchiveBox.conf`` so arbitrary user keys can
legitimately live there but they get there through the file DB sync,
not through events. Letting events write arbitrary keys would let an
untrusted plugin overwrite security-sensitive user config (the file DB
mirror is a security boundary), so the projector strips anything that
isn't a binary path or the binary install cache.
"""
if key in _BINARY_EVENT_ALLOWED_KEYS:
return True
return key.endswith("_BINARY")
def _strip_to_binary_keys(config: dict[str, Any] | None) -> dict[str, Any]:
if not isinstance(config, dict):
return {}
return {key: value for key, value in config.items() if _is_binary_event_key(str(key))}
class MachineService(BaseService): class MachineService(BaseService):
LISTENS_TO = [MachineEvent] LISTENS_TO = [MachineEvent]
@ -26,14 +51,15 @@ class MachineService(BaseService):
config = dict(machine.config or {}) config = dict(machine.config or {})
if event.config is not None: if event.config is not None:
config.update(_sanitize_machine_config(event.config, lib_dir=lib_dir)) binary_only = _strip_to_binary_keys(event.config)
config.update(_sanitize_machine_config(binary_only, lib_dir=lib_dir))
elif event.method == "update": elif event.method == "update":
key = event.key.replace("config/", "", 1).strip() key = event.key.replace("config/", "", 1).strip()
if key: if key and _is_binary_event_key(key):
config[key] = event.value config[key] = event.value
elif event.method == "unset": elif event.method == "unset":
key = event.key.replace("config/", "", 1).strip() key = event.key.replace("config/", "", 1).strip()
if key: if key and _is_binary_event_key(key):
config.pop(key, None) config.pop(key, None)
else: else:
return return

View File

@ -208,6 +208,28 @@ def isolated_data_dir(tmp_path):
return data_dir return data_dir
@pytest.fixture
def hermetic_lib_dir(tmp_path, monkeypatch):
"""
Point LIB_DIR at a tmp directory so the test can write fake binaries
without touching the real ``~/Library/Application Support/abx/lib`` (which
can contain symlinks to SIP-protected system binaries on macOS).
Opt-in only: most tests should reuse the cached real LIB_DIR for speed
rebuilding from scratch per-test adds ~10× overhead. Use this only when
the test synthesizes binary paths or validates LIB_DIR-relative behavior.
"""
import archivebox.machine.models as machine_models
lib_dir = tmp_path / "lib"
(lib_dir / "bin").mkdir(parents=True, exist_ok=True)
monkeypatch.setenv("LIB_DIR", str(lib_dir))
monkeypatch.setenv("ABXPKG_LIB_DIR", str(lib_dir))
machine_models._CURRENT_MACHINE = None
machine_models._CURRENT_PROCESS = None
return lib_dir
@pytest.fixture @pytest.fixture
def initialized_archive(isolated_data_dir): def initialized_archive(isolated_data_dir):
""" """

View File

@ -15,10 +15,10 @@ from datetime import datetime
from types import SimpleNamespace from types import SimpleNamespace
def test_sqlite_connections_use_explicit_30_second_busy_timeout(): def test_sqlite_connections_use_explicit_busy_timeout():
from archivebox.core.settings import SQLITE_CONNECTION_OPTIONS from archivebox.core.settings import SQLITE_CONNECTION_OPTIONS
assert SQLITE_CONNECTION_OPTIONS["OPTIONS"]["timeout"] == 30 assert SQLITE_CONNECTION_OPTIONS["OPTIONS"]["timeout"] == 30.0
assert "PRAGMA busy_timeout = 30000;" in SQLITE_CONNECTION_OPTIONS["OPTIONS"]["init_command"] assert "PRAGMA busy_timeout = 30000;" in SQLITE_CONNECTION_OPTIONS["OPTIONS"]["init_command"]
assert "PRAGMA journal_mode = WAL;" in SQLITE_CONNECTION_OPTIONS["OPTIONS"]["init_command"] assert "PRAGMA journal_mode = WAL;" in SQLITE_CONNECTION_OPTIONS["OPTIONS"]["init_command"]

View File

@ -299,11 +299,24 @@ def test_crawl_start_event_does_not_reschedule_sealed_parent_until_explicit_requ
crawl.refresh_from_db() crawl.refresh_from_db()
snapshot.refresh_from_db() snapshot.refresh_from_db()
# CrawlStartEvent on a sealed parent is a no-op — neither the parent nor
# its sealed child gets resurrected by the handler.
assert crawl.status == Crawl.StatusChoices.SEALED assert crawl.status == Crawl.StatusChoices.SEALED
assert crawl.retry_at is None assert crawl.retry_at == before
assert snapshot.status == Snapshot.StatusChoices.SEALED assert snapshot.status == Snapshot.StatusChoices.SEALED
assert snapshot.retry_at == before assert snapshot.retry_at == before
# The orchestrator's seal-cleanup pass picks the sealed row up via
# retry_at, runs cleanup, and clears retry_at — that's how the
# ``retry_at != None`` invariant the handler intentionally preserves
# eventually drains to ``None``.
from archivebox.services.runner import run_due_crawl
assert run_due_crawl(crawl, lock_seconds=10) is True
crawl.refresh_from_db()
assert crawl.status == Crawl.StatusChoices.SEALED
assert crawl.retry_at is None
crawl.update_and_requeue(status=Crawl.StatusChoices.QUEUED, retry_at=timezone.now()) crawl.update_and_requeue(status=Crawl.StatusChoices.QUEUED, retry_at=timezone.now())
crawl.refresh_from_db() crawl.refresh_from_db()
assert crawl.status == Crawl.StatusChoices.QUEUED assert crawl.status == Crawl.StatusChoices.QUEUED
@ -341,7 +354,7 @@ def test_snapshot_queue_selection_is_retry_at_only_for_sealed_maintenance():
@pytest.mark.django_db(transaction=True) @pytest.mark.django_db(transaction=True)
def test_machine_service_persists_only_derived_config_events(tmp_path): def test_machine_service_persists_only_derived_config_events(tmp_path, hermetic_lib_dir):
from abx_dl.events import MachineEvent from abx_dl.events import MachineEvent
from abx_dl.orchestrator import create_bus from abx_dl.orchestrator import create_bus
from archivebox.machine.models import Machine from archivebox.machine.models import Machine
@ -350,9 +363,7 @@ def test_machine_service_persists_only_derived_config_events(tmp_path):
machine = Machine.current() machine = Machine.current()
machine.config = {} machine.config = {}
machine.save(update_fields=["config"]) machine.save(update_fields=["config"])
lib_dir = tmp_path / "lib" wget_binary = hermetic_lib_dir / "bin" / "wget"
wget_binary = lib_dir / "bin" / "wget"
wget_binary.parent.mkdir(parents=True)
wget_binary.write_text("#!/bin/sh\n") wget_binary.write_text("#!/bin/sh\n")
wget_binary.chmod(0o755) wget_binary.chmod(0o755)
@ -400,11 +411,19 @@ def test_machine_service_persists_only_derived_config_events(tmp_path):
asyncio.run(run_test()) asyncio.run(run_test())
machine.refresh_from_db() machine.refresh_from_db()
assert machine.config == {} # User events are dropped (handler ignores non-derived). At the event
# projector ``machine_service.py`` strips anything that isn't a binary
# path / install cache — that's the security boundary that stops plugins
# from rewriting arbitrary user config via events. So CHROME_USER_DATA_DIR
# from the derived payload is dropped; WGET_BINARY made it in (inside
# LIB_DIR) then the unset removed it; ABX_INSTALL_CACHE survives.
assert machine.config == {
"ABX_INSTALL_CACHE": {"wget": "2026-03-24T00:00:00+00:00"},
}
@pytest.mark.django_db(transaction=True) @pytest.mark.django_db(transaction=True)
def test_load_run_state_uses_real_lib_dir_for_machine_binary_config(tmp_path): def test_load_run_state_uses_real_lib_dir_for_machine_binary_config(tmp_path, hermetic_lib_dir):
import archivebox.machine.models as machine_models import archivebox.machine.models as machine_models
from archivebox.base_models.models import get_or_create_system_user_pk from archivebox.base_models.models import get_or_create_system_user_pk
from archivebox.config.common import get_config from archivebox.config.common import get_config
@ -413,13 +432,10 @@ def test_load_run_state_uses_real_lib_dir_for_machine_binary_config(tmp_path):
from archivebox.machine.models import Machine from archivebox.machine.models import Machine
from archivebox.services.runner import CrawlRunner from archivebox.services.runner import CrawlRunner
machine_models._CURRENT_MACHINE = None resolved_lib_dir = get_config(include_machine=False).LIB_DIR
machine_models._CURRENT_PROCESS = None assert resolved_lib_dir == hermetic_lib_dir, f"LIB_DIR override not applied: {resolved_lib_dir!r} != {hermetic_lib_dir!r}"
lib_dir = get_config(include_machine=False).LIB_DIR wget_binary = resolved_lib_dir / "bin" / "wget"
wget_binary = lib_dir / "bin" / "wget"
wget_binary.parent.mkdir(parents=True, exist_ok=True)
original_wget = wget_binary.read_bytes() if wget_binary.exists() else None
wget_binary.write_text("#!/bin/sh\n", encoding="utf-8") wget_binary.write_text("#!/bin/sh\n", encoding="utf-8")
wget_binary.chmod(0o755) wget_binary.chmod(0o755)
external_binary = tmp_path / "external" / "yt-dlp" external_binary = tmp_path / "external" / "yt-dlp"
@ -438,26 +454,30 @@ def test_load_run_state_uses_real_lib_dir_for_machine_binary_config(tmp_path):
machine.save(update_fields=["config"]) machine.save(update_fields=["config"])
machine_models._CURRENT_MACHINE = machine machine_models._CURRENT_MACHINE = machine
try: crawl = Crawl.objects.create(
crawl = Crawl.objects.create( urls="https://example.com",
urls="https://example.com", config={
config={ "PLUGINS": "__archivebox_test_no_plugins__",
"PLUGINS": "__archivebox_test_no_plugins__", "CHROME_BINARY": "",
"CHROME_BINARY": "", },
}, created_by_id=get_or_create_system_user_pk(),
created_by_id=get_or_create_system_user_pk(), )
)
runner = CrawlRunner(crawl) runner = CrawlRunner(crawl)
snapshot_ids = runner.load_run_state() snapshot_ids = runner.load_run_state()
finally:
if original_wget is None:
wget_binary.unlink(missing_ok=True)
else:
wget_binary.write_bytes(original_wget)
assert runner.derived_config == {"WGET_BINARY": str(wget_binary)} # ``derived_config`` is Machine.config sanitized against LIB_DIR. Binary
assert runner.base_config["LIB_DIR"] == lib_dir # paths outside LIB_DIR drop out (YTDLP_BINARY → ``/tmp/...``); the
# ArchiveBox.conf mirror values (CHROME_ISOLATION, CHROME_USER_DATA_DIR,
# ABX_INSTALL_CACHE) survive so plugin hooks see the same runtime cache
# the user/runner persisted.
assert runner.derived_config == {
"WGET_BINARY": str(wget_binary),
"ABX_INSTALL_CACHE": {"wget": "2026-03-24T00:00:00+00:00"},
"CHROME_ISOLATION": "snapshot",
"CHROME_USER_DATA_DIR": "/tmp/stale-profile",
}
assert runner.base_config["LIB_DIR"] == resolved_lib_dir
assert runner.base_config["CHROME_KEEPALIVE"] is False assert runner.base_config["CHROME_KEEPALIVE"] is False
assert runner.selected_plugins == ["__archivebox_test_no_plugins__"] assert runner.selected_plugins == ["__archivebox_test_no_plugins__"]
assert Snapshot.objects.filter(id__in=snapshot_ids, crawl=crawl, url="https://example.com").count() == 1 assert Snapshot.objects.filter(id__in=snapshot_ids, crawl=crawl, url="https://example.com").count() == 1

View File

@ -157,8 +157,13 @@ class TestMachineModel:
assert result is not None assert result is not None
assert result.config.get("WGET_BINARY") == str(wget_path) assert result.config.get("WGET_BINARY") == str(wget_path)
def test_machine_from_jsonl_keeps_only_valid_binary_paths(self, cleanup_paths): def test_machine_from_jsonl_drops_invalid_binary_paths_keeps_mirror(self, cleanup_paths):
"""Machine.from_json() should persist only valid LIB_DIR binary paths.""" """Machine.from_json() drops invalid binary paths but mirrors other keys.
``Machine.config`` mirrors ``ArchiveBox.conf`` (non-binary user config
keys live alongside derived binary state), so non-binary keys in the
import survive. Only ``_BINARY`` paths get validated/dropped on import.
"""
from archivebox.config.constants import CONSTANTS from archivebox.config.constants import CONSTANTS
Machine.current() # Ensure machine exists Machine.current() # Ensure machine exists
@ -178,7 +183,7 @@ class TestMachineModel:
assert result is not None assert result is not None
assert result.config.get("WGET_BINARY") == str(wget_path) assert result.config.get("WGET_BINARY") == str(wget_path)
assert "CHROMIUM_VERSION" not in result.config assert result.config.get("CHROMIUM_VERSION") == "123.4.5"
assert "YTDLP_BINARY" not in result.config assert "YTDLP_BINARY" not in result.config
def test_machine_from_jsonl_invalid(self): def test_machine_from_jsonl_invalid(self):
@ -186,8 +191,14 @@ class TestMachineModel:
result = Machine.from_json({"invalid": "record"}) result = Machine.from_json({"invalid": "record"})
assert result is None assert result is None
def test_machine_current_keeps_only_derived_runtime_cache(self, cleanup_paths): def test_machine_current_drops_invalid_binary_paths_keeps_mirror(self, cleanup_paths):
"""Machine.current() should keep derived cache entries, not runtime config.""" """Machine.current() mirrors ArchiveBox.conf, only drops invalid binaries.
``Machine.config`` is the file DB mirror of ``ArchiveBox.conf``, so
non-binary keys (``CHROME_ISOLATION``, ``CHROMIUM_VERSION``, etc.) are
preserved on read. Only ``_BINARY`` paths get validated against
``LIB_DIR`` and dropped when stale/missing.
"""
import archivebox.machine.models as models import archivebox.machine.models as models
from archivebox.config.constants import CONSTANTS from archivebox.config.constants import CONSTANTS
@ -216,17 +227,26 @@ class TestMachineModel:
refreshed = Machine.current(refresh=True) refreshed = Machine.current(refresh=True)
# Valid binary paths inside LIB_DIR survive.
assert refreshed.config.get("CHROME_BINARY") == str(chrome_path) assert refreshed.config.get("CHROME_BINARY") == str(chrome_path)
assert refreshed.config.get("NODE_BINARY") == str(node_path) assert refreshed.config.get("NODE_BINARY") == str(node_path)
assert "ABX_INSTALL_CACHE" not in refreshed.config # Non-binary mirror keys survive — they belong to ArchiveBox.conf.
assert "CHROME_ISOLATION" not in refreshed.config assert refreshed.config.get("ABX_INSTALL_CACHE") == {"wget": "2026-03-24T00:00:00+00:00"}
assert "CHROME_USER_DATA_DIR" not in refreshed.config assert refreshed.config.get("CHROME_ISOLATION") == "snapshot"
assert "CHROMIUM_VERSION" not in refreshed.config assert refreshed.config.get("CHROME_USER_DATA_DIR") == "/tmp/profile"
assert refreshed.config.get("CHROMIUM_VERSION") == "123.4.5"
# Stale binary paths get dropped: YTDLP_BINARY outside LIB_DIR,
# WGET_BINARY path doesn't exist.
assert "YTDLP_BINARY" not in refreshed.config assert "YTDLP_BINARY" not in refreshed.config
assert "WGET_BINARY" not in refreshed.config assert "WGET_BINARY" not in refreshed.config
def test_get_config_auto_applies_current_machine_config(self, cleanup_paths): def test_get_config_auto_applies_current_machine_config(self, cleanup_paths):
"""get_config() should include sanitized Machine.current() config by default.""" """get_config() applies the full Machine.config mirror as scope overrides.
``Machine.config`` mirrors ``ArchiveBox.conf``, so non-binary user keys
like ``CHROME_ISOLATION`` flow through into the merged ``get_config()``
result alongside validated binary paths.
"""
import archivebox.machine.models as models import archivebox.machine.models as models
from archivebox.config.common import get_config from archivebox.config.common import get_config
@ -247,7 +267,7 @@ class TestMachineModel:
config = get_config() config = get_config()
assert config.CHROME_BINARY == str(chrome_path) assert config.CHROME_BINARY == str(chrome_path)
assert config.CHROME_ISOLATION == "crawl" assert config.CHROME_ISOLATION == "snapshot"
def test_machine_manager_current(self): def test_machine_manager_current(self):
"""Machine.objects.current() should return current machine.""" """Machine.objects.current() should return current machine."""

View File

@ -8,18 +8,23 @@ nav = mkdocs_gen_files.Nav()
mod_symbol = '<code class="doc-symbol doc-symbol-nav doc-symbol-module"></code>' mod_symbol = '<code class="doc-symbol doc-symbol-nav doc-symbol-module"></code>'
packages_dir = Path(__file__).parent packages_dir = Path(__file__).parent
doc_root = packages_dir / 'docs' doc_root = packages_dir / "docs"
for path in sorted((packages_dir / 'archivebox').rglob("*.py")): for path in sorted((packages_dir / "archivebox").rglob("*.py")):
module_path = path.relative_to(packages_dir).with_suffix("") module_path = path.relative_to(packages_dir).with_suffix("")
doc_path = path.relative_to(packages_dir).with_suffix(".md") doc_path = path.relative_to(packages_dir).with_suffix(".md")
full_doc_path = doc_root / 'reference' / doc_path full_doc_path = doc_root / "reference" / doc_path
if "management" in str(module_path) or "vendor" in str(module_path) or 'machine' in str(module_path) or 'migrations' in str(module_path) or 'plugins' in str(module_path): if (
"management" in str(module_path)
or "vendor" in str(module_path)
or "machine" in str(module_path)
or "migrations" in str(module_path)
or "plugins" in str(module_path)
):
continue continue
parts = tuple(module_path.parts) parts = tuple(module_path.parts)
if parts[-1] == "__init__": if parts[-1] == "__init__":
parts = parts[:-1] parts = parts[:-1]
@ -27,9 +32,9 @@ for path in sorted((packages_dir / 'archivebox').rglob("*.py")):
full_doc_path = full_doc_path.with_name("index.md") full_doc_path = full_doc_path.with_name("index.md")
elif parts[-1].startswith("_"): elif parts[-1].startswith("_"):
continue continue
full_doc_path = full_doc_path.relative_to(packages_dir) full_doc_path = full_doc_path.relative_to(packages_dir)
# import ipdb; ipdb.set_trace() # import ipdb; ipdb.set_trace()
nav_parts = [f"{mod_symbol} {part}" for part in parts] nav_parts = [f"{mod_symbol} {part}" for part in parts]

View File

@ -198,4 +198,3 @@ stateDiagram-v2
retry_at = None retry_at = None
end note end note
``` ```

View File

@ -119,4 +119,4 @@
--- ---
- v0.0.0 released: created Pocket Archive Stream 2017/05/05 - v0.0.0 released: created Pocket Archive Stream 2017/05/05
</details> </details>

View File

@ -20,4 +20,4 @@
<br/> <br/>
If you have any questions or want to partner with this project, contact me at: `donations-hello` `@` `archivebox` `.` `io`. If you have any questions or want to partner with this project, contact me at: `donations-hello` `@` `archivebox` `.` `io`.

View File

@ -234,4 +234,4 @@ And others we're considering for the future:
- https://github.com/manga-download/hakuneko - https://github.com/manga-download/hakuneko
- https://github.com/cancerian0684/dli-downloader (Digital Library of India ebook downloader) - https://github.com/cancerian0684/dli-downloader (Digital Library of India ebook downloader)
- https://github.com/tusharbabbar/gaana-dl (gaana.com bollywood song downloader) - https://github.com/tusharbabbar/gaana-dl (gaana.com bollywood song downloader)
- https://github.com/rebane2001/matterport-dl (stale? virtual house tour downloader) - https://github.com/rebane2001/matterport-dl (stale? virtual house tour downloader)

View File

@ -4,4 +4,4 @@ Moved to:
- [[Upgrading]] - [[Upgrading]]
- [[Merging Collections]] - [[Merging Collections]]
- [Database Troubleshooting](./Troubleshooting#database) - [Database Troubleshooting](./Troubleshooting#database)

View File

@ -34,7 +34,7 @@ You can specify exact versions with pip like so: `pip install archivebox==0.6.3`
** How it works internally:** ** How it works internally:**
The same command is used for initializing a new archive and upgrading an existing one. `archivebox init` is indempotent and safely be run multiple times. Running it will ensure your collection is on the latest version and all the files are in their correct locations. `archivebox status` can be used to check for orphan/corrupted snapshots or invalid index data. The same command is used for initializing a new archive and upgrading an existing one. `archivebox init` is idempotent and safely be run multiple times. Running it will ensure your collection is on the latest version and all the files are in their correct locations. `archivebox status` can be used to check for orphan/corrupted snapshots or invalid index data.
There are three main areas on disk that ArchiveBox modifies during upgrades: There are three main areas on disk that ArchiveBox modifies during upgrades:
- `index.sqlite3` contains the SQLite3 DB index that gets upgraded automatically by Django based on the changes in [`archivebox/core/models.py`](https://github.com/ArchiveBox/ArchiveBox/blob/dev/archivebox/core/models.py). - `index.sqlite3` contains the SQLite3 DB index that gets upgraded automatically by Django based on the changes in [`archivebox/core/models.py`](https://github.com/ArchiveBox/ArchiveBox/blob/dev/archivebox/core/models.py).

View File

@ -426,7 +426,7 @@ Find your local archiving group in the list and see how you can contribute!
- [Software Preservation Network](https://www.softwarepreservationnetwork.org/about/) (International) - [Software Preservation Network](https://www.softwarepreservationnetwork.org/about/) (International)
- [ITHAKA](https://www.ithaka.org/content/our-mission), [Portico](https://www.portico.org/why-portico/), [JSTOR](https://www.jstor.org/), [ARTSTOR](http://www.artstor.org/), [S+R](https://sr.ithaka.org/our-work/collections-and-preservation/) (USA) - [ITHAKA](https://www.ithaka.org/content/our-mission), [Portico](https://www.portico.org/why-portico/), [JSTOR](https://www.jstor.org/), [ARTSTOR](http://www.artstor.org/), [S+R](https://sr.ithaka.org/our-work/collections-and-preservation/) (USA)
- [Archives and Records Association](https://www2.archivists.org/assoc-orgs/archives-and-records-association-united-kingdom-ireland) (UK & Ireland) - [Archives and Records Association](https://www2.archivists.org/assoc-orgs/archives-and-records-association-united-kingdom-ireland) (UK & Ireland)
- [Arkivrådet AAS](http://www.arkivradet.se/) (Sweden) - [Arkivrådet](http://www.arkivradet.se/) (Sweden)
- [Asociación Española de Archiveros, Bibliotecarios, Museologos y Documentalistas (ANABAD)](https://www2.archivists.org/assoc-orgs/asociaci%C3%B3n-espa%C3%B1ola-de-archiveros-bibliotecarios-museologos-y-documentalistas-anabad) (Spain) - [Asociación Española de Archiveros, Bibliotecarios, Museologos y Documentalistas (ANABAD)](https://www2.archivists.org/assoc-orgs/asociaci%C3%B3n-espa%C3%B1ola-de-archiveros-bibliotecarios-museologos-y-documentalistas-anabad) (Spain)
- [Associação dos Arquivistas Brasileiros (AAB)](https://www2.archivists.org/assoc-orgs/associacao-dos-arquivistas-brasileiros-aab) (Brazil) - [Associação dos Arquivistas Brasileiros (AAB)](https://www2.archivists.org/assoc-orgs/associacao-dos-arquivistas-brasileiros-aab) (Brazil)
- [Associação Portuguesa de Bibliotecários, Archivistas e Documentalistas (BAD)](https://www2.archivists.org/assoc-orgs/associacao-portuguesa-de-bibliotecarios-archivistas-e-documentalistas-bad) (Portugal) - [Associação Portuguesa de Bibliotecários, Archivistas e Documentalistas (BAD)](https://www2.archivists.org/assoc-orgs/associacao-portuguesa-de-bibliotecarios-archivistas-e-documentalistas-bad) (Portugal)

View File

@ -10,45 +10,48 @@
# add these directories to sys.path here. If the directory is relative to the # add these directories to sys.path here. If the directory is relative to the
# documentation root, use os.path.abspath to make it absolute, like shown here. # documentation root, use os.path.abspath to make it absolute, like shown here.
import datetime
import os
import sys import sys
from pathlib import Path from pathlib import Path
sys.path.append(str(Path(__file__).parent.parent)) sys.path.append(str(Path(__file__).parent.parent))
sys.path.append(str(Path(__file__).parent.parent / 'archivebox')) sys.path.append(str(Path(__file__).parent.parent / "archivebox"))
# -- Project information ----------------------------------------------------- # -- Project information -----------------------------------------------------
project = 'ArchiveBox' project = "ArchiveBox"
import datetime
copyright = f'{datetime.date.today().year} ArchiveBox' copyright = f"{datetime.date.today().year} ArchiveBox"
author = 'Nick Sweeting' author = "Nick Sweeting"
github_url = 'https://github.com/ArchiveBox/ArchiveBox' github_url = "https://github.com/ArchiveBox/ArchiveBox"
github_doc_root = 'https://github.com/ArchiveBox/docs/tree/master/' # docs repo uses master branch github_doc_root = "https://github.com/ArchiveBox/docs/tree/master/" # docs repo uses master branch
github_view_style = 'blob' github_view_style = "blob"
language = 'en' language = "en"
# The full version, including alpha/beta/rc tags # The full version, including alpha/beta/rc tags
release = (Path(__file__).parent.parent / 'pyproject.toml').read_text().split('version = ', 1)[-1].split('\n', 1)[0].strip('"').strip("'") release = (Path(__file__).parent.parent / "pyproject.toml").read_text().split("version = ", 1)[-1].split("\n", 1)[0].strip('"').strip("'")
tag = release tag = release
# 0.8.5 -> v0.8.5 # 0.8.5 -> v0.8.5
if release[0].isdigit(): if release[0].isdigit():
tag = f"v{release}" # .split('rc')[0] tag = f"v{release}" # .split('rc')[0]
# Detect if this is a dev/pre-release build using PEP 440 parsing. # Detect if this is a dev/pre-release build using PEP 440 parsing.
# A version like "0.9.10" with no suffix is stable. # A version like "0.9.10" with no suffix is stable.
# A version like "0.9.10rc1", "0.9.10.dev1", "0.9.10a1" is pre-release. # A version like "0.9.10rc1", "0.9.10.dev1", "0.9.10a1" is pre-release.
# When you release 0.9.0 final (no suffix), it auto-becomes stable. # When you release 0.9.0 final (no suffix), it auto-becomes stable.
import os
try: try:
from packaging.version import Version from packaging.version import Version
is_dev = Version(release).is_prerelease or Version(release).is_devrelease is_dev = Version(release).is_prerelease or Version(release).is_devrelease
except Exception: except Exception:
# fallback if packaging is not installed # fallback if packaging is not installed
is_dev = any(label in release for label in ('dev', 'rc', 'alpha', 'beta')) is_dev = any(label in release for label in ("dev", "rc", "alpha", "beta"))
# RTD "latest" always builds from the default branch = dev docs # RTD "latest" always builds from the default branch = dev docs
rtd_version = os.environ.get('READTHEDOCS_VERSION', '') rtd_version = os.environ.get("READTHEDOCS_VERSION", "")
if rtd_version in ('latest', 'dev', 'main', 'master'): if rtd_version in ("latest", "dev", "main", "master"):
is_dev = True is_dev = True
# -- General configuration --------------------------------------------------- # -- General configuration ---------------------------------------------------
@ -57,15 +60,15 @@ if rtd_version in ('latest', 'dev', 'main', 'master'):
# extensions coming with Sphinx (named 'sphinx.ext.*') or your custom # extensions coming with Sphinx (named 'sphinx.ext.*') or your custom
# ones. # ones.
extensions = [ extensions = [
'sphinx.ext.autodoc', "sphinx.ext.autodoc",
'sphinx.ext.napoleon', "sphinx.ext.napoleon",
'sphinx.ext.linkcode', "sphinx.ext.linkcode",
'sphinx.ext.autosummary', "sphinx.ext.autosummary",
# 'sphinx.ext.graphviz', # 'sphinx.ext.graphviz',
# 'sphinx.ext.inheritance_diagram' # 'sphinx.ext.inheritance_diagram'
'myst_parser', # pip install myst-parser "myst_parser", # pip install myst-parser
'autodoc2', "autodoc2",
'sphinxcontrib.mermaid', # pip install sphinxcontrib-mermaid "sphinxcontrib.mermaid", # pip install sphinxcontrib-mermaid
# 'recommonmark', # 'recommonmark',
] ]
autodoc2_packages = [ autodoc2_packages = [
@ -73,46 +76,46 @@ autodoc2_packages = [
"path": "../archivebox", "path": "../archivebox",
"module": "archivebox", "module": "archivebox",
"exclude_dirs": [ "exclude_dirs": [
'__pycache__', "__pycache__",
'migrations', "migrations",
'vendor', "vendor",
'typings', "typings",
'templates', "templates",
'static', "static",
'tests', "tests",
'tmp', "tmp",
], ],
"exclude_files": [ "exclude_files": [
'tests.py', "tests.py",
'conftest.py', "conftest.py",
'fixtures.py', "fixtures.py",
], ],
}, },
] ]
autodoc2_output_dir = 'apidocs' autodoc2_output_dir = "apidocs"
autodoc2_render_plugin = "myst" autodoc2_render_plugin = "myst"
autodoc2_skip_module_regexes = [ autodoc2_skip_module_regexes = [
r'.*migrations.*', r".*migrations.*",
r'.*vendor.*', r".*vendor.*",
r'.*\.tests($|\..*)', r".*\.tests($|\..*)",
r'.*\.conftest$', r".*\.conftest$",
r'.*\.fixtures$', r".*\.fixtures$",
] ]
# autodoc2_hidden_objects = ['inherited', 'dunder'] # autodoc2_hidden_objects = ['inherited', 'dunder']
autodoc2_hidden_regexes = [ autodoc2_hidden_regexes = [
r'.*__package__', r".*__package__",
] ]
myst_enable_extensions = ['linkify'] # pip install linkify-it-py myst_enable_extensions = ["linkify"] # pip install linkify-it-py
myst_fence_as_directive = ['mermaid'] # render ```mermaid blocks via sphinxcontrib-mermaid myst_fence_as_directive = ["mermaid"] # render ```mermaid blocks via sphinxcontrib-mermaid
source_suffix = { source_suffix = {
'.rst': 'restructuredtext', ".rst": "restructuredtext",
'.txt': 'markdown', ".txt": "markdown",
'.md': 'markdown', ".md": "markdown",
} }
master_doc = 'index' master_doc = "index"
napoleon_google_docstring = True napoleon_google_docstring = True
napoleon_use_param = True napoleon_use_param = True
napoleon_use_ivar = False napoleon_use_ivar = False
@ -120,30 +123,30 @@ napoleon_use_rtype = True
napoleon_include_special_with_doc = False napoleon_include_special_with_doc = False
# Add any paths that contain templates here, relative to this directory. # Add any paths that contain templates here, relative to this directory.
templates_path = ['_templates'] templates_path = ["_templates"]
# List of patterns, relative to source directory, that match files and # List of patterns, relative to source directory, that match files and
# directories to ignore when looking for source files. # directories to ignore when looking for source files.
# This pattern also affects html_static_path and html_extra_path. # This pattern also affects html_static_path and html_extra_path.
exclude_patterns = [ exclude_patterns = [
'_build', "_build",
'**/Thumbs.db', "**/Thumbs.db",
'**/.DS_Store', "**/.DS_Store",
'data*', "data*",
'requirements.txt', "requirements.txt",
'**/requirements.txt', "**/requirements.txt",
'**/tests/**', "**/tests/**",
'**/templates/**', "**/templates/**",
'**/migrations/**', "**/migrations/**",
'_Sidebar.md', "_Sidebar.md",
'_Footer.md', "_Footer.md",
] ]
suppress_warnings = [ suppress_warnings = [
'myst.header', # non-consecutive header levels (common in wiki markdown) "myst.header", # non-consecutive header levels (common in wiki markdown)
'myst.xref_missing', # cross-reference targets from wiki-style links "myst.xref_missing", # cross-reference targets from wiki-style links
'myst.xref_ambiguous', # ambiguous cross-references across modules "myst.xref_ambiguous", # ambiguous cross-references across modules
'autodoc2.dup_item', # duplicate items from Django model inheritance "autodoc2.dup_item", # duplicate items from Django model inheritance
] ]
@ -152,15 +155,15 @@ suppress_warnings = [
# The theme to use for HTML and HTML Help pages. See the documentation for # The theme to use for HTML and HTML Help pages. See the documentation for
# a list of builtin themes. # a list of builtin themes.
# #
html_logo = 'logo.png' html_logo = "logo.png"
html_theme = 'sphinx_rtd_theme' html_theme = "sphinx_rtd_theme"
html_theme_options = { html_theme_options = {
'navigation_depth': 5, "navigation_depth": 5,
'collapse_navigation': False, "collapse_navigation": False,
'sticky_navigation': True, "sticky_navigation": True,
'version_selector': True, "version_selector": True,
'language_selector': False, "language_selector": False,
'style_external_links': True, "style_external_links": True,
} }
html_context = { html_context = {
"display_github": True, "display_github": True,
@ -176,58 +179,67 @@ html_context = {
html_show_sphinx = False html_show_sphinx = False
# Display the version prominently so users know which docs they're reading # Display the version prominently so users know which docs they're reading
version = release # short X.Y version shown in sidebar version = release # short X.Y version shown in sidebar
# release is already set above # full version with alpha/beta/rc tags # release is already set above # full version with alpha/beta/rc tags
texinfo_documents = [ texinfo_documents = [
(master_doc, 'archivebox', 'archivebox Documentation', author, 'archivebox', 'The open-source self-hosted internet archive.', 'Miscellaneous'), (
master_doc,
"archivebox",
"archivebox Documentation",
author,
"archivebox",
"The open-source self-hosted internet archive.",
"Miscellaneous",
),
] ]
autodoc_default_flags = ['members'] autodoc_default_flags = ["members"]
autodoc_member_order = 'bysource' autodoc_member_order = "bysource"
autosummary_generate = True autosummary_generate = True
pygments_style = 'sphinx' pygments_style = "sphinx"
# Add any paths that contain custom static files (such as style sheets) here, # Add any paths that contain custom static files (such as style sheets) here,
# relative to this directory. They are copied after the builtin static files, # relative to this directory. They are copied after the builtin static files,
# so a file named "default.css" will overwrite the builtin "default.css". # so a file named "default.css" will overwrite the builtin "default.css".
html_static_path = ['_static'] html_static_path = ["_static"]
man_pages = [ man_pages = [
(master_doc, 'archivebox', 'archivebox Documentation', [author], 1) (master_doc, "archivebox", "archivebox Documentation", [author], 1),
] ]
def linkcode_resolve(domain, info): def linkcode_resolve(domain, info):
""" """
Calculate link to source code on Github Calculate link to source code on Github
Docs: https://www.sphinx-doc.org/en/master/usage/extensions/linkcode.html Docs: https://www.sphinx-doc.org/en/master/usage/extensions/linkcode.html
""" """
module_name = str(info['module'] or '') module_name = str(info["module"] or "")
package_name = module_name.split('.', 1)[0] # archivebox package_name = module_name.split(".", 1)[0] # archivebox
submodule_name = module_name.split(f'{package_name}', 1)[-1].strip('.') # core.models submodule_name = module_name.split(f"{package_name}", 1)[-1].strip(".") # core.models
symbol_name = str(info['fullname'] or '') # Crawl.abid_ts_src symbol_name = str(info["fullname"] or "") # Crawl.abid_ts_src
full_name = f'{package_name}.{submodule_name}.{symbol_name}'.replace('..', '.') # archivebox.core.models.Crawl.abid_ts_src full_name = f"{package_name}.{submodule_name}.{symbol_name}".replace("..", ".") # archivebox.core.models.Crawl.abid_ts_src
fallback_url = f'https://github.com/search?type=code&q=repo%3AArchiveBox%2FArchiveBox%20{full_name.replace(".", "%20")}' fallback_url = f"https://github.com/search?type=code&q=repo%3AArchiveBox%2FArchiveBox%20{full_name.replace('.', '%20')}"
# 'archivebox.core.models.Crawl' -> archivebox/core/models.py # 'archivebox.core.models.Crawl' -> archivebox/core/models.py
file_path = f'{package_name}/{submodule_name.replace(".", "/")}.py' # archivebox/core/models.py file_path = f"{package_name}/{submodule_name.replace('.', '/')}.py" # archivebox/core/models.py
# correct for any extra / or .py # correct for any extra / or .py
file_path = file_path.strip('/').strip('.py') + '.py' file_path = file_path.strip("/").strip(".py") + ".py"
# fallback to using Github search instead if URL doesn't look like a valid file path # fallback to using Github search instead if URL doesn't look like a valid file path
if not file_path.startswith('archivebox/'): if not file_path.startswith("archivebox/"):
return fallback_url return fallback_url
if '//' in file_path: if "//" in file_path:
return fallback_url return fallback_url
if file_path.count('.py') > 1: if file_path.count(".py") > 1:
return fallback_url return fallback_url
# correct for archivebox/cli.py -> archivebox/cli/__init__.py # correct for archivebox/cli.py -> archivebox/cli/__init__.py
init_path = f'{package_name}/{submodule_name.replace(".", "/")}/__init__.py' init_path = f"{package_name}/{submodule_name.replace('.', '/')}/__init__.py"
if not Path(f'../{file_path}').is_file(): if not Path(f"../{file_path}").is_file():
if Path(f'../{init_path}').is_file(): if Path(f"../{init_path}").is_file():
file_path = init_path file_path = init_path
else: else:
return fallback_url return fallback_url

View File

@ -1,6 +1,6 @@
{ {
"name": "archivebox", "name": "archivebox",
"version": "0.9.33rc56", "version": "0.9.33rc58",
"repository": "github:ArchiveBox/ArchiveBox", "repository": "github:ArchiveBox/ArchiveBox",
"license": "MIT", "license": "MIT",
"dependencies": { "dependencies": {

View File

@ -1,6 +1,6 @@
[project] [project]
name = "archivebox" name = "archivebox"
version = "0.9.33rc56" version = "0.9.33rc58"
requires-python = ">=3.13" requires-python = ">=3.13"
description = "Self-hosted internet archiving solution." description = "Self-hosted internet archiving solution."
authors = [{name = "Nick Sweeting", email = "pyproject.toml@archivebox.io"}] authors = [{name = "Nick Sweeting", email = "pyproject.toml@archivebox.io"}]
@ -79,9 +79,9 @@ dependencies = [
### Extractor dependencies (optional binaries detected at runtime via shutil.which) ### Extractor dependencies (optional binaries detected at runtime via shutil.which)
### Binary/Package Management ### Binary/Package Management
"abxbus==2.5.8", # EventBus API "abxbus==2.5.8", # EventBus API
"abxpkg>=1.11.81", # for: detecting, versioning, and installing binaries via apt/brew/pip/npm "abxpkg>=1.11.83", # for: detecting, versioning, and installing binaries via apt/brew/pip/npm
"abx-plugins>=1.11.90", # shared ArchiveBox plugin package with Chrome/Puppeteer dependency wiring "abx-plugins>=1.11.92", # shared ArchiveBox plugin package with Chrome/Puppeteer dependency wiring
"abx-dl>=1.11.90", # shared ArchiveBox downloader package with blocking install preflight "abx-dl>=1.11.92", # shared ArchiveBox downloader package with blocking install preflight
### UUID7 backport for Python <3.14 ### UUID7 backport for Python <3.14
"uuid7>=0.1.0; python_version < '3.14'", # provides the uuid_extensions module on Python 3.13 "uuid7>=0.1.0; python_version < '3.14'", # provides the uuid_extensions module on Python 3.13
] ]