Merge remote-tracking branch 'origin/dev' into codex/all-tests-ci

This commit is contained in:
Nick Sweeting 2026-07-25 11:51:21 -07:00
commit d4ff0d033f
No known key found for this signature in database
37 changed files with 1314 additions and 188 deletions

View File

@ -120,6 +120,14 @@ jobs:
fi
uv sync --locked --dev "${extra_args[@]}"
- name: Install PostgreSQL server binaries
if: contains(matrix.test.path, 'test_postgres_backend')
run: |
set -Eeuo pipefail
if ! ls /usr/lib/postgresql/*/bin/initdb >/dev/null 2>&1 && ! command -v initdb >/dev/null 2>&1; then
sudo apt-get update && sudo apt-get install -y postgresql
fi
- name: Run ${{ matrix.test.name }}
env:
TEST_PATH: ${{ matrix.test.path }}

View File

@ -12,19 +12,9 @@ import signal_webhooks.fields
import signal_webhooks.utils
class Migration(migrations.Migration):
initial = True
dependencies = [
("auth", "0012_alter_user_first_name_max_length"),
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
]
operations = [
migrations.SeparateDatabaseAndState(
database_operations=[
migrations.RunSQL(
sql="""
# Raw sqlite DDL kept EXACTLY as-is; executed only on sqlite via Django's
# RunSQL (identical statement splitting → byte-for-byte identical behavior).
INITIAL_SQL = """
-- Create api_apitoken table
CREATE TABLE IF NOT EXISTS api_apitoken (
id TEXT PRIMARY KEY NOT NULL,
@ -70,12 +60,60 @@ class Migration(migrations.Migration):
CREATE INDEX IF NOT EXISTS api_outboundwebhook_created_at_idx ON api_outboundwebhook(created_at);
CREATE INDEX IF NOT EXISTS api_outboundwebhook_name_idx ON api_outboundwebhook(name);
CREATE INDEX IF NOT EXISTS api_outboundwebhook_ref_idx ON api_outboundwebhook(ref);
""",
reverse_sql="""
"""
INITIAL_REVERSE_SQL = """
DROP TABLE IF EXISTS api_outboundwebhook;
DROP TABLE IF EXISTS api_apitoken;
""",
),
"""
def _run_sqlite_only_sql(apps, schema_editor):
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_forwards(
"api",
schema_editor,
None,
None,
)
def _run_sqlite_only_sql_reverse(apps, schema_editor):
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_backwards(
"api",
schema_editor,
None,
None,
)
def _pg_sync_schema(apps, schema_editor):
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(apps, schema_editor, "api", ["APIToken", "OutboundWebhook"])
def _pg_drop_schema(apps, schema_editor):
from archivebox.misc.db import drop_models_on_postgres
drop_models_on_postgres(apps, schema_editor, "api", ["OutboundWebhook", "APIToken"])
class Migration(migrations.Migration):
initial = True
dependencies = [
("auth", "0012_alter_user_first_name_max_length"),
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
]
operations = [
migrations.SeparateDatabaseAndState(
database_operations=[
migrations.RunPython(_run_sqlite_only_sql, reverse_code=_run_sqlite_only_sql_reverse),
],
state_operations=[
migrations.CreateModel(
@ -236,4 +274,8 @@ class Migration(migrations.Migration):
),
],
),
# On non-sqlite backends the raw DDL above is skipped, so the real
# schema diverges from Django state. Rebuild both tables (including the
# unique constraint) from the post-migration state (no-op on sqlite).
migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
]

View File

@ -32,13 +32,13 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
from archivebox.config import CONSTANTS, VERSION
from archivebox.config.common import get_config
from archivebox.config.collection import write_config_file
from archivebox.misc.db import apply_migrations
from archivebox.misc.db import apply_migrations, database_exists, ensure_database_ready
from archivebox.misc.checks import check_migrations
config = get_config()
is_empty = not len(set(os.listdir(CONSTANTS.DATA_DIR)) - CONSTANTS.ALLOWED_IN_DATA_DIR)
existing_index = os.path.isfile(CONSTANTS.DATABASE_FILE)
existing_index = database_exists()
if is_empty and not existing_index:
print(f"[turquoise4][+] Initializing a new ArchiveBox v{VERSION} collection...[/turquoise4]")
print("[green]----------------------------------------------------------------------[/green]")
@ -86,11 +86,15 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
# create the ArchiveBox.conf file
write_config_file({"SECRET_KEY": config.SECRET_KEY})
if os.access(CONSTANTS.DATABASE_FILE, os.F_OK):
if existing_index:
print("\n[green][*] Verifying main SQL index and running any migrations needed...[/green]")
else:
print("\n[green][+] Building main SQL index and running initial migrations...[/green]")
# For postgres, make sure the server is reachable and create the database
# if it doesn't exist yet (sqlite creates its file automatically).
ensure_database_ready()
from archivebox.config.django import setup_django
setup_django()
@ -107,9 +111,14 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
else:
os.environ["ARCHIVEBOX_WANTS_INIT"] = previous_wants_init
assert os.path.isfile(CONSTANTS.DATABASE_FILE) and os.access(CONSTANTS.DATABASE_FILE, os.R_OK)
from archivebox.misc.db import database_display_location, is_postgres
assert database_exists()
print()
print(f"{_display_data_path(CONSTANTS.DATABASE_FILE, CONSTANTS.DATA_DIR)}")
if is_postgres():
print(f"{database_display_location()}")
else:
print(f"{_display_data_path(CONSTANTS.DATABASE_FILE, CONSTANTS.DATA_DIR)}")
print()
print("[dodger_blue3][*] Checking links from indexes and archive folders (safe to Ctrl+C)...[/dodger_blue3]")

View File

@ -44,7 +44,10 @@ def status(out_dir: Path = CONSTANTS.DATA_DIR) -> None:
legacy_snapshot_dirs = [
entry for entry in archive_dir.iterdir() if entry.is_dir() and not entry.is_symlink() and Snapshot.is_legacy_archive_dir(entry)
]
print(f" > SQL Main Index: {num_sql_links} links".ljust(36), f"(found in {CONSTANTS.SQL_INDEX_FILENAME})")
from archivebox.misc.db import database_display_location, is_postgres
index_location = database_display_location() if is_postgres() else CONSTANTS.SQL_INDEX_FILENAME
print(f" > SQL Main Index: {num_sql_links} links".ljust(36), f"(found in {index_location})")
print(f" > JSON Link Details: {len(legacy_snapshot_dirs)} links".ljust(36), f"(found in {archive_dir.name}/*/index.json)")
print()
print("[green]\\[*] Scanning archive data directories...[/green]")

View File

@ -350,7 +350,16 @@ class DatabaseConfig(BaseConfigSet):
toml_section_header: str = "DATABASE_CONFIG"
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
DATABASE_ENGINE: str = Field(
default="sqlite",
alias="ARCHIVEBOX_DATABASE_ENGINE",
pattern=r"(?i)^(sqlite|postgres|postgresql)$",
)
DATABASE_NAME: str = Field(default=str(CONSTANTS.DATABASE_FILE), alias="ARCHIVEBOX_DATABASE_NAME")
DATABASE_HOST: str = Field(default="127.0.0.1", alias="ARCHIVEBOX_DATABASE_HOST")
DATABASE_PORT: int = Field(default=5432, alias="ARCHIVEBOX_DATABASE_PORT", ge=1, le=65535)
DATABASE_USER: str = Field(default="archivebox", alias="ARCHIVEBOX_DATABASE_USER")
DATABASE_PASSWORD: str = Field(default="", alias="ARCHIVEBOX_DATABASE_PASSWORD")
SQLITE_JOURNAL_MODE: str = Field(
default="WAL",
alias="ARCHIVEBOX_SQLITE_JOURNAL_MODE",

View File

@ -152,9 +152,10 @@ def setup_django(check_db=False, in_memory_db=False) -> None:
for conn in connections.all():
conn.close_if_unusable_or_obsolete()
sql_index_path = CONSTANTS.DATABASE_FILE
assert os.access(sql_index_path, os.F_OK), (
f"No database file {sql_index_path} found in: {CONSTANTS.DATA_DIR} (Are you in an ArchiveBox collection directory?)"
from archivebox.misc.db import database_display_location, database_exists
assert database_exists(), (
f"No database {database_display_location()} found for: {CONSTANTS.DATA_DIR} (Are you in an ArchiveBox collection directory?)"
)
except KeyboardInterrupt:

View File

@ -58,9 +58,11 @@ def _get_collection_id(DATA_DIR=DATA_DIR, force_create=False) -> str:
collection_id = hashlib.sha256(f"{machine_id}:{collection_path}@{creation_date}".encode()).hexdigest()[:8]
try:
# only persist collection_id file if we already have an index.sqlite3 file present
# only persist collection_id file if this dir already looks like a real collection
# (has an index.sqlite3, or an ArchiveBox.conf when the DB lives in postgres),
# otherwise we might be running in a directory that is not a collection, no point creating cruft files
collection_is_active = os.path.isfile(DATABASE_FILE) and os.path.isdir(ARCHIVE_DIR) and os.access(DATA_DIR, os.W_OK)
collection_marker = os.path.isfile(DATABASE_FILE) or os.path.isfile(DATA_DIR / "ArchiveBox.conf")
collection_is_active = collection_marker and os.path.isdir(ARCHIVE_DIR) and os.access(DATA_DIR, os.W_OK)
if collection_is_active or force_create:
collection_id_file.write_text(collection_id)
@ -270,6 +272,24 @@ def get_or_create_working_lib_dir(autofix=True, quiet=False, config: "ArchiveBox
raise OSError(f"ArchiveBox is unable to find a writable ABXPKG_LIB_DIR, tried {CANDIDATES}!")
def _sql_index_location() -> dict:
from archivebox.misc.db import database_display_location, database_exists, is_postgres
if is_postgres():
return {
"path": database_display_location(),
"enabled": True,
"is_valid": database_exists(),
"is_mount": False,
}
return {
"path": DATABASE_FILE.resolve(),
"enabled": True,
"is_valid": os.path.isfile(DATABASE_FILE) and os.access(DATABASE_FILE, os.R_OK) and os.access(DATABASE_FILE, os.W_OK),
"is_mount": os.path.ismount(DATABASE_FILE.resolve()),
}
def get_data_locations(config: "ArchiveBoxConfig | None" = None, **config_kwargs):
from archivebox.config.constants import CONSTANTS
from archivebox.config.common import get_config
@ -296,12 +316,7 @@ def get_data_locations(config: "ArchiveBoxConfig | None" = None, **config_kwargs
and os.access(CONSTANTS.CONFIG_FILE, os.R_OK)
and os.access(CONSTANTS.CONFIG_FILE, os.W_OK),
},
"SQL_INDEX": {
"path": DATABASE_FILE.resolve(),
"enabled": True,
"is_valid": os.path.isfile(DATABASE_FILE) and os.access(DATABASE_FILE, os.R_OK) and os.access(DATABASE_FILE, os.W_OK),
"is_mount": os.path.ismount(DATABASE_FILE.resolve()),
},
"SQL_INDEX": _sql_index_location(),
"ARCHIVE_DIR": {
"path": CONSTANTS.ARCHIVE_DIR.resolve(),
"enabled": True,

View File

@ -122,8 +122,6 @@ class ArchiveBoxAdmin(admin.AdminSite):
def index(self, request: "HttpRequest", extra_context: dict[str, Any] | None = None) -> "TemplateResponse":
response = super().index(request, extra_context)
if connection.vendor != "sqlite":
return response
models_by_table: dict[str, list[dict[str, Any]]] = {}
for app in response.context_data.get("app_list", []):
@ -136,23 +134,18 @@ class ArchiveBoxAdmin(admin.AdminSite):
if not models_by_table:
return response
try:
with connection.cursor() as cursor:
cursor.execute("SELECT tbl, stat FROM sqlite_stat1")
for table, stat in cursor.fetchall():
try:
count = int(str(stat).split()[0])
except (IndexError, TypeError, ValueError):
continue
self._set_model_object_count(
models_by_table,
table,
count,
title=f"Approximate count from SQLite stats: {count:,}",
)
models_by_table.pop(table, None)
except DatabaseError:
pass
from archivebox.misc.db import approximate_row_counts
for table, count in approximate_row_counts(connection).items():
if table not in models_by_table:
continue
self._set_model_object_count(
models_by_table,
table,
count,
title=f"Approximate count from database stats: {count:,}",
)
models_by_table.pop(table, None)
for table in list(models_by_table):
try:

View File

@ -20,6 +20,14 @@ class CoreConfig(AppConfig):
ModelWithOutputDir.register_delete_signal()
# SQLite ignores VARCHAR(n) limits but PostgreSQL enforces them; clamp
# CharField values on save so writes behave the same on both backends.
from django.db.models.signals import pre_save
from archivebox.misc.db import truncate_overlong_charfields
pre_save.connect(truncate_overlong_charfields, dispatch_uid="archivebox_truncate_overlong_charfields")
# Import models to register state machines with the registry
# Skip during makemigrations to avoid premature state machine access
if "makemigrations" not in sys.argv:

View File

@ -47,6 +47,14 @@ def normalize_status(status):
def upgrade_core_tables(apps, schema_editor):
"""Upgrade core tables from v0.7.2 or v0.8.6rc0 to v0.9.0."""
# sqlite-only legacy repair/rebuild. On postgres this raw SQL (PRAGMA,
# sqlite_master, INSERT OR IGNORE, DATETIME, table-rebuild dance) is both
# invalid and unnecessary: a postgres database can never contain legacy
# data at this point. The final _pg_sync_schema op resyncs the real
# postgres schema to migration state instead.
if schema_editor.connection.vendor != "sqlite":
return
from archivebox.uuid_compat import uuid7
cursor = connection.cursor()
@ -491,6 +499,15 @@ def upgrade_core_tables(apps, schema_editor):
print(" ✓ Core table rebuild complete")
def _pg_sync_schema(apps, schema_editor):
# On postgres, upgrade_core_tables is a no-op above, so the raw sqlite
# rebuilds of core_tag/core_snapshot/core_archiveresult never ran. Resync
# the real schema to this migration's end-state (tables are empty on pg).
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(apps, schema_editor, "core", ["Tag", "Snapshot", "SnapshotTag", "ArchiveResult"])
class Migration(migrations.Migration):
dependencies = [
("core", "0022_auto_20231023_2008"),
@ -571,4 +588,5 @@ class Migration(migrations.Migration):
),
],
),
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
]

View File

@ -4,11 +4,102 @@
from django.db import migrations, models
# Raw sqlite table-rebuild that makes core_snapshot.crawl_id NOT NULL. Kept
# byte-for-byte and replayed through Django's own RunSQL only on sqlite.
_MAKE_CRAWL_ID_NOT_NULL_SQL = """
-- Rebuild snapshot table with NOT NULL crawl_id
CREATE TABLE core_snapshot_final (
id TEXT PRIMARY KEY NOT NULL,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
modified_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
url TEXT NOT NULL,
timestamp VARCHAR(32) NOT NULL UNIQUE,
bookmarked_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
crawl_id TEXT NOT NULL,
parent_snapshot_id TEXT,
title VARCHAR(512),
downloaded_at DATETIME,
depth INTEGER NOT NULL DEFAULT 0,
fs_version VARCHAR(10) NOT NULL DEFAULT '0.9.0',
config TEXT NOT NULL DEFAULT '{}',
notes TEXT NOT NULL DEFAULT '',
num_uses_succeeded INTEGER NOT NULL DEFAULT 0,
num_uses_failed INTEGER NOT NULL DEFAULT 0,
status VARCHAR(15) NOT NULL DEFAULT 'queued',
retry_at DATETIME,
current_step INTEGER NOT NULL DEFAULT 0,
FOREIGN KEY (crawl_id) REFERENCES crawls_crawl(id) ON DELETE CASCADE,
FOREIGN KEY (parent_snapshot_id) REFERENCES core_snapshot(id) ON DELETE SET NULL
);
INSERT INTO core_snapshot_final (
id, url, timestamp, title,
bookmarked_at, created_at, modified_at,
crawl_id, parent_snapshot_id,
downloaded_at, depth, fs_version,
config, notes,
num_uses_succeeded, num_uses_failed,
status, retry_at, current_step
)
SELECT
id, url, timestamp, title,
bookmarked_at, created_at, modified_at,
REPLACE(crawl_id, '-', ''), REPLACE(parent_snapshot_id, '-', ''),
downloaded_at, depth, fs_version,
COALESCE(config, '{}'), COALESCE(notes, ''),
num_uses_succeeded, num_uses_failed,
status, retry_at, current_step
FROM core_snapshot;
DROP TABLE core_snapshot;
ALTER TABLE core_snapshot_final RENAME TO core_snapshot;
CREATE INDEX core_snapshot_url_idx ON core_snapshot(url);
CREATE INDEX core_snapshot_timestamp_idx ON core_snapshot(timestamp);
CREATE INDEX core_snapshot_bookmarked_at_idx ON core_snapshot(bookmarked_at);
CREATE INDEX core_snapshot_crawl_id_idx ON core_snapshot(crawl_id);
CREATE INDEX core_snapshot_status_idx ON core_snapshot(status);
CREATE INDEX core_snapshot_retry_at_idx ON core_snapshot(retry_at);
CREATE INDEX core_snapshot_created_at_idx ON core_snapshot(created_at);
CREATE UNIQUE INDEX core_snapshot_url_crawl_unique ON core_snapshot(url, crawl_id);
"""
def _make_crawl_id_not_null(apps, schema_editor):
# sqlite-only table rebuild. Reuse Django's own RunSQL statement splitting so
# the sqlite behavior is byte-for-byte identical to the original RunSQL op.
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(sql=_MAKE_CRAWL_ID_NOT_NULL_SQL).database_forwards("core", schema_editor, None, None)
def _pg_sync_schema(apps, schema_editor):
# On postgres the raw sqlite rebuilds above are skipped and crawl was only
# added to migration state; resync the real schema to this migration's
# end-state (empty tables). Snapshot is referenced by SnapshotTag and
# ArchiveResult, so those are rebuilt too to restore FK constraints.
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(apps, schema_editor, "core", ["Snapshot", "SnapshotTag", "ArchiveResult"])
def create_default_crawl_and_assign_snapshots(apps, schema_editor):
"""
Create a default crawl for migrated snapshots and assign all snapshots without a crawl to it.
Uses raw SQL because the app registry isn't fully populated during migrations.
"""
# Legacy-data-only, sqlite-specific (PRAGMA + '?' placeholders). On a fresh
# postgres install core_snapshot has no crawl_id column yet and there are no
# unassigned snapshots, so gate before touching any SQL.
if schema_editor.connection.vendor != "sqlite":
return
from django.db import connection
import uuid as uuid_lib
from datetime import datetime
@ -89,72 +180,10 @@ class Migration(migrations.Migration):
),
migrations.SeparateDatabaseAndState(
database_operations=[
# Now make crawl_id NOT NULL
migrations.RunSQL(
sql="""
-- Rebuild snapshot table with NOT NULL crawl_id
CREATE TABLE core_snapshot_final (
id TEXT PRIMARY KEY NOT NULL,
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
modified_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
url TEXT NOT NULL,
timestamp VARCHAR(32) NOT NULL UNIQUE,
bookmarked_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
crawl_id TEXT NOT NULL,
parent_snapshot_id TEXT,
title VARCHAR(512),
downloaded_at DATETIME,
depth INTEGER NOT NULL DEFAULT 0,
fs_version VARCHAR(10) NOT NULL DEFAULT '0.9.0',
config TEXT NOT NULL DEFAULT '{}',
notes TEXT NOT NULL DEFAULT '',
num_uses_succeeded INTEGER NOT NULL DEFAULT 0,
num_uses_failed INTEGER NOT NULL DEFAULT 0,
status VARCHAR(15) NOT NULL DEFAULT 'queued',
retry_at DATETIME,
current_step INTEGER NOT NULL DEFAULT 0,
FOREIGN KEY (crawl_id) REFERENCES crawls_crawl(id) ON DELETE CASCADE,
FOREIGN KEY (parent_snapshot_id) REFERENCES core_snapshot(id) ON DELETE SET NULL
);
INSERT INTO core_snapshot_final (
id, url, timestamp, title,
bookmarked_at, created_at, modified_at,
crawl_id, parent_snapshot_id,
downloaded_at, depth, fs_version,
config, notes,
num_uses_succeeded, num_uses_failed,
status, retry_at, current_step
)
SELECT
id, url, timestamp, title,
bookmarked_at, created_at, modified_at,
REPLACE(crawl_id, '-', ''), REPLACE(parent_snapshot_id, '-', ''),
downloaded_at, depth, fs_version,
COALESCE(config, '{}'), COALESCE(notes, ''),
num_uses_succeeded, num_uses_failed,
status, retry_at, current_step
FROM core_snapshot;
DROP TABLE core_snapshot;
ALTER TABLE core_snapshot_final RENAME TO core_snapshot;
CREATE INDEX core_snapshot_url_idx ON core_snapshot(url);
CREATE INDEX core_snapshot_timestamp_idx ON core_snapshot(timestamp);
CREATE INDEX core_snapshot_bookmarked_at_idx ON core_snapshot(bookmarked_at);
CREATE INDEX core_snapshot_crawl_id_idx ON core_snapshot(crawl_id);
CREATE INDEX core_snapshot_status_idx ON core_snapshot(status);
CREATE INDEX core_snapshot_retry_at_idx ON core_snapshot(retry_at);
CREATE INDEX core_snapshot_created_at_idx ON core_snapshot(created_at);
CREATE UNIQUE INDEX core_snapshot_url_crawl_unique ON core_snapshot(url, crawl_id);
""",
reverse_sql=migrations.RunSQL.noop,
# Now make crawl_id NOT NULL (sqlite-only table rebuild)
migrations.RunPython(
_make_crawl_id_not_null,
reverse_code=migrations.RunPython.noop,
),
],
state_operations=[
@ -169,4 +198,5 @@ class Migration(migrations.Migration):
),
],
),
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
]

View File

@ -11,6 +11,12 @@ from archivebox.uuid_compat import uuid7
def copy_old_fields_to_new(apps, schema_editor):
"""Copy data from old field names to new field names after AddField operations."""
# sqlite-only legacy data copy (PRAGMA introspection). On postgres there is
# no legacy data and the state-only AddFields below are resynced by
# _pg_sync_schema at the end of this migration.
if schema_editor.connection.vendor != "sqlite":
return
cursor = connection.cursor()
# Check if old fields still exist
@ -43,6 +49,19 @@ def copy_old_fields_to_new(apps, schema_editor):
# NOTE: UUIDs are already populated by migration 0023 for all migration paths.
def _pg_sync_schema(apps, schema_editor):
# This migration mixes real ORM AddFields with state-only AddFields (the
# snapshot config/current_step/depth/notes/num_uses_*/parent_snapshot/
# retry_at/status columns and archiveresult created_at/modified_at were
# created on sqlite by 0023's raw rebuild, so here they are state-only).
# On postgres those columns would never land, desyncing the real schema.
# Resync to this migration's final state (empty tables on pg). Snapshot is
# referenced by SnapshotTag and ArchiveResult, so rebuild those too.
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(apps, schema_editor, "core", ["Snapshot", "SnapshotTag", "ArchiveResult"])
class Migration(migrations.Migration):
dependencies = [
("core", "0024_assign_default_crawl"),
@ -335,4 +354,5 @@ class Migration(migrations.Migration):
model_name="snapshot",
constraint=models.UniqueConstraint(fields=("timestamp",), name="unique_timestamp"),
),
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
]

View File

@ -275,6 +275,14 @@ def copy_archiveresult_data_to_process(apps, schema_editor):
- failed exited (exit_code=1)
- skipped exited (exit_code=None)
"""
# sqlite-only legacy data copy (PRAGMA introspection, '?' placeholders).
# A postgres install can never contain legacy ArchiveResult cmd/pwd data at
# this point, so there is nothing to copy into machine_process. The
# RemoveField ops below are real (non-state-only) and drop the now-empty
# cmd/pwd/cmd_version columns on both vendors, so no resync is needed.
if schema_editor.connection.vendor != "sqlite":
return
cursor = connection.cursor()
# Check if old fields still exist (skip if fresh install or already migrated)

View File

@ -26,6 +26,12 @@ def migrate_archiveresult_id_to_uuid(apps, schema_editor):
Result: Clean schema with ONLY id as UUIDField (no old_id, no uuid)
"""
# sqlite-only table rebuild (sqlite_master, PRAGMA, table copy/rename). On
# postgres this is unnecessary and invalid: the table is empty and the
# state flip of id -> UUIDField is applied by _pg_sync_schema below.
if schema_editor.connection.vendor != "sqlite":
return
cursor = connection.cursor()
# Check if table exists and has data
@ -190,6 +196,16 @@ def migrate_archiveresult_id_to_uuid(apps, schema_editor):
print(f" ✓ ArchiveResult UUID primary key migration complete ({row_count} records)")
def _pg_sync_schema(apps, schema_editor):
# On postgres the sqlite table rebuild above is skipped; the id -> UUIDField
# flip and uuid-field removal only reach migration state. Resync the real
# (empty) core_archiveresult table to this migration's end-state. Nothing in
# core references ArchiveResult, so no other models need rebuilding.
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(apps, schema_editor, "core", ["ArchiveResult"])
class Migration(migrations.Migration):
dependencies = [
("core", "0028_alter_snapshot_fs_version"),
@ -217,4 +233,5 @@ class Migration(migrations.Migration):
),
],
),
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
]

View File

@ -13,6 +13,13 @@ def _repair_snapshot_permissions(apps, schema_editor):
no underlying column. Fresh installs added the column via 0041 and
this is a no-op.
"""
# sqlite-only legacy repair (PRAGMA table_xinfo + sqlite VIRTUAL generated
# column syntax). On postgres the permissions column was created by 0041's
# portable GeneratedField AddField (STORED generated column), so there is
# nothing to repair.
if schema_editor.connection.vendor != "sqlite":
return
cursor = schema_editor.connection.cursor()
# ``table_xinfo`` lists STORED/VIRTUAL generated columns; ``table_info``
# silently drops them, so a prior 0041 that landed the STORED column

View File

@ -0,0 +1,30 @@
# Postgres-only: btree pattern-ops index on core_snapshot.url so LIKE 'prefix%'
# queries (url__startswith and the URL prefix search) stay index scans under any
# database collation. SQLite needs nothing here: its plain url index already
# serves the bytewise range comparisons used on that backend.
from django.db import migrations
def add_pg_url_pattern_index(apps, schema_editor):
if schema_editor.connection.vendor != "postgresql":
return
schema_editor.execute(
"CREATE INDEX IF NOT EXISTS core_snapshot_url_pattern_ops_idx ON core_snapshot (url text_pattern_ops)",
)
def remove_pg_url_pattern_index(apps, schema_editor):
if schema_editor.connection.vendor != "postgresql":
return
schema_editor.execute("DROP INDEX IF EXISTS core_snapshot_url_pattern_ops_idx")
class Migration(migrations.Migration):
dependencies = [
("core", "0050_snapshot_permissions_not_null"),
]
operations = [
migrations.RunPython(add_pg_url_pattern_index, remove_pg_url_pattern_index),
]

View File

@ -183,9 +183,14 @@ class SnapshotQuerySet(models.QuerySet):
for crawl_id, permissions in Crawl.objects.filter(pk__in=missing_crawl_ids).values_list("pk", "permissions")
}
from archivebox.misc.db import truncate_overlong_charfields
for obj in objs:
if isinstance(obj, self.model):
obj.ensure_permissions_config(crawl_permissions=crawl_permissions_by_id.get(str(obj.crawl_id)))
# bulk_create bypasses pre_save, so clamp CharFields here too
# (e.g. page titles) to stay within postgres VARCHAR limits.
truncate_overlong_charfields(obj)
return super().bulk_create(objs, *args, **kwargs)
def paged_iterator(self, chunk_size: int = 500):

View File

@ -14,6 +14,10 @@ import archivebox
from archivebox.config.constants import CONSTANTS
from archivebox.config.common import get_config
from archivebox.core.routes_util import get_api_base_url, get_admin_base_url, get_base_url, normalize_base_url
# DATABASE_ENGINE config selects the backend (sqlite by default); the
# sqlite-vs-postgres helpers live in archivebox.misc.db.
from archivebox.misc.db import is_postgres, postgres_db_params
from .settings_logging import SETTINGS_LOGGING
@ -245,12 +249,21 @@ SQLITE_CONNECTION_OPTIONS = {
},
}
DATABASES = {
"default": {
"NAME": DATABASE_NAME,
**SQLITE_CONNECTION_OPTIONS,
},
}
if is_postgres():
DATABASES = {
"default": {
"ENGINE": "django.db.backends.postgresql",
**postgres_db_params(),
"OPTIONS": {"connect_timeout": 10},
},
}
else:
DATABASES = {
"default": {
"NAME": DATABASE_NAME,
**SQLITE_CONNECTION_OPTIONS,
},
}
MIGRATION_MODULES = {"signal_webhooks": None}
# Django requires DEFAULT_AUTO_FIELD to subclass AutoField (BigAutoField, SmallAutoField, etc.)

View File

@ -276,12 +276,20 @@ class SnapshotView(View):
"""
def _fragmentless_url_query(url: str) -> Q:
# Use a range comparison (url >= 'canonical#' AND url < 'canonical#\U0010ffff')
# instead of LIKE/__startswith — SQLite's case-insensitive LIKE bypasses the
# url index and forces a full-table scan over ~1M rows (~250ms). The range
# form lets SQLite use a MULTI-INDEX OR and stays under 1ms.
from archivebox.misc.db import is_postgres
canonical = without_fragment(url)
return Q(url=canonical) | (Q(url__gte=f"{canonical}#") & Q(url__lt=f"{canonical}#\U0010ffff"))
if not is_postgres():
# Use a range comparison (url >= 'canonical#' AND url < 'canonical#\U0010ffff')
# instead of LIKE/__startswith — SQLite's case-insensitive LIKE bypasses the
# url index and forces a full-table scan over ~1M rows (~250ms). The range
# form lets SQLite use a MULTI-INDEX OR and stays under 1ms.
return Q(url=canonical) | (Q(url__gte=f"{canonical}#") & Q(url__lt=f"{canonical}#\U0010ffff"))
# On postgres the range trick is unsafe: linguistic (ICU/libc) collations
# don't compare '#'-suffixed strings bytewise, so the range can miss rows.
# startswith compiles to LIKE 'prefix%' with wildcards escaped, which is
# correct under any collation and uses the url pattern-ops index.
return Q(url=canonical) | Q(url__startswith=f"{canonical}#")
normalized = without_fragment(path)
if path.startswith(("http://", "https://")):

View File

@ -10,19 +10,9 @@ from archivebox.uuid_compat import uuid7
from archivebox.base_models.models import get_or_create_system_user_pk
class Migration(migrations.Migration):
initial = True
dependencies = [
("auth", "0012_alter_user_first_name_max_length"),
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
]
operations = [
migrations.SeparateDatabaseAndState(
database_operations=[
migrations.RunSQL(
sql="""
# Raw sqlite DDL kept EXACTLY as-is; executed only on sqlite via Django's
# RunSQL (identical statement splitting → byte-for-byte identical behavior).
INITIAL_SQL = """
-- Create crawls_crawlschedule table first (circular FK will be added later)
CREATE TABLE IF NOT EXISTS crawls_crawlschedule (
id TEXT PRIMARY KEY NOT NULL,
@ -75,12 +65,60 @@ class Migration(migrations.Migration):
CREATE INDEX IF NOT EXISTS crawls_crawl_created_at_idx ON crawls_crawl(created_at);
CREATE INDEX IF NOT EXISTS crawls_crawl_created_by_id_idx ON crawls_crawl(created_by_id);
CREATE INDEX IF NOT EXISTS crawls_crawl_schedule_id_idx ON crawls_crawl(schedule_id);
""",
reverse_sql="""
"""
INITIAL_REVERSE_SQL = """
DROP TABLE IF EXISTS crawls_crawl;
DROP TABLE IF EXISTS crawls_crawlschedule;
""",
),
"""
def _run_sqlite_only_sql(apps, schema_editor):
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_forwards(
"crawls",
schema_editor,
None,
None,
)
def _run_sqlite_only_sql_reverse(apps, schema_editor):
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_backwards(
"crawls",
schema_editor,
None,
None,
)
def _pg_sync_schema(apps, schema_editor):
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(apps, schema_editor, "crawls", ["CrawlSchedule", "Crawl"])
def _pg_drop_schema(apps, schema_editor):
from archivebox.misc.db import drop_models_on_postgres
drop_models_on_postgres(apps, schema_editor, "crawls", ["Crawl", "CrawlSchedule"])
class Migration(migrations.Migration):
initial = True
dependencies = [
("auth", "0012_alter_user_first_name_max_length"),
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
]
operations = [
migrations.SeparateDatabaseAndState(
database_operations=[
migrations.RunPython(_run_sqlite_only_sql, reverse_code=_run_sqlite_only_sql_reverse),
],
state_operations=[
migrations.CreateModel(
@ -174,4 +212,9 @@ class Migration(migrations.Migration):
),
],
),
# On non-sqlite backends the raw DDL above is skipped, so the real
# schema diverges from Django state. Rebuild both tables from the
# post-migration state (no-op on sqlite). Runs as a top-level op so
# ``apps`` reflects the full state including the circular template FK.
migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
]

View File

@ -6,6 +6,10 @@ from django.db import migrations, connection
def upgrade_crawl_table_from_v086(apps, schema_editor):
"""Upgrade crawls_crawl table from v0.8.6rc0 schema to v0.9.0 schema."""
# sqlite-only legacy repair (sqlite_master/PRAGMA/GLOB). Postgres support
# postdates all legacy data, so this never has work to do there.
if schema_editor.connection.vendor != "sqlite":
return
cursor = connection.cursor()
# Check if crawls_crawl table exists

View File

@ -97,6 +97,12 @@ def _ensure_permissions_column(cursor):
def hydrate_crawl_permissions(apps, schema_editor):
# sqlite-only legacy hydration + generated-column repair (PRAGMA
# table_xinfo / VIRTUAL generated column). On postgres the ``permissions``
# column is created portably by 0013 (AddField GeneratedField) and there is
# never legacy data to hydrate, so this is a no-op there.
if schema_editor.connection.vendor != "sqlite":
return
Crawl = apps.get_model("crawls", "Crawl")
base_config = raw_base_config(apps)
default_permissions = resolve_permissions(base_config, "public")

View File

@ -7,16 +7,45 @@ import django.utils.timezone
from archivebox.uuid_compat import uuid7
class Migration(migrations.Migration):
initial = True
def _run_sqlite_initial_ddl(apps, schema_editor):
# sqlite-only: raw DDL creates the real tables byte-for-byte as historically
# shipped. On other vendors the tables are built from migration state by
# _pg_sync_schema below (a fresh non-sqlite DB never holds legacy data here).
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(
sql=_SQLITE_INITIAL_SQL,
reverse_sql=_SQLITE_INITIAL_REVERSE_SQL,
).database_forwards("machine", schema_editor, None, None)
dependencies = []
operations = [
migrations.SeparateDatabaseAndState(
database_operations=[
migrations.RunSQL(
sql="""
def _reverse_sqlite_initial_ddl(apps, schema_editor):
if schema_editor.connection.vendor != "sqlite":
return
migrations.RunSQL(
sql=_SQLITE_INITIAL_SQL,
reverse_sql=_SQLITE_INITIAL_REVERSE_SQL,
).database_backwards("machine", schema_editor, None, None)
def _pg_sync_schema(apps, schema_editor):
from archivebox.misc.db import rebuild_models_from_migration_state
rebuild_models_from_migration_state(
apps,
schema_editor,
"machine",
["Machine", "NetworkInterface", "Binary"],
)
def _pg_drop_schema(apps, schema_editor):
from archivebox.misc.db import drop_models_on_postgres
drop_models_on_postgres(apps, schema_editor, "machine", ["Binary", "NetworkInterface", "Machine"])
_SQLITE_INITIAL_SQL = """
-- Create machine_machine table
CREATE TABLE IF NOT EXISTS machine_machine (
id TEXT PRIMARY KEY NOT NULL,
@ -98,13 +127,25 @@ class Migration(migrations.Migration):
CREATE INDEX IF NOT EXISTS machine_binary_status_idx ON machine_binary(status);
CREATE INDEX IF NOT EXISTS machine_binary_retry_at_idx ON machine_binary(retry_at);
""",
reverse_sql="""
"""
_SQLITE_INITIAL_REVERSE_SQL = """
DROP TABLE IF EXISTS machine_binary;
DROP TABLE IF EXISTS machine_networkinterface;
DROP TABLE IF EXISTS machine_machine;
""",
),
"""
class Migration(migrations.Migration):
initial = True
dependencies = []
operations = [
migrations.SeparateDatabaseAndState(
database_operations=[
migrations.RunPython(_run_sqlite_initial_ddl, _reverse_sqlite_initial_ddl),
],
state_operations=[
migrations.CreateModel(
@ -244,4 +285,5 @@ class Migration(migrations.Migration):
),
],
),
migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
]

View File

@ -10,6 +10,13 @@ def converge_binary_table(apps, schema_editor):
Drop machine_installedbinary if it exists (0.8.6rc0 path).
Create machine_binary if it doesn't exist (needed by Process model).
"""
# sqlite-only legacy convergence: on a fresh install machine_binary already
# exists with the final column set (created by 0001), so every branch below
# is a no-op schema-wise. A non-sqlite DB never carries the 0.8.x legacy
# shapes this repairs, and its Binary table already matches migration state.
if schema_editor.connection.vendor != "sqlite":
return
cursor = connection.cursor()
# Check what tables exist

View File

@ -2,9 +2,12 @@ from django.db import migrations
def remove_output_dir_if_exists(apps, schema_editor):
cursor = schema_editor.connection.cursor()
cursor.execute("PRAGMA table_info(machine_binary)")
columns = {row[1] for row in cursor.fetchall()}
# On non-sqlite the RemoveField below is state-only, so the real
# ALTER TABLE DROP COLUMN must run here too (a fresh DB has machine_binary
# with output_dir from 0001). Portable introspection works on both backends.
connection = schema_editor.connection
with connection.cursor() as cursor:
columns = {col.name for col in connection.introspection.get_table_description(cursor, "machine_binary")}
if "output_dir" not in columns:
return

View File

@ -2,6 +2,12 @@ from django.db import migrations
def add_machine_config_if_missing(apps, schema_editor):
# sqlite-only: uses PRAGMA introspection. On a fresh non-sqlite install the
# config column already exists (created from 0001 migration state), so there
# is nothing to add.
if schema_editor.connection.vendor != "sqlite":
return
cursor = schema_editor.connection.cursor()
cursor.execute("PRAGMA table_info(machine_machine)")
columns = {row[1] for row in cursor.fetchall()}

View File

@ -12,6 +12,7 @@ from pathlib import Path
from typing import TextIO
from typing import Any
import fcntl
import os
import time
from collections.abc import Callable
from contextlib import contextmanager
@ -21,6 +22,213 @@ from archivebox.config import CONSTANTS
from archivebox.misc.util import enforce_types
# ============================================================================
# Database backend adapter (sqlite / postgresql)
# ============================================================================
# All sqlite-vs-postgres branching in ArchiveBox is centralized in this
# section. Code elsewhere should call these helpers instead of checking
# ``connection.vendor``, building ``DATABASES`` entries, or touching
# ``CONSTANTS.DATABASE_FILE`` directly.
def is_postgres() -> bool:
"""True if DATABASE_ENGINE selects postgres (sqlite is the default)."""
from archivebox.config.common import get_config
return (get_config().DATABASE_ENGINE or "sqlite").strip().lower().startswith("postgres")
def postgres_db_params() -> dict[str, str]:
"""Postgres connection params from config (NAME/USER/PASSWORD/HOST/PORT)."""
from archivebox.config.common import get_config
config = get_config()
name = config.DATABASE_NAME
# DATABASE_NAME defaults to the sqlite file path; that default makes no
# sense as a postgres database name, so fall back to 'archivebox'.
if name == str(CONSTANTS.DATABASE_FILE) or name.endswith(".sqlite3"):
name = "archivebox"
return {
"NAME": name,
"USER": config.DATABASE_USER,
"PASSWORD": config.DATABASE_PASSWORD,
"HOST": config.DATABASE_HOST,
"PORT": str(config.DATABASE_PORT),
}
def _psycopg_connect(dbname: str | None = None, connect_timeout: int = 5):
import psycopg
params = postgres_db_params()
return psycopg.connect(
dbname=dbname or params["NAME"],
user=params["USER"],
password=params["PASSWORD"] or None,
host=params["HOST"],
port=params["PORT"],
connect_timeout=connect_timeout,
)
def database_exists() -> bool:
"""True if this collection's database has been initialized.
sqlite: the index.sqlite3 file exists on disk.
postgres: the configured database is reachable and contains the
django_migrations table. Safe to call before Django is set up.
"""
if not is_postgres():
return os.path.isfile(CONSTANTS.DATABASE_FILE)
try:
with _psycopg_connect() as conn:
row = conn.execute("SELECT to_regclass('django_migrations')").fetchone()
return bool(row and row[0])
except Exception:
return False
def database_display_location() -> str:
"""Human-readable location of the database (file path or postgres DSN)."""
if not is_postgres():
return str(CONSTANTS.DATABASE_FILE)
params = postgres_db_params()
return f"postgresql://{params['USER']}@{params['HOST']}:{params['PORT']}/{params['NAME']}"
def ensure_database_ready() -> None:
"""Make sure a database server is reachable before running migrations.
sqlite: no-op (the file is created on first connection).
postgres: verify the server accepts connections and create the configured
database if it does not exist yet. Raises SystemExit with a helpful
message if the server is unreachable.
"""
if not is_postgres():
return
import psycopg
from rich import print as rich_print
params = postgres_db_params()
try:
with _psycopg_connect():
return
except psycopg.OperationalError as err:
# 3D000 invalid_catalog_name: server is up but the database is missing
if getattr(err, "sqlstate", None) != "3D000" and "does not exist" not in str(err):
rich_print(f"[red][X] Error: Unable to connect to PostgreSQL at {database_display_location()}[/red]")
rich_print(f" {err}")
rich_print(" [violet]Hint:[/violet] Check ARCHIVEBOX_DATABASE_HOST/PORT/USER/PASSWORD and that the server is running.")
raise SystemExit(4) from err
with _psycopg_connect(dbname="postgres") as conn:
conn.autocommit = True
safe_name = params["NAME"].replace('"', '""')
conn.execute(f'CREATE DATABASE "{safe_name}"')
rich_print(f" + Created PostgreSQL database {params['NAME']}")
def approximate_row_counts(connection) -> dict[str, int]:
"""Cheap per-table approximate row counts from the backend's optimizer stats.
sqlite: reads sqlite_stat1 (populated by ANALYZE).
postgres: reads pg_class.reltuples (maintained by autovacuum/ANALYZE).
Returns {} on any failure; tables never analyzed may be absent.
"""
counts: dict[str, int] = {}
try:
with connection.cursor() as cursor:
if connection.vendor == "sqlite":
cursor.execute("SELECT tbl, stat FROM sqlite_stat1")
for table, stat in cursor.fetchall():
try:
counts[str(table)] = int(str(stat).split()[0])
except (IndexError, TypeError, ValueError):
continue
elif connection.vendor == "postgresql":
cursor.execute(
"""
SELECT c.relname, c.reltuples::bigint
FROM pg_class c
JOIN pg_namespace n ON n.oid = c.relnamespace
WHERE c.relkind = 'r'
AND n.nspname = current_schema()
AND c.reltuples >= 0
""",
)
counts = {str(table): int(estimate) for table, estimate in cursor.fetchall()}
except Exception:
return {}
return counts
def truncate_overlong_charfields(instance=None, **kwargs) -> None:
"""Clamp a model instance's CharField values to their declared max_length.
SQLite never enforces VARCHAR(n) limits, so ArchiveBox has always stored
overlong values (e.g. long crawl labels or page titles) untruncated.
PostgreSQL enforces them and would raise DataError on save instead.
Truncating keeps writes succeeding identically on both backends.
Dual-use: works as a ``pre_save`` receiver (Django passes ``instance=`` and
``sender=`` as kwargs; registered in ``CoreConfig.ready()``) and as a plain
``truncate_overlong_charfields(obj)`` call for ``bulk_create`` paths, which
bypass signals.
"""
from django.db import models as dj_models
if instance is None:
return
for field in instance._meta.local_concrete_fields:
if isinstance(field, dj_models.CharField) and field.max_length:
value = getattr(instance, field.attname, None)
if isinstance(value, str) and len(value) > field.max_length:
setattr(instance, field.attname, value[: field.max_length])
# --- migration helpers ------------------------------------------------------
def rebuild_models_from_migration_state(apps, schema_editor, app_label: str, model_names: list[str]) -> None:
"""(non-sqlite only) Drop and recreate the given models' tables from the
current migration state.
ArchiveBox's historical sqlite migrations rebuild tables with raw SQL that
intentionally diverges from Django migration state (state-only AddFields
reconciled by later sqlite rebuilds). Postgres support postdates all of
them, so a non-sqlite database can never contain legacy data at these
points in history: every affected table is empty, and dropping + recreating
it from state is always equivalent, keeping the real schema in lockstep
with migration state at each divergence point. No-op on sqlite.
"""
if schema_editor.connection.vendor == "sqlite":
return
existing_tables = set(schema_editor.connection.introspection.table_names())
models = [apps.get_model(app_label, model_name) for model_name in model_names]
for model in models:
if model._meta.db_table in existing_tables:
schema_editor.delete_model(model)
for model in models:
schema_editor.create_model(model)
def drop_models_on_postgres(apps, schema_editor, app_label: str, model_names: list[str]) -> None:
"""Reverse companion to ``rebuild_models_from_migration_state``.
Drops the given models' tables on non-sqlite backends (in the order given,
so callers pass reverse-dependency order). No-op on sqlite, whose reverse is
handled by the gated ``RunSQL`` reverse_sql instead.
"""
if schema_editor.connection.vendor == "sqlite":
return
existing_tables = set(schema_editor.connection.introspection.table_names())
for model_name in model_names:
model = apps.get_model(app_label, model_name)
if model._meta.db_table in existing_tables:
schema_editor.delete_model(model)
def run_db_analyze_batch(
remaining: list[str] | None,
*,
@ -357,7 +565,8 @@ def migration_state(out_dir: Path = CONSTANTS.DATA_DIR) -> tuple[list[str], list
try:
cursor.execute("SELECT app, name FROM django_migrations")
except Exception as err:
if "no such table" in str(err).lower():
msg = str(err).lower()
if "no such table" in msg or ("relation" in msg and "does not exist" in msg):
return set()
raise
return {(str(app), str(name)) for app, name in cursor.fetchall()}

View File

@ -410,7 +410,8 @@ def printable_folder_status(name: str, folder: dict) -> str:
else:
color, symbol, note, num_files = "grey53", "-", "unused", "-"
if folder["path"]:
if folder["path"] and "://" not in str(folder["path"]):
# file-count probing only makes sense for filesystem paths, not DSNs
if os.access(folder["path"], os.R_OK):
try:
num_files = (

View File

@ -92,6 +92,12 @@ def _ensure_permissions_column(cursor):
def hydrate_persona_permissions(apps, schema_editor):
# sqlite-only legacy hydration + generated-column repair (PRAGMA
# table_xinfo / VIRTUAL generated column). On postgres the ``permissions``
# column is created portably by 0003 (AddField GeneratedField) and there is
# never legacy data to hydrate, so this is a no-op there.
if schema_editor.connection.vendor != "sqlite":
return
Persona = apps.get_model("personas", "Persona")
base_config = raw_base_config(apps)
default_permissions = resolve_permissions(base_config, "public")

View File

@ -16,31 +16,48 @@ MAX_SEARCH_RANK_IDS = 500
def escape_like_query(query: str) -> str:
"""Escape a string for SQLite LIKE matching."""
"""Escape a string for SQL LIKE matching (used with ESCAPE '\\')."""
return query.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
def crawl_config_values_search_wave(query: str) -> Q | None:
"""Build a Snapshot Q predicate matching values inside Crawl.config."""
if connection.vendor != "sqlite":
return None
from archivebox.crawls.models import Crawl
pattern = f"%{escape_like_query(query).lower()}%"
matching_crawls = Crawl.objects.extra(
where=[
"""
EXISTS (
SELECT 1
FROM json_tree(config)
WHERE json_tree.atom IS NOT NULL
AND LOWER(CAST(json_tree.atom AS TEXT)) LIKE %s ESCAPE '\\'
)
""",
],
params=[pattern],
)
if connection.vendor == "sqlite":
matching_crawls = Crawl.objects.extra(
where=[
"""
EXISTS (
SELECT 1
FROM json_tree(config)
WHERE json_tree.atom IS NOT NULL
AND LOWER(CAST(json_tree.atom AS TEXT)) LIKE %s ESCAPE '\\'
)
""",
],
params=[pattern],
)
elif connection.vendor == "postgresql":
# Match only scalar config *values*, not keys (mirrors SQLite's
# json_tree.atom). jsonb_path_query('$.**') walks every nested node;
# keep the non-container leaves and compare their text form.
matching_crawls = Crawl.objects.extra(
where=[
"""
EXISTS (
SELECT 1
FROM jsonb_path_query(config, '$.**') AS leaf
WHERE jsonb_typeof(leaf) NOT IN ('object', 'array')
AND LOWER(leaf #>> '{}') LIKE %s ESCAPE '\\'
)
""",
],
params=[pattern],
)
else:
return None
return Q(crawl_id__in=matching_crawls.values("pk"))

View File

@ -124,19 +124,32 @@ def iter_url_prefix_search_ids(prefix: str, queryset):
table = connection.ops.quote_name(model._meta.db_table)
pk_column = connection.ops.quote_name(model._meta.pk.column)
url_column = connection.ops.quote_name(model._meta.get_field("url").column)
upper_bound = url_prefix_upper_bound(prefix)
raw_ids = []
if connection.vendor == "sqlite":
# Bytewise range comparison uses the plain url btree index directly.
where_clause = f"{url_column} >= %s AND {url_column} < %s"
where_params = [prefix, url_prefix_upper_bound(prefix)]
else:
# Range comparisons are collation-dependent on postgres (linguistic
# collations don't compare bytewise), so use LIKE with escaped
# wildcards instead — correct under any collation and able to use the
# url pattern-ops index.
from archivebox.search.query import escape_like_query
where_clause = f"{url_column} LIKE %s ESCAPE '\\'"
where_params = [f"{escape_like_query(prefix)}%"]
with connection.cursor() as cursor:
cursor.execute(
f"""
SELECT {pk_column}
FROM {table}
WHERE {url_column} >= %s AND {url_column} < %s
WHERE {where_clause}
ORDER BY {url_column}
LIMIT %s
""",
[prefix, upper_bound, URL_PREFIX_SEARCH_LIMIT],
[*where_params, URL_PREFIX_SEARCH_LIMIT],
)
raw_ids = [str(row[0]).replace("-", "") for row in cursor.fetchall()]

View File

@ -0,0 +1,228 @@
#!/usr/bin/env python3
"""
End-to-end tests for the PostgreSQL database backend (DATABASE_ENGINE=postgres).
Spins up a real throwaway PostgreSQL cluster (initdb + pg_ctl) for the module,
then exercises real archivebox CLI flows against it: init, status, re-init,
add --index-only, list, remove, and a full schema-vs-models parity check.
Requires PostgreSQL server binaries (initdb/pg_ctl) to be installed.
"""
import os
import shlex
import shutil
import socket
import subprocess
import tempfile
import uuid
from pathlib import Path
import pytest
from .conftest import cli_env, run_archivebox_cmd, run_queued_crawls
def _find_pg_bindir() -> Path:
"""Locate PostgreSQL server binaries (initdb) on this machine."""
initdb_on_path = shutil.which("initdb")
if initdb_on_path:
return Path(initdb_on_path).resolve().parent
candidates = []
for base in (Path("/usr/lib/postgresql"), Path("/opt/homebrew/opt"), Path("/usr/local/opt"), Path("/opt/homebrew/Cellar/postgresql")):
if base.is_dir():
for sub in sorted(base.iterdir(), reverse=True):
initdb = sub / "bin" / "initdb"
if initdb.is_file():
candidates.append(initdb.parent)
assert candidates, "PostgreSQL server binaries (initdb) are required for test_postgres_backend tests"
return candidates[0]
def _free_port() -> int:
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
sock.bind(("127.0.0.1", 0))
return sock.getsockname()[1]
PG_TEST_USER = "abxtest"
@pytest.fixture(scope="module")
def pg_server():
"""A real throwaway PostgreSQL server for the whole test module."""
bindir = _find_pg_bindir()
workdir = Path(tempfile.mkdtemp(prefix="abx-pg-test-"))
datadir = workdir / "data"
logfile = workdir / "server.log"
port = _free_port()
# postgres refuses to run as root; when the suite runs as root (e.g. in a
# dev container), run the server as the system postgres user instead.
run_as_postgres_user = os.geteuid() == 0
if run_as_postgres_user:
workdir.chmod(0o755)
shutil.chown(workdir, user="postgres")
def pg_cmd(args: list[str]) -> subprocess.CompletedProcess:
if run_as_postgres_user:
full = ["su", "postgres", "-s", "/bin/sh", "-c", shlex.join(args)]
else:
full = args
return subprocess.run(full, capture_output=True, text=True, check=False, timeout=120)
result = pg_cmd([str(bindir / "initdb"), "-D", str(datadir), "-E", "UTF8", "-A", "trust", "-U", PG_TEST_USER])
assert result.returncode == 0, f"initdb failed: {result.stderr}"
server_opts = f"-p {port} -k {workdir} -c listen_addresses=127.0.0.1 -c fsync=off -c synchronous_commit=off"
result = pg_cmd([str(bindir / "pg_ctl"), "-D", str(datadir), "-l", str(logfile), "-o", server_opts, "-w", "start"])
assert result.returncode == 0, f"pg_ctl start failed: {result.stderr}\n{logfile.read_text() if logfile.exists() else ''}"
try:
yield {"host": "127.0.0.1", "port": port, "user": PG_TEST_USER}
finally:
pg_cmd([str(bindir / "pg_ctl"), "-D", str(datadir), "-m", "immediate", "stop"])
shutil.rmtree(workdir, ignore_errors=True)
def pg_cli_env(pg_server: dict, dbname: str, **extra) -> dict:
env = cli_env(**extra)
env.update(
{
"ARCHIVEBOX_DATABASE_ENGINE": "postgres",
"ARCHIVEBOX_DATABASE_HOST": pg_server["host"],
"ARCHIVEBOX_DATABASE_PORT": str(pg_server["port"]),
"ARCHIVEBOX_DATABASE_USER": pg_server["user"],
"ARCHIVEBOX_DATABASE_NAME": dbname,
},
)
return env
def unique_dbname() -> str:
return f"abx_test_{uuid.uuid4().hex[:12]}"
def pg_query(pg_server: dict, dbname: str, query: str) -> list[tuple]:
import psycopg
with psycopg.connect(host=pg_server["host"], port=pg_server["port"], user=pg_server["user"], dbname=dbname) as conn:
return conn.execute(query).fetchall()
def test_init_creates_postgres_schema(pg_server, tmp_path):
"""Fresh init against postgres should create the database + full schema, and no sqlite file."""
dbname = unique_dbname()
env = pg_cli_env(pg_server, dbname)
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
assert result.returncode == 0, f"Init failed: {result.stderr}\n{result.stdout}"
assert not (tmp_path / "index.sqlite3").exists(), "sqlite file should not be created when using postgres"
assert (tmp_path / "archive").is_dir(), "Archive dir not created"
(applied_migrations,) = pg_query(pg_server, dbname, "SELECT COUNT(*) FROM django_migrations")[0]
assert applied_migrations > 50, f"Expected all migrations applied, got {applied_migrations}"
for table in ("core_snapshot", "core_archiveresult", "core_tag", "crawls_crawl", "machine_machine", "api_apitoken", "personas_persona"):
(regclass,) = pg_query(pg_server, dbname, f"SELECT to_regclass('{table}')")[0]
assert regclass == table, f"Table {table} missing from postgres schema"
def test_postgres_schema_matches_models(pg_server, tmp_path):
"""Every model column must exist in postgres and vice versa (no state/schema drift)."""
dbname = unique_dbname()
env = pg_cli_env(pg_server, dbname)
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
assert result.returncode == 0, f"Init failed: {result.stderr}"
parity_script = (
"from django.apps import apps\n"
"from django.db import connection\n"
"problems = []\n"
"with connection.cursor() as cursor:\n"
" tables = set(connection.introspection.table_names(cursor))\n"
" for model in apps.get_models(include_auto_created=True):\n"
" meta = model._meta\n"
" if not meta.managed or meta.proxy:\n"
" continue\n"
" if meta.db_table not in tables:\n"
" problems.append(f'missing table {meta.db_table}')\n"
" continue\n"
" db_cols = {col.name for col in connection.introspection.get_table_description(cursor, meta.db_table)}\n"
" model_cols = {field.column for field in meta.local_concrete_fields}\n"
" for col in sorted(model_cols - db_cols):\n"
" problems.append(f'{meta.db_table}: missing column {col}')\n"
" for col in sorted(db_cols - model_cols):\n"
" problems.append(f'{meta.db_table}: extra column {col}')\n"
"print('SCHEMA_PROBLEMS=' + repr(sorted(problems)))\n"
)
result = run_archivebox_cmd(["manage", "shell", "-c", parity_script], cwd=tmp_path, env=env, timeout=120)
assert result.returncode == 0, f"manage shell failed: {result.stderr}"
assert "SCHEMA_PROBLEMS=[]" in result.stdout, f"Postgres schema diverges from models:\n{result.stdout}\n{result.stderr}"
result = run_archivebox_cmd(["manage", "makemigrations", "--check", "--dry-run"], cwd=tmp_path, env=env, timeout=120)
assert result.returncode == 0, f"Model state does not match migrations: {result.stdout}\n{result.stderr}"
def test_status_and_reinit_on_postgres(pg_server, tmp_path):
"""status works against postgres, and a second init takes the 'verify existing' path."""
dbname = unique_dbname()
env = pg_cli_env(pg_server, dbname)
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
assert result.returncode == 0, f"Init failed: {result.stderr}"
assert "Initializing a new ArchiveBox" in result.stdout
result = run_archivebox_cmd(["status"], cwd=tmp_path, env=env, timeout=120)
assert result.returncode == 0, f"Status failed: {result.stderr}"
assert f"postgresql://{PG_TEST_USER}@" in result.stdout, f"status should show the postgres DSN:\n{result.stdout}"
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
assert result.returncode == 0, f"Re-init failed: {result.stderr}"
assert "Verifying and updating existing ArchiveBox collection" in result.stdout
def test_add_list_remove_on_postgres(pg_server, tmp_path):
"""Real add/list/remove CLI flows store and retrieve rows from postgres."""
dbname = unique_dbname()
env = pg_cli_env(pg_server, dbname, disable_extractors=True)
test_url = "https://example.com/abx-postgres-test"
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
assert result.returncode == 0, f"Init failed: {result.stderr}"
result = run_archivebox_cmd(["add", "--index-only", test_url], cwd=tmp_path, env=env, timeout=300)
assert result.returncode == 0, f"Add failed: {result.stderr}\n{result.stdout}"
run_queued_crawls(tmp_path, env)
(snapshot_count,) = pg_query(pg_server, dbname, "SELECT COUNT(*) FROM core_snapshot")[0]
assert snapshot_count >= 1, "Snapshot row not written to postgres"
(crawl_count,) = pg_query(pg_server, dbname, "SELECT COUNT(*) FROM crawls_crawl")[0]
assert crawl_count >= 1, "Crawl row not written to postgres"
result = run_archivebox_cmd(["list"], cwd=tmp_path, env=env, timeout=120)
assert result.returncode == 0, f"List failed: {result.stderr}"
assert "example.com/abx-postgres-test" in result.stdout, f"Added URL missing from list output:\n{result.stdout}"
result = run_archivebox_cmd(["search", "abx-postgres-test"], cwd=tmp_path, env=env, timeout=120)
assert result.returncode == 0, f"Search failed: {result.stderr}"
assert "example.com/abx-postgres-test" in result.stdout, f"Added URL missing from search output:\n{result.stdout}"
result = run_archivebox_cmd(
["remove", "--yes", "--delete", "--filter-type=exact", test_url],
cwd=tmp_path,
env=env,
timeout=120,
)
assert result.returncode == 0, f"Remove failed: {result.stderr}\n{result.stdout}"
(snapshot_count,) = pg_query(pg_server, dbname, f"SELECT COUNT(*) FROM core_snapshot WHERE url = '{test_url}'")[0]
assert snapshot_count == 0, "Snapshot row should be deleted from postgres"
def test_sqlite_remains_the_default(tmp_path):
"""Without DATABASE_ENGINE config, init keeps using the sqlite file backend."""
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=cli_env(), timeout=300)
assert result.returncode == 0, f"Init failed: {result.stderr}"
assert (tmp_path / "index.sqlite3").exists(), "sqlite file should be created by default"

202
bin/benchmark_db_backends.py Executable file
View File

@ -0,0 +1,202 @@
#!/usr/bin/env python3
"""Benchmark ArchiveBox hot-path index queries at large row counts.
Run from inside an initialized (empty) collection directory, with the same
ARCHIVEBOX_DATABASE_* env vars the collection was initialized with:
cd /path/to/collection
uv run --project /path/to/ArchiveBox python /path/to/ArchiveBox/bin/benchmark_db_backends.py --rows 1000000
Seeds N snapshots (+1 archiveresult each) via bulk_create, runs ANALYZE, then
times the hot queries used by the admin UI, snapshot detail views, URL prefix
search, and the worker queue/claim paths. Works on both sqlite and postgres
use it to compare backends or to catch performance regressions.
"""
import argparse
import json
import os
import statistics
import sys
import time
from datetime import datetime, timedelta, timezone
def seed(rows: int, batch_size: int = 20_000) -> None:
from django.db import transaction
from archivebox.base_models.models import get_or_create_system_user_pk
from archivebox.core.models import ArchiveResult, Snapshot
from archivebox.crawls.models import Crawl
from archivebox.uuid_compat import uuid7
user_pk = get_or_create_system_user_pk()
crawl = Crawl.objects.create(urls="https://example.com/", created_by_id=user_pk, status="sealed", label="benchmark seed")
existing = Snapshot.objects.count()
if existing >= rows:
print(f"already seeded ({existing} snapshots)", flush=True)
return
base_time = datetime(2020, 1, 1, tzinfo=timezone.utc)
started = time.monotonic()
for start in range(existing, rows, batch_size):
end = min(start + batch_size, rows)
snapshots = []
for i in range(start, end):
created = base_time + timedelta(seconds=i)
snapshots.append(
Snapshot(
id=uuid7(),
url=f"https://site{i % 1000}.example.org/path/{i}" + ("#section" if i % 7 == 0 else ""),
timestamp=f"{1577836800 + i}.{i}",
title=f"Benchmark page {i}",
crawl=crawl,
bookmarked_at=created,
created_at=created,
modified_at=created,
downloaded_at=created if i % 10 else None,
status="sealed" if i % 20 else "queued",
retry_at=None if i % 20 else created,
fs_version="0.9.0",
config={},
depth=0,
),
)
with transaction.atomic():
Snapshot.objects.bulk_create(snapshots, batch_size=batch_size)
ArchiveResult.objects.bulk_create(
[
ArchiveResult(
id=uuid7(),
snapshot=snapshot,
plugin="wget",
hook_name="on_Snapshot__06_wget.py",
status="succeeded",
created_at=snapshot.created_at,
modified_at=snapshot.created_at,
start_ts=snapshot.created_at,
end_ts=snapshot.created_at,
output_str="benchmark",
output_files={},
)
for snapshot in snapshots
],
batch_size=batch_size,
)
if (end // batch_size) % 5 == 0 or end == rows:
rate = (end - existing) / max(time.monotonic() - started, 0.001)
print(f" seeded {end}/{rows} snapshots ({rate:,.0f} rows/s)", flush=True)
def analyze_tables() -> None:
from django.db import connection
with connection.cursor() as cursor:
for table in ("core_snapshot", "core_archiveresult"):
cursor.execute(f"ANALYZE {table}")
def timed(func, repeat: int = 5) -> tuple[float, object]:
times = []
result = None
for _ in range(repeat):
started = time.perf_counter()
result = func()
times.append((time.perf_counter() - started) * 1000)
return statistics.median(times), result
def run_benchmarks(rows: int) -> dict[str, float]:
from django.db import connection
from django.db.models import Count, Q
from django.utils import timezone as dj_timezone
from archivebox.core.models import ArchiveResult, Snapshot
from archivebox.misc.db import approximate_row_counts
from archivebox.search.views import iter_url_prefix_search_ids
now = dj_timezone.now()
results: dict[str, float] = {}
target_i = rows // 2
target_url = f"https://site{target_i % 1000}.example.org/path/{target_i}"
def fragmentless_q(url: str) -> Q:
if connection.vendor == "sqlite":
return Q(url=url) | (Q(url__gte=f"{url}#") & Q(url__lt=f"{url}#\U0010ffff"))
return Q(url=url) | Q(url__startswith=f"{url}#")
benchmarks = {
"exact_count": lambda: Snapshot.objects.count(),
"approximate_row_counts": lambda: approximate_row_counts(connection),
"admin_list_page": lambda: list(
Snapshot.objects.order_by("-bookmarked_at").values("id", "url", "title", "status", "bookmarked_at")[:40],
),
"admin_list_page_offset_10k": lambda: list(Snapshot.objects.order_by("-bookmarked_at").values("id", "url", "title")[10_000:10_040]),
"snapshot_detail_by_url": lambda: list(Snapshot.objects.filter(fragmentless_q(target_url))[:10]),
"snapshot_detail_archiveresults": lambda: list(
ArchiveResult.objects.filter(snapshot__url=target_url).order_by("start_ts").values("id", "plugin", "status")[:100],
),
"url_prefix_search": lambda: list(iter_url_prefix_search_ids("https://site500.example.org/", Snapshot.objects.all())),
"worker_queue_scan": lambda: list(
Snapshot.objects.filter(status="queued", retry_at__lte=now)
.order_by("retry_at", "created_at")
.values_list("id", flat=True)[:100],
),
"status_facet_counts": lambda: dict(Snapshot.objects.values_list("status").annotate(n=Count("id")).values_list("status", "n")),
"tag_join_filter": lambda: list(Snapshot.objects.filter(title__icontains="page 4242").values("id")[:20]),
}
for name, func in benchmarks.items():
median_ms, _ = timed(func)
results[name] = round(median_ms, 2)
print(f" {name:35s} {median_ms:10.2f} ms", flush=True)
def claim_one() -> int:
snapshot = Snapshot.objects.filter(status="queued").order_by("retry_at").first()
if snapshot is None:
return 0
return Snapshot.objects.filter(pk=snapshot.pk, retry_at=snapshot.retry_at).update(
retry_at=now + timedelta(seconds=60),
modified_at=now,
)
median_ms, _ = timed(claim_one)
results["worker_cas_claim"] = round(median_ms, 2)
print(f" {'worker_cas_claim':35s} {median_ms:10.2f} ms", flush=True)
return results
def main() -> None:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--rows", type=int, default=1_000_000)
parser.add_argument("--seed-only", action="store_true")
parser.add_argument("--json", dest="json_path", default=None, help="also write results to this JSON file")
args = parser.parse_args()
from archivebox.config.django import setup_django
from archivebox.misc.db import database_exists
setup_django()
assert database_exists(), "run archivebox init in this directory first"
from django.db import connection
print(f"backend: {connection.vendor}, target rows: {args.rows}", flush=True)
print("seeding...", flush=True)
seed(args.rows)
if args.seed_only:
return
print("running ANALYZE...", flush=True)
analyze_tables()
print("benchmarks (median of 5):", flush=True)
results = run_benchmarks(args.rows)
results["_backend"] = connection.vendor
results["_rows"] = args.rows
if args.json_path:
with open(args.json_path, "w", encoding="utf-8") as f:
json.dump(results, f, indent=2)
if __name__ == "__main__":
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
main()

View File

@ -645,9 +645,9 @@ Skip the startup check that verifies [`TMP_DIR`](#tmp_dir) can host unix-domain
## Database Settings
*Options for tuning the SQLite index database that backs ArchiveBox's snapshot, tag, and crawl metadata.*
*Options for choosing and tuning the index database that backs ArchiveBox's snapshot, tag, and crawl metadata.*
ArchiveBox stores all of its index metadata in a single SQLite database file (`index.sqlite3` inside your data directory). The defaults are tuned for nearly all users — the knobs below mostly govern **lock-contention behavior**, which matters when multiple workers touch the database concurrently (e.g. supervised orchestrators, parallel `archivebox add` runs, container restarts that race against an in-flight write, or long-running web/admin processes alongside CLI commands).
ArchiveBox stores all of its index metadata in a single SQLite database file (`index.sqlite3` inside your data directory) by default, or optionally in a PostgreSQL database (see [`DATABASE_ENGINE`](#database_engine)). The defaults are tuned for nearly all users — the knobs below mostly govern **lock-contention behavior**, which matters when multiple workers touch the database concurrently (e.g. supervised orchestrators, parallel `archivebox add` runs, container restarts that race against an in-flight write, or long-running web/admin processes alongside CLI commands).
> [!NOTE]
> These are advanced operator tuning options. If you are not actively diagnosing `database is locked` errors or planning a non-default storage layout, you can safely leave everything in this section at its default.
@ -657,9 +657,59 @@ ArchiveBox stores all of its index metadata in a single SQLite database file (`i
- https://www.sqlite.org/wal.html
- https://www.sqlite.org/pragma.html
---
<a id="database_engine"></a>
<a id="archivebox_database_engine"></a>
#### `DATABASE_ENGINE`
**Possible Values:** [`sqlite`]/`postgres`
Which database backend to use for the main index. Settable as `ARCHIVEBOX_DATABASE_ENGINE` or under `[DATABASE_CONFIG]` in `ArchiveBox.conf`.
The default `sqlite` keeps everything in a single `index.sqlite3` file inside the data directory and requires no external services. Set this to `postgres` to store the index in a PostgreSQL database instead — useful for large collections with many concurrent writers, or when the data directory lives on a filesystem where SQLite performs poorly (e.g. network mounts).
With `postgres`, connection details come from the options below. `archivebox init` will create the configured database automatically if the server is reachable and the database does not exist yet. Only the *index* moves to PostgreSQL — snapshot output files stay in `./archive/` in the data directory, and plugin-owned sidecar databases (e.g. the `search.sqlite3` full-text index) are unaffected.
```ini
# example ArchiveBox.conf
[DATABASE_CONFIG]
DATABASE_ENGINE = postgres
DATABASE_NAME = archivebox
DATABASE_HOST = 127.0.0.1
DATABASE_PORT = 5432
DATABASE_USER = archivebox
DATABASE_PASSWORD = s3cret
```
> [!WARNING]
> Pick a backend when you first run `archivebox init` and stick with it. There is no built-in tool (yet) to move an existing collection's index between SQLite and PostgreSQL.
*Related options:*
[`DATABASE_NAME`](#database_name), [`DATABASE_HOST`](#database_host)
---
<a id="database_host"></a>
<a id="archivebox_database_host"></a>
<a id="database_port"></a>
<a id="archivebox_database_port"></a>
<a id="database_user"></a>
<a id="archivebox_database_user"></a>
<a id="database_password"></a>
<a id="archivebox_database_password"></a>
#### `DATABASE_HOST` / `DATABASE_PORT` / `DATABASE_USER` / `DATABASE_PASSWORD`
**Possible Values:** [`127.0.0.1`] / [`5432`] / [`archivebox`] / [empty]
PostgreSQL connection settings, used only when [`DATABASE_ENGINE`](#database_engine)`=postgres`. Settable as `ARCHIVEBOX_DATABASE_HOST`, `ARCHIVEBOX_DATABASE_PORT`, `ARCHIVEBOX_DATABASE_USER`, and `ARCHIVEBOX_DATABASE_PASSWORD`.
`DATABASE_HOST` may also be a path to a directory containing a PostgreSQL unix socket (e.g. `/var/run/postgresql`).
---
<a id="database_name"></a>
<a id="archivebox_database_name"></a>
#### `DATABASE_NAME`
**Possible Values:** [`index.sqlite3`] / `archivebox` / ...
The main index database. Settable as `ARCHIVEBOX_DATABASE_NAME`.
With the default [`DATABASE_ENGINE`](#database_engine)`=sqlite`, this is the path to the SQLite index file inside the data directory (`index.sqlite3`). With `DATABASE_ENGINE=postgres`, it is the name of the PostgreSQL database instead (default: `archivebox`), created automatically by `archivebox init` if it does not exist.
---
#### `SQLITE_JOURNAL_MODE`
**Possible Values:** [`WAL`]/`DELETE`/`TRUNCATE`/`PERSIST`/`MEMORY`/`OFF`
SQLite [journal mode](https://www.sqlite.org/pragma.html#pragma_journal_mode), applied via `PRAGMA journal_mode = ...` on every new connection. Settable as `ARCHIVEBOX_SQLITE_JOURNAL_MODE`.

View File

@ -92,6 +92,7 @@ version = 2
"7890998169008ad0-1" = "illustration"
# docs/Configuration.md
"48b48da014bd91dd-1" = "illustration"
"129eb9ca56f17500-1" = "illustration"
"798793d9d6b2e12e-1" = "run"
"3fecc495abdc20ef-1" = "illustration"

View File

@ -41,6 +41,7 @@ dependencies = [
### Django libraries
"setuptools>=74.1.0", # for: django 5 on python >=3.12, distutils is no longer in stdlib but django 5.1 expects distutils (TODO: check if this can be removed eventually)
"django>=6.0",
"psycopg[binary]>=3.2", # for: PostgreSQL database backend (ARCHIVEBOX_DATABASE_ENGINE=postgres)
"daphne>=4.2.1", # ASGI server for Django (no channels needed - websockets not used)
"django-ninja>=1.5.1",
"django-extensions>=3.2.3",

47
uv.lock
View File

@ -13,11 +13,11 @@ supported-markers = [
]
[options]
exclude-newer = "2026-07-18T21:19:18.486818Z"
exclude-newer = "2026-07-19T10:29:25.327679866Z"
exclude-newer-span = "P5D"
[options.exclude-newer-package]
abxbus = { timestamp = "2026-07-23T21:19:17.486835Z", span = "PT1S" }
abxbus = { timestamp = "2026-07-24T10:29:24.328016284Z", span = "PT1S" }
abx-plugins = "2100-01-01T00:00:00Z"
abx-dl = "2100-01-01T00:00:00Z"
abxpkg = "2100-01-01T00:00:00Z"
@ -149,6 +149,7 @@ dependencies = [
{ name = "ipython", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "platformdirs", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "psutil", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "psycopg", extra = ["binary"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "py-machineid", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "pydantic", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
{ name = "pydantic-settings", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
@ -251,6 +252,7 @@ requires-dist = [
{ name = "ipython", specifier = ">=8.27.0" },
{ name = "platformdirs", specifier = ">=4.3.6" },
{ name = "psutil", specifier = ">=6.0.0" },
{ name = "psycopg", extras = ["binary"], specifier = ">=3.2" },
{ name = "py-machineid", specifier = ">=0.6.0" },
{ name = "pydantic", specifier = ">=2.8.0" },
{ name = "pydantic-settings", specifier = ">=2.5.2" },
@ -1695,6 +1697,47 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/04/78/0acd37ca84ce3ddffaa92ef0f571e073faa6d8ff1f0559ab1272188ea2be/psutil-7.2.2-cp36-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b58fabe35e80b264a4e3bb23e6b96f9e45a3df7fb7eed419ac0e5947c61e47cc", size = 148266, upload-time = "2026-01-28T18:15:31.597Z" },
]
[[package]]
name = "psycopg"
version = "3.3.4"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/db/2f/cb91e5502ec9de1de6f1b76cfbf69531932725361168bb06963620c77e2e/psycopg-3.3.4.tar.gz", hash = "sha256:e21207764952cff81b6b8bdacad9a3939f2793367fdac2987b3aac36a651b5bc", size = 165799, upload-time = "2026-05-01T23:31:55.179Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/5c/e0/7b3dee031daae7743609ce3c746565d4a3ed7c2c186479eb48e34e838c64/psycopg-3.3.4-py3-none-any.whl", hash = "sha256:b6bbc25ccf05c8fad3b061d9db2ef0909a555171b84b07f29458a447253d679a", size = 213001, upload-time = "2026-05-01T23:20:50.816Z" },
]
[package.optional-dependencies]
binary = [
{ name = "psycopg-binary", marker = "(implementation_name != 'pypy' and sys_platform == 'darwin') or (implementation_name != 'pypy' and sys_platform == 'linux')" },
]
[[package]]
name = "psycopg-binary"
version = "3.3.4"
source = { registry = "https://pypi.org/simple" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/09/43/13e9c406fbbf354580476e248a16b64802a376873ebe6339e30bb655572d/psycopg_binary-3.3.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:fbd1d4ed566895ad2d3bf4ddfd8bae90026930ddf29df3b9d91d32c8c47866a7", size = 4590377, upload-time = "2026-05-01T23:29:18.782Z" },
{ url = "https://files.pythonhosted.org/packages/22/be/2923cd7c3683e7afdecf4f10796a18de02f5c5ddc0969aa2ad0a8cdd3bbd/psycopg_binary-3.3.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:75a9067e236f9b9ae3535b66fe99bddb33d39c0de10112e49b9ab11eee53dc31", size = 4669023, upload-time = "2026-05-01T23:29:25.884Z" },
{ url = "https://files.pythonhosted.org/packages/96/a0/2c913d6fe13d6a8bd13597d36739bf47af063ad9399e402cfecab16f3c1e/psycopg_binary-3.3.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:b56b603ebcea8aa10b46228b8410ba7f13e7c2ee54389d4d9be0927fd8ce2a70", size = 5467423, upload-time = "2026-05-01T23:29:33.416Z" },
{ url = "https://files.pythonhosted.org/packages/e7/38/205d10bc1ad0df4a21c5c51659126bd3ea0ef98fcad1e852f78c249bb9c3/psycopg_binary-3.3.4-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:c677c4ad433cb7150c8cd304a0769ae3bcfbe5ea0676eb53faa7b1443b16d0d3", size = 5151137, upload-time = "2026-05-01T23:29:42.013Z" },
{ url = "https://files.pythonhosted.org/packages/36/fc/f0381ddcd45eff3bb70dbca6823a996048d7f507b2ec3fc92c6fabc0fe87/psycopg_binary-3.3.4-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:26df2717e59c0473e4465a97dfb1b7afebaa479277870fd5784d1436470db47c", size = 6736671, upload-time = "2026-05-01T23:29:51.626Z" },
{ url = "https://files.pythonhosted.org/packages/95/40/fa545ae152c24327651e5624e4902121e808270be36c10b12e9939be09bc/psycopg_binary-3.3.4-cp313-cp313-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:1dc1f79fd16bb1f3f4421417a514607539f17804d95c7ed617265369d1981cae", size = 4979601, upload-time = "2026-05-01T23:29:56.961Z" },
{ url = "https://files.pythonhosted.org/packages/86/e4/2f8a47ee97f90cd2b933d0463081d35631ff419de2b8c984a5f369857de0/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:136f199a407b5348b9b857c504aff60c77622a28482e7195839ce1b51238c4cc", size = 4510513, upload-time = "2026-05-01T23:30:07.243Z" },
{ url = "https://files.pythonhosted.org/packages/0e/0e/94e842ff4a7f98ed162580ca2e8b8864b28c1e0350f2443f8ee47f821167/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:b6f5a29e9c775b9f12a1a717aa7a2c80f9e1db6f27ba44a5b59c80ac61d2ffcf", size = 4187243, upload-time = "2026-05-01T23:30:15.352Z" },
{ url = "https://files.pythonhosted.org/packages/d0/83/fc6c174b672e29b7de996ea77b6cbddf46c891751c3355f6974292baa6b4/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:ee17a2cf4943cde261adfad1bbc5bf38d6b3776d7afff74c7cabcbeaeb08c260", size = 3927347, upload-time = "2026-05-01T23:30:21.186Z" },
{ url = "https://files.pythonhosted.org/packages/e9/65/768364d4a97a15b1a7f47ba52688c1686f22941d8332a8398cefc468e25f/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:5c4ab71be17bdca30cb34c34c4e1496e2f5d6f20c199c12bad226070b22ef9bf", size = 4236393, upload-time = "2026-05-01T23:30:26.211Z" },
{ url = "https://files.pythonhosted.org/packages/48/a6/828c9185701dab71b234c2a76c38a08b098ebfec5020716b4e93807492b5/psycopg_binary-3.3.4-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:28b7398fdd19db3232c884fb24550bdfe951221f510e195e233299e4c9b78f97", size = 4607292, upload-time = "2026-05-01T23:30:38.962Z" },
{ url = "https://files.pythonhosted.org/packages/92/58/5b40dbc9d839045c9dae956960e4fb6d20bcabe6c59a2aa34fc3a371913f/psycopg_binary-3.3.4-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:1fbaa292a3c8bb61b45df1ad3da1908ccee7cb889db9425e3557d9e34e2a4829", size = 4687023, upload-time = "2026-05-01T23:30:47.227Z" },
{ url = "https://files.pythonhosted.org/packages/85/a9/793f0ac107a9003b48441d0d1f9f616d96e0f37458dd8dc12528ceff55fb/psycopg_binary-3.3.4-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:94596f9e7633ee3f6440711d43bb70aa31cc0a46a900ab8b4201a366ace5c9e7", size = 5486985, upload-time = "2026-05-01T23:30:55.517Z" },
{ url = "https://files.pythonhosted.org/packages/8f/26/42e8533497e2592334f68ec529cf5f840f7fa4e99575a4bb61aa184dbfbf/psycopg_binary-3.3.4-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:8c0056529e68dbe9184cd4019a1f3d8f3a4ead2f6fc7a5afcf27d3314edd1277", size = 5168745, upload-time = "2026-05-01T23:31:01.904Z" },
{ url = "https://files.pythonhosted.org/packages/15/af/b7151776cc08d5935d45c833ec818a9beb417cf7c08239af1aafbdae78ee/psycopg_binary-3.3.4-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2c09aad7051326e7603c14e50636db9c01f78272dc54b3accff03d46370461e6", size = 6761486, upload-time = "2026-05-01T23:31:14.511Z" },
{ url = "https://files.pythonhosted.org/packages/d0/ed/c92533b9124712d592cbf1cd6c76da933a2e0acea81dfe1fbe7e735f0cff/psycopg_binary-3.3.4-cp314-cp314-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:514404ed543efd620c85602b747df2a23cf1241b4067199e1a66f2d2757aaa41", size = 4997427, upload-time = "2026-05-01T23:31:20.901Z" },
{ url = "https://files.pythonhosted.org/packages/a2/23/ccadfd0de416aa188356daa199453af24087b042e296088706d190ae0295/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:46893c26858be12cc49ca4226ed6a60b4bfccadd946b3bebb783a60b38788228", size = 4533549, upload-time = "2026-05-01T23:31:26.204Z" },
{ url = "https://files.pythonhosted.org/packages/fd/a0/c8f43cee36386f7bc891ab41a9d31ea07cf9826038e732da79f26b1e5f34/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:df1d567fc430f6df15c9fcf67d87685fc49bdb325adc0db5af1adfb2f44eb5c9", size = 4210256, upload-time = "2026-05-01T23:31:33.884Z" },
{ url = "https://files.pythonhosted.org/packages/4e/2c/c1547871be3790676e8868b38655496422f94f0978dfb66b74bdba2f1676/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:6b9016b1714da4dd5ecaaa75b82098aa5a0b87854ce9b092e21c27c4ae23e014", size = 3946204, upload-time = "2026-05-01T23:31:39.626Z" },
{ url = "https://files.pythonhosted.org/packages/c4/b1/f6670f00fa7ea601584623f6c11602ab92117d83eaff885e0210f6de7418/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:47c656a8a7ba6eb0cff1801a4caaa9c8bdc12d03080e273aff1c8ac39971a77e", size = 4255811, upload-time = "2026-05-01T23:31:44.986Z" },
]
[[package]]
name = "ptyprocess"
version = "0.7.0"