mirror of
https://github.com/ArchiveBox/ArchiveBox.git
synced 2026-09-12 19:50:57 +05:00
Merge remote-tracking branch 'origin/dev' into codex/all-tests-ci
This commit is contained in:
commit
d4ff0d033f
8
.github/workflows/test-parallel.yml
vendored
8
.github/workflows/test-parallel.yml
vendored
@ -120,6 +120,14 @@ jobs:
|
||||
fi
|
||||
uv sync --locked --dev "${extra_args[@]}"
|
||||
|
||||
- name: Install PostgreSQL server binaries
|
||||
if: contains(matrix.test.path, 'test_postgres_backend')
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
if ! ls /usr/lib/postgresql/*/bin/initdb >/dev/null 2>&1 && ! command -v initdb >/dev/null 2>&1; then
|
||||
sudo apt-get update && sudo apt-get install -y postgresql
|
||||
fi
|
||||
|
||||
- name: Run ${{ matrix.test.name }}
|
||||
env:
|
||||
TEST_PATH: ${{ matrix.test.path }}
|
||||
|
||||
@ -12,19 +12,9 @@ import signal_webhooks.fields
|
||||
import signal_webhooks.utils
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
initial = True
|
||||
|
||||
dependencies = [
|
||||
("auth", "0012_alter_user_first_name_max_length"),
|
||||
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
migrations.RunSQL(
|
||||
sql="""
|
||||
# Raw sqlite DDL kept EXACTLY as-is; executed only on sqlite via Django's
|
||||
# RunSQL (identical statement splitting → byte-for-byte identical behavior).
|
||||
INITIAL_SQL = """
|
||||
-- Create api_apitoken table
|
||||
CREATE TABLE IF NOT EXISTS api_apitoken (
|
||||
id TEXT PRIMARY KEY NOT NULL,
|
||||
@ -70,12 +60,60 @@ class Migration(migrations.Migration):
|
||||
CREATE INDEX IF NOT EXISTS api_outboundwebhook_created_at_idx ON api_outboundwebhook(created_at);
|
||||
CREATE INDEX IF NOT EXISTS api_outboundwebhook_name_idx ON api_outboundwebhook(name);
|
||||
CREATE INDEX IF NOT EXISTS api_outboundwebhook_ref_idx ON api_outboundwebhook(ref);
|
||||
""",
|
||||
reverse_sql="""
|
||||
"""
|
||||
|
||||
INITIAL_REVERSE_SQL = """
|
||||
DROP TABLE IF EXISTS api_outboundwebhook;
|
||||
DROP TABLE IF EXISTS api_apitoken;
|
||||
""",
|
||||
),
|
||||
"""
|
||||
|
||||
|
||||
def _run_sqlite_only_sql(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_forwards(
|
||||
"api",
|
||||
schema_editor,
|
||||
None,
|
||||
None,
|
||||
)
|
||||
|
||||
|
||||
def _run_sqlite_only_sql_reverse(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_backwards(
|
||||
"api",
|
||||
schema_editor,
|
||||
None,
|
||||
None,
|
||||
)
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(apps, schema_editor, "api", ["APIToken", "OutboundWebhook"])
|
||||
|
||||
|
||||
def _pg_drop_schema(apps, schema_editor):
|
||||
from archivebox.misc.db import drop_models_on_postgres
|
||||
|
||||
drop_models_on_postgres(apps, schema_editor, "api", ["OutboundWebhook", "APIToken"])
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
initial = True
|
||||
|
||||
dependencies = [
|
||||
("auth", "0012_alter_user_first_name_max_length"),
|
||||
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
migrations.RunPython(_run_sqlite_only_sql, reverse_code=_run_sqlite_only_sql_reverse),
|
||||
],
|
||||
state_operations=[
|
||||
migrations.CreateModel(
|
||||
@ -236,4 +274,8 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
],
|
||||
),
|
||||
# On non-sqlite backends the raw DDL above is skipped, so the real
|
||||
# schema diverges from Django state. Rebuild both tables (including the
|
||||
# unique constraint) from the post-migration state (no-op on sqlite).
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
|
||||
]
|
||||
|
||||
@ -32,13 +32,13 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
|
||||
from archivebox.config import CONSTANTS, VERSION
|
||||
from archivebox.config.common import get_config
|
||||
from archivebox.config.collection import write_config_file
|
||||
from archivebox.misc.db import apply_migrations
|
||||
from archivebox.misc.db import apply_migrations, database_exists, ensure_database_ready
|
||||
from archivebox.misc.checks import check_migrations
|
||||
|
||||
config = get_config()
|
||||
|
||||
is_empty = not len(set(os.listdir(CONSTANTS.DATA_DIR)) - CONSTANTS.ALLOWED_IN_DATA_DIR)
|
||||
existing_index = os.path.isfile(CONSTANTS.DATABASE_FILE)
|
||||
existing_index = database_exists()
|
||||
if is_empty and not existing_index:
|
||||
print(f"[turquoise4][+] Initializing a new ArchiveBox v{VERSION} collection...[/turquoise4]")
|
||||
print("[green]----------------------------------------------------------------------[/green]")
|
||||
@ -86,11 +86,15 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
|
||||
# create the ArchiveBox.conf file
|
||||
write_config_file({"SECRET_KEY": config.SECRET_KEY})
|
||||
|
||||
if os.access(CONSTANTS.DATABASE_FILE, os.F_OK):
|
||||
if existing_index:
|
||||
print("\n[green][*] Verifying main SQL index and running any migrations needed...[/green]")
|
||||
else:
|
||||
print("\n[green][+] Building main SQL index and running initial migrations...[/green]")
|
||||
|
||||
# For postgres, make sure the server is reachable and create the database
|
||||
# if it doesn't exist yet (sqlite creates its file automatically).
|
||||
ensure_database_ready()
|
||||
|
||||
from archivebox.config.django import setup_django
|
||||
|
||||
setup_django()
|
||||
@ -107,9 +111,14 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
|
||||
else:
|
||||
os.environ["ARCHIVEBOX_WANTS_INIT"] = previous_wants_init
|
||||
|
||||
assert os.path.isfile(CONSTANTS.DATABASE_FILE) and os.access(CONSTANTS.DATABASE_FILE, os.R_OK)
|
||||
from archivebox.misc.db import database_display_location, is_postgres
|
||||
|
||||
assert database_exists()
|
||||
print()
|
||||
print(f" √ {_display_data_path(CONSTANTS.DATABASE_FILE, CONSTANTS.DATA_DIR)}")
|
||||
if is_postgres():
|
||||
print(f" √ {database_display_location()}")
|
||||
else:
|
||||
print(f" √ {_display_data_path(CONSTANTS.DATABASE_FILE, CONSTANTS.DATA_DIR)}")
|
||||
|
||||
print()
|
||||
print("[dodger_blue3][*] Checking links from indexes and archive folders (safe to Ctrl+C)...[/dodger_blue3]")
|
||||
|
||||
@ -44,7 +44,10 @@ def status(out_dir: Path = CONSTANTS.DATA_DIR) -> None:
|
||||
legacy_snapshot_dirs = [
|
||||
entry for entry in archive_dir.iterdir() if entry.is_dir() and not entry.is_symlink() and Snapshot.is_legacy_archive_dir(entry)
|
||||
]
|
||||
print(f" > SQL Main Index: {num_sql_links} links".ljust(36), f"(found in {CONSTANTS.SQL_INDEX_FILENAME})")
|
||||
from archivebox.misc.db import database_display_location, is_postgres
|
||||
|
||||
index_location = database_display_location() if is_postgres() else CONSTANTS.SQL_INDEX_FILENAME
|
||||
print(f" > SQL Main Index: {num_sql_links} links".ljust(36), f"(found in {index_location})")
|
||||
print(f" > JSON Link Details: {len(legacy_snapshot_dirs)} links".ljust(36), f"(found in {archive_dir.name}/*/index.json)")
|
||||
print()
|
||||
print("[green]\\[*] Scanning archive data directories...[/green]")
|
||||
|
||||
@ -350,7 +350,16 @@ class DatabaseConfig(BaseConfigSet):
|
||||
toml_section_header: str = "DATABASE_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
|
||||
|
||||
DATABASE_ENGINE: str = Field(
|
||||
default="sqlite",
|
||||
alias="ARCHIVEBOX_DATABASE_ENGINE",
|
||||
pattern=r"(?i)^(sqlite|postgres|postgresql)$",
|
||||
)
|
||||
DATABASE_NAME: str = Field(default=str(CONSTANTS.DATABASE_FILE), alias="ARCHIVEBOX_DATABASE_NAME")
|
||||
DATABASE_HOST: str = Field(default="127.0.0.1", alias="ARCHIVEBOX_DATABASE_HOST")
|
||||
DATABASE_PORT: int = Field(default=5432, alias="ARCHIVEBOX_DATABASE_PORT", ge=1, le=65535)
|
||||
DATABASE_USER: str = Field(default="archivebox", alias="ARCHIVEBOX_DATABASE_USER")
|
||||
DATABASE_PASSWORD: str = Field(default="", alias="ARCHIVEBOX_DATABASE_PASSWORD")
|
||||
SQLITE_JOURNAL_MODE: str = Field(
|
||||
default="WAL",
|
||||
alias="ARCHIVEBOX_SQLITE_JOURNAL_MODE",
|
||||
|
||||
@ -152,9 +152,10 @@ def setup_django(check_db=False, in_memory_db=False) -> None:
|
||||
for conn in connections.all():
|
||||
conn.close_if_unusable_or_obsolete()
|
||||
|
||||
sql_index_path = CONSTANTS.DATABASE_FILE
|
||||
assert os.access(sql_index_path, os.F_OK), (
|
||||
f"No database file {sql_index_path} found in: {CONSTANTS.DATA_DIR} (Are you in an ArchiveBox collection directory?)"
|
||||
from archivebox.misc.db import database_display_location, database_exists
|
||||
|
||||
assert database_exists(), (
|
||||
f"No database {database_display_location()} found for: {CONSTANTS.DATA_DIR} (Are you in an ArchiveBox collection directory?)"
|
||||
)
|
||||
|
||||
except KeyboardInterrupt:
|
||||
|
||||
@ -58,9 +58,11 @@ def _get_collection_id(DATA_DIR=DATA_DIR, force_create=False) -> str:
|
||||
collection_id = hashlib.sha256(f"{machine_id}:{collection_path}@{creation_date}".encode()).hexdigest()[:8]
|
||||
|
||||
try:
|
||||
# only persist collection_id file if we already have an index.sqlite3 file present
|
||||
# only persist collection_id file if this dir already looks like a real collection
|
||||
# (has an index.sqlite3, or an ArchiveBox.conf when the DB lives in postgres),
|
||||
# otherwise we might be running in a directory that is not a collection, no point creating cruft files
|
||||
collection_is_active = os.path.isfile(DATABASE_FILE) and os.path.isdir(ARCHIVE_DIR) and os.access(DATA_DIR, os.W_OK)
|
||||
collection_marker = os.path.isfile(DATABASE_FILE) or os.path.isfile(DATA_DIR / "ArchiveBox.conf")
|
||||
collection_is_active = collection_marker and os.path.isdir(ARCHIVE_DIR) and os.access(DATA_DIR, os.W_OK)
|
||||
if collection_is_active or force_create:
|
||||
collection_id_file.write_text(collection_id)
|
||||
|
||||
@ -270,6 +272,24 @@ def get_or_create_working_lib_dir(autofix=True, quiet=False, config: "ArchiveBox
|
||||
raise OSError(f"ArchiveBox is unable to find a writable ABXPKG_LIB_DIR, tried {CANDIDATES}!")
|
||||
|
||||
|
||||
def _sql_index_location() -> dict:
|
||||
from archivebox.misc.db import database_display_location, database_exists, is_postgres
|
||||
|
||||
if is_postgres():
|
||||
return {
|
||||
"path": database_display_location(),
|
||||
"enabled": True,
|
||||
"is_valid": database_exists(),
|
||||
"is_mount": False,
|
||||
}
|
||||
return {
|
||||
"path": DATABASE_FILE.resolve(),
|
||||
"enabled": True,
|
||||
"is_valid": os.path.isfile(DATABASE_FILE) and os.access(DATABASE_FILE, os.R_OK) and os.access(DATABASE_FILE, os.W_OK),
|
||||
"is_mount": os.path.ismount(DATABASE_FILE.resolve()),
|
||||
}
|
||||
|
||||
|
||||
def get_data_locations(config: "ArchiveBoxConfig | None" = None, **config_kwargs):
|
||||
from archivebox.config.constants import CONSTANTS
|
||||
from archivebox.config.common import get_config
|
||||
@ -296,12 +316,7 @@ def get_data_locations(config: "ArchiveBoxConfig | None" = None, **config_kwargs
|
||||
and os.access(CONSTANTS.CONFIG_FILE, os.R_OK)
|
||||
and os.access(CONSTANTS.CONFIG_FILE, os.W_OK),
|
||||
},
|
||||
"SQL_INDEX": {
|
||||
"path": DATABASE_FILE.resolve(),
|
||||
"enabled": True,
|
||||
"is_valid": os.path.isfile(DATABASE_FILE) and os.access(DATABASE_FILE, os.R_OK) and os.access(DATABASE_FILE, os.W_OK),
|
||||
"is_mount": os.path.ismount(DATABASE_FILE.resolve()),
|
||||
},
|
||||
"SQL_INDEX": _sql_index_location(),
|
||||
"ARCHIVE_DIR": {
|
||||
"path": CONSTANTS.ARCHIVE_DIR.resolve(),
|
||||
"enabled": True,
|
||||
|
||||
@ -122,8 +122,6 @@ class ArchiveBoxAdmin(admin.AdminSite):
|
||||
|
||||
def index(self, request: "HttpRequest", extra_context: dict[str, Any] | None = None) -> "TemplateResponse":
|
||||
response = super().index(request, extra_context)
|
||||
if connection.vendor != "sqlite":
|
||||
return response
|
||||
|
||||
models_by_table: dict[str, list[dict[str, Any]]] = {}
|
||||
for app in response.context_data.get("app_list", []):
|
||||
@ -136,23 +134,18 @@ class ArchiveBoxAdmin(admin.AdminSite):
|
||||
if not models_by_table:
|
||||
return response
|
||||
|
||||
try:
|
||||
with connection.cursor() as cursor:
|
||||
cursor.execute("SELECT tbl, stat FROM sqlite_stat1")
|
||||
for table, stat in cursor.fetchall():
|
||||
try:
|
||||
count = int(str(stat).split()[0])
|
||||
except (IndexError, TypeError, ValueError):
|
||||
continue
|
||||
self._set_model_object_count(
|
||||
models_by_table,
|
||||
table,
|
||||
count,
|
||||
title=f"Approximate count from SQLite stats: {count:,}",
|
||||
)
|
||||
models_by_table.pop(table, None)
|
||||
except DatabaseError:
|
||||
pass
|
||||
from archivebox.misc.db import approximate_row_counts
|
||||
|
||||
for table, count in approximate_row_counts(connection).items():
|
||||
if table not in models_by_table:
|
||||
continue
|
||||
self._set_model_object_count(
|
||||
models_by_table,
|
||||
table,
|
||||
count,
|
||||
title=f"Approximate count from database stats: {count:,}",
|
||||
)
|
||||
models_by_table.pop(table, None)
|
||||
|
||||
for table in list(models_by_table):
|
||||
try:
|
||||
|
||||
@ -20,6 +20,14 @@ class CoreConfig(AppConfig):
|
||||
|
||||
ModelWithOutputDir.register_delete_signal()
|
||||
|
||||
# SQLite ignores VARCHAR(n) limits but PostgreSQL enforces them; clamp
|
||||
# CharField values on save so writes behave the same on both backends.
|
||||
from django.db.models.signals import pre_save
|
||||
|
||||
from archivebox.misc.db import truncate_overlong_charfields
|
||||
|
||||
pre_save.connect(truncate_overlong_charfields, dispatch_uid="archivebox_truncate_overlong_charfields")
|
||||
|
||||
# Import models to register state machines with the registry
|
||||
# Skip during makemigrations to avoid premature state machine access
|
||||
if "makemigrations" not in sys.argv:
|
||||
|
||||
@ -47,6 +47,14 @@ def normalize_status(status):
|
||||
|
||||
def upgrade_core_tables(apps, schema_editor):
|
||||
"""Upgrade core tables from v0.7.2 or v0.8.6rc0 to v0.9.0."""
|
||||
# sqlite-only legacy repair/rebuild. On postgres this raw SQL (PRAGMA,
|
||||
# sqlite_master, INSERT OR IGNORE, DATETIME, table-rebuild dance) is both
|
||||
# invalid and unnecessary: a postgres database can never contain legacy
|
||||
# data at this point. The final _pg_sync_schema op resyncs the real
|
||||
# postgres schema to migration state instead.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
from archivebox.uuid_compat import uuid7
|
||||
|
||||
cursor = connection.cursor()
|
||||
@ -491,6 +499,15 @@ def upgrade_core_tables(apps, schema_editor):
|
||||
print(" ✓ Core table rebuild complete")
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
# On postgres, upgrade_core_tables is a no-op above, so the raw sqlite
|
||||
# rebuilds of core_tag/core_snapshot/core_archiveresult never ran. Resync
|
||||
# the real schema to this migration's end-state (tables are empty on pg).
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(apps, schema_editor, "core", ["Tag", "Snapshot", "SnapshotTag", "ArchiveResult"])
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("core", "0022_auto_20231023_2008"),
|
||||
@ -571,4 +588,5 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
],
|
||||
),
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
|
||||
]
|
||||
|
||||
@ -4,11 +4,102 @@
|
||||
from django.db import migrations, models
|
||||
|
||||
|
||||
# Raw sqlite table-rebuild that makes core_snapshot.crawl_id NOT NULL. Kept
|
||||
# byte-for-byte and replayed through Django's own RunSQL only on sqlite.
|
||||
_MAKE_CRAWL_ID_NOT_NULL_SQL = """
|
||||
-- Rebuild snapshot table with NOT NULL crawl_id
|
||||
CREATE TABLE core_snapshot_final (
|
||||
id TEXT PRIMARY KEY NOT NULL,
|
||||
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
modified_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
url TEXT NOT NULL,
|
||||
timestamp VARCHAR(32) NOT NULL UNIQUE,
|
||||
bookmarked_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
crawl_id TEXT NOT NULL,
|
||||
parent_snapshot_id TEXT,
|
||||
|
||||
title VARCHAR(512),
|
||||
downloaded_at DATETIME,
|
||||
depth INTEGER NOT NULL DEFAULT 0,
|
||||
fs_version VARCHAR(10) NOT NULL DEFAULT '0.9.0',
|
||||
|
||||
config TEXT NOT NULL DEFAULT '{}',
|
||||
notes TEXT NOT NULL DEFAULT '',
|
||||
num_uses_succeeded INTEGER NOT NULL DEFAULT 0,
|
||||
num_uses_failed INTEGER NOT NULL DEFAULT 0,
|
||||
|
||||
status VARCHAR(15) NOT NULL DEFAULT 'queued',
|
||||
retry_at DATETIME,
|
||||
current_step INTEGER NOT NULL DEFAULT 0,
|
||||
|
||||
FOREIGN KEY (crawl_id) REFERENCES crawls_crawl(id) ON DELETE CASCADE,
|
||||
FOREIGN KEY (parent_snapshot_id) REFERENCES core_snapshot(id) ON DELETE SET NULL
|
||||
);
|
||||
|
||||
INSERT INTO core_snapshot_final (
|
||||
id, url, timestamp, title,
|
||||
bookmarked_at, created_at, modified_at,
|
||||
crawl_id, parent_snapshot_id,
|
||||
downloaded_at, depth, fs_version,
|
||||
config, notes,
|
||||
num_uses_succeeded, num_uses_failed,
|
||||
status, retry_at, current_step
|
||||
)
|
||||
SELECT
|
||||
id, url, timestamp, title,
|
||||
bookmarked_at, created_at, modified_at,
|
||||
REPLACE(crawl_id, '-', ''), REPLACE(parent_snapshot_id, '-', ''),
|
||||
downloaded_at, depth, fs_version,
|
||||
COALESCE(config, '{}'), COALESCE(notes, ''),
|
||||
num_uses_succeeded, num_uses_failed,
|
||||
status, retry_at, current_step
|
||||
FROM core_snapshot;
|
||||
|
||||
DROP TABLE core_snapshot;
|
||||
ALTER TABLE core_snapshot_final RENAME TO core_snapshot;
|
||||
|
||||
CREATE INDEX core_snapshot_url_idx ON core_snapshot(url);
|
||||
CREATE INDEX core_snapshot_timestamp_idx ON core_snapshot(timestamp);
|
||||
CREATE INDEX core_snapshot_bookmarked_at_idx ON core_snapshot(bookmarked_at);
|
||||
CREATE INDEX core_snapshot_crawl_id_idx ON core_snapshot(crawl_id);
|
||||
CREATE INDEX core_snapshot_status_idx ON core_snapshot(status);
|
||||
CREATE INDEX core_snapshot_retry_at_idx ON core_snapshot(retry_at);
|
||||
CREATE INDEX core_snapshot_created_at_idx ON core_snapshot(created_at);
|
||||
CREATE UNIQUE INDEX core_snapshot_url_crawl_unique ON core_snapshot(url, crawl_id);
|
||||
"""
|
||||
|
||||
|
||||
def _make_crawl_id_not_null(apps, schema_editor):
|
||||
# sqlite-only table rebuild. Reuse Django's own RunSQL statement splitting so
|
||||
# the sqlite behavior is byte-for-byte identical to the original RunSQL op.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(sql=_MAKE_CRAWL_ID_NOT_NULL_SQL).database_forwards("core", schema_editor, None, None)
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
# On postgres the raw sqlite rebuilds above are skipped and crawl was only
|
||||
# added to migration state; resync the real schema to this migration's
|
||||
# end-state (empty tables). Snapshot is referenced by SnapshotTag and
|
||||
# ArchiveResult, so those are rebuilt too to restore FK constraints.
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(apps, schema_editor, "core", ["Snapshot", "SnapshotTag", "ArchiveResult"])
|
||||
|
||||
|
||||
def create_default_crawl_and_assign_snapshots(apps, schema_editor):
|
||||
"""
|
||||
Create a default crawl for migrated snapshots and assign all snapshots without a crawl to it.
|
||||
Uses raw SQL because the app registry isn't fully populated during migrations.
|
||||
"""
|
||||
# Legacy-data-only, sqlite-specific (PRAGMA + '?' placeholders). On a fresh
|
||||
# postgres install core_snapshot has no crawl_id column yet and there are no
|
||||
# unassigned snapshots, so gate before touching any SQL.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
from django.db import connection
|
||||
import uuid as uuid_lib
|
||||
from datetime import datetime
|
||||
@ -89,72 +180,10 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
# Now make crawl_id NOT NULL
|
||||
migrations.RunSQL(
|
||||
sql="""
|
||||
-- Rebuild snapshot table with NOT NULL crawl_id
|
||||
CREATE TABLE core_snapshot_final (
|
||||
id TEXT PRIMARY KEY NOT NULL,
|
||||
created_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
modified_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
url TEXT NOT NULL,
|
||||
timestamp VARCHAR(32) NOT NULL UNIQUE,
|
||||
bookmarked_at DATETIME NOT NULL DEFAULT CURRENT_TIMESTAMP,
|
||||
|
||||
crawl_id TEXT NOT NULL,
|
||||
parent_snapshot_id TEXT,
|
||||
|
||||
title VARCHAR(512),
|
||||
downloaded_at DATETIME,
|
||||
depth INTEGER NOT NULL DEFAULT 0,
|
||||
fs_version VARCHAR(10) NOT NULL DEFAULT '0.9.0',
|
||||
|
||||
config TEXT NOT NULL DEFAULT '{}',
|
||||
notes TEXT NOT NULL DEFAULT '',
|
||||
num_uses_succeeded INTEGER NOT NULL DEFAULT 0,
|
||||
num_uses_failed INTEGER NOT NULL DEFAULT 0,
|
||||
|
||||
status VARCHAR(15) NOT NULL DEFAULT 'queued',
|
||||
retry_at DATETIME,
|
||||
current_step INTEGER NOT NULL DEFAULT 0,
|
||||
|
||||
FOREIGN KEY (crawl_id) REFERENCES crawls_crawl(id) ON DELETE CASCADE,
|
||||
FOREIGN KEY (parent_snapshot_id) REFERENCES core_snapshot(id) ON DELETE SET NULL
|
||||
);
|
||||
|
||||
INSERT INTO core_snapshot_final (
|
||||
id, url, timestamp, title,
|
||||
bookmarked_at, created_at, modified_at,
|
||||
crawl_id, parent_snapshot_id,
|
||||
downloaded_at, depth, fs_version,
|
||||
config, notes,
|
||||
num_uses_succeeded, num_uses_failed,
|
||||
status, retry_at, current_step
|
||||
)
|
||||
SELECT
|
||||
id, url, timestamp, title,
|
||||
bookmarked_at, created_at, modified_at,
|
||||
REPLACE(crawl_id, '-', ''), REPLACE(parent_snapshot_id, '-', ''),
|
||||
downloaded_at, depth, fs_version,
|
||||
COALESCE(config, '{}'), COALESCE(notes, ''),
|
||||
num_uses_succeeded, num_uses_failed,
|
||||
status, retry_at, current_step
|
||||
FROM core_snapshot;
|
||||
|
||||
DROP TABLE core_snapshot;
|
||||
ALTER TABLE core_snapshot_final RENAME TO core_snapshot;
|
||||
|
||||
CREATE INDEX core_snapshot_url_idx ON core_snapshot(url);
|
||||
CREATE INDEX core_snapshot_timestamp_idx ON core_snapshot(timestamp);
|
||||
CREATE INDEX core_snapshot_bookmarked_at_idx ON core_snapshot(bookmarked_at);
|
||||
CREATE INDEX core_snapshot_crawl_id_idx ON core_snapshot(crawl_id);
|
||||
CREATE INDEX core_snapshot_status_idx ON core_snapshot(status);
|
||||
CREATE INDEX core_snapshot_retry_at_idx ON core_snapshot(retry_at);
|
||||
CREATE INDEX core_snapshot_created_at_idx ON core_snapshot(created_at);
|
||||
CREATE UNIQUE INDEX core_snapshot_url_crawl_unique ON core_snapshot(url, crawl_id);
|
||||
""",
|
||||
reverse_sql=migrations.RunSQL.noop,
|
||||
# Now make crawl_id NOT NULL (sqlite-only table rebuild)
|
||||
migrations.RunPython(
|
||||
_make_crawl_id_not_null,
|
||||
reverse_code=migrations.RunPython.noop,
|
||||
),
|
||||
],
|
||||
state_operations=[
|
||||
@ -169,4 +198,5 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
],
|
||||
),
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
|
||||
]
|
||||
|
||||
@ -11,6 +11,12 @@ from archivebox.uuid_compat import uuid7
|
||||
|
||||
def copy_old_fields_to_new(apps, schema_editor):
|
||||
"""Copy data from old field names to new field names after AddField operations."""
|
||||
# sqlite-only legacy data copy (PRAGMA introspection). On postgres there is
|
||||
# no legacy data and the state-only AddFields below are resynced by
|
||||
# _pg_sync_schema at the end of this migration.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
cursor = connection.cursor()
|
||||
|
||||
# Check if old fields still exist
|
||||
@ -43,6 +49,19 @@ def copy_old_fields_to_new(apps, schema_editor):
|
||||
# NOTE: UUIDs are already populated by migration 0023 for all migration paths.
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
# This migration mixes real ORM AddFields with state-only AddFields (the
|
||||
# snapshot config/current_step/depth/notes/num_uses_*/parent_snapshot/
|
||||
# retry_at/status columns and archiveresult created_at/modified_at were
|
||||
# created on sqlite by 0023's raw rebuild, so here they are state-only).
|
||||
# On postgres those columns would never land, desyncing the real schema.
|
||||
# Resync to this migration's final state (empty tables on pg). Snapshot is
|
||||
# referenced by SnapshotTag and ArchiveResult, so rebuild those too.
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(apps, schema_editor, "core", ["Snapshot", "SnapshotTag", "ArchiveResult"])
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("core", "0024_assign_default_crawl"),
|
||||
@ -335,4 +354,5 @@ class Migration(migrations.Migration):
|
||||
model_name="snapshot",
|
||||
constraint=models.UniqueConstraint(fields=("timestamp",), name="unique_timestamp"),
|
||||
),
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
|
||||
]
|
||||
|
||||
@ -275,6 +275,14 @@ def copy_archiveresult_data_to_process(apps, schema_editor):
|
||||
- failed → exited (exit_code=1)
|
||||
- skipped → exited (exit_code=None)
|
||||
"""
|
||||
# sqlite-only legacy data copy (PRAGMA introspection, '?' placeholders).
|
||||
# A postgres install can never contain legacy ArchiveResult cmd/pwd data at
|
||||
# this point, so there is nothing to copy into machine_process. The
|
||||
# RemoveField ops below are real (non-state-only) and drop the now-empty
|
||||
# cmd/pwd/cmd_version columns on both vendors, so no resync is needed.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
cursor = connection.cursor()
|
||||
|
||||
# Check if old fields still exist (skip if fresh install or already migrated)
|
||||
|
||||
@ -26,6 +26,12 @@ def migrate_archiveresult_id_to_uuid(apps, schema_editor):
|
||||
|
||||
Result: Clean schema with ONLY id as UUIDField (no old_id, no uuid)
|
||||
"""
|
||||
# sqlite-only table rebuild (sqlite_master, PRAGMA, table copy/rename). On
|
||||
# postgres this is unnecessary and invalid: the table is empty and the
|
||||
# state flip of id -> UUIDField is applied by _pg_sync_schema below.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
cursor = connection.cursor()
|
||||
|
||||
# Check if table exists and has data
|
||||
@ -190,6 +196,16 @@ def migrate_archiveresult_id_to_uuid(apps, schema_editor):
|
||||
print(f" ✓ ArchiveResult UUID primary key migration complete ({row_count} records)")
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
# On postgres the sqlite table rebuild above is skipped; the id -> UUIDField
|
||||
# flip and uuid-field removal only reach migration state. Resync the real
|
||||
# (empty) core_archiveresult table to this migration's end-state. Nothing in
|
||||
# core references ArchiveResult, so no other models need rebuilding.
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(apps, schema_editor, "core", ["ArchiveResult"])
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("core", "0028_alter_snapshot_fs_version"),
|
||||
@ -217,4 +233,5 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
],
|
||||
),
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
|
||||
]
|
||||
|
||||
@ -13,6 +13,13 @@ def _repair_snapshot_permissions(apps, schema_editor):
|
||||
no underlying column. Fresh installs added the column via 0041 and
|
||||
this is a no-op.
|
||||
"""
|
||||
# sqlite-only legacy repair (PRAGMA table_xinfo + sqlite VIRTUAL generated
|
||||
# column syntax). On postgres the permissions column was created by 0041's
|
||||
# portable GeneratedField AddField (STORED generated column), so there is
|
||||
# nothing to repair.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
cursor = schema_editor.connection.cursor()
|
||||
# ``table_xinfo`` lists STORED/VIRTUAL generated columns; ``table_info``
|
||||
# silently drops them, so a prior 0041 that landed the STORED column
|
||||
|
||||
@ -0,0 +1,30 @@
|
||||
# Postgres-only: btree pattern-ops index on core_snapshot.url so LIKE 'prefix%'
|
||||
# queries (url__startswith and the URL prefix search) stay index scans under any
|
||||
# database collation. SQLite needs nothing here: its plain url index already
|
||||
# serves the bytewise range comparisons used on that backend.
|
||||
|
||||
from django.db import migrations
|
||||
|
||||
|
||||
def add_pg_url_pattern_index(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "postgresql":
|
||||
return
|
||||
schema_editor.execute(
|
||||
"CREATE INDEX IF NOT EXISTS core_snapshot_url_pattern_ops_idx ON core_snapshot (url text_pattern_ops)",
|
||||
)
|
||||
|
||||
|
||||
def remove_pg_url_pattern_index(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "postgresql":
|
||||
return
|
||||
schema_editor.execute("DROP INDEX IF EXISTS core_snapshot_url_pattern_ops_idx")
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("core", "0050_snapshot_permissions_not_null"),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.RunPython(add_pg_url_pattern_index, remove_pg_url_pattern_index),
|
||||
]
|
||||
@ -183,9 +183,14 @@ class SnapshotQuerySet(models.QuerySet):
|
||||
for crawl_id, permissions in Crawl.objects.filter(pk__in=missing_crawl_ids).values_list("pk", "permissions")
|
||||
}
|
||||
|
||||
from archivebox.misc.db import truncate_overlong_charfields
|
||||
|
||||
for obj in objs:
|
||||
if isinstance(obj, self.model):
|
||||
obj.ensure_permissions_config(crawl_permissions=crawl_permissions_by_id.get(str(obj.crawl_id)))
|
||||
# bulk_create bypasses pre_save, so clamp CharFields here too
|
||||
# (e.g. page titles) to stay within postgres VARCHAR limits.
|
||||
truncate_overlong_charfields(obj)
|
||||
return super().bulk_create(objs, *args, **kwargs)
|
||||
|
||||
def paged_iterator(self, chunk_size: int = 500):
|
||||
|
||||
@ -14,6 +14,10 @@ import archivebox
|
||||
from archivebox.config.constants import CONSTANTS
|
||||
from archivebox.config.common import get_config
|
||||
from archivebox.core.routes_util import get_api_base_url, get_admin_base_url, get_base_url, normalize_base_url
|
||||
|
||||
# DATABASE_ENGINE config selects the backend (sqlite by default); the
|
||||
# sqlite-vs-postgres helpers live in archivebox.misc.db.
|
||||
from archivebox.misc.db import is_postgres, postgres_db_params
|
||||
from .settings_logging import SETTINGS_LOGGING
|
||||
|
||||
|
||||
@ -245,12 +249,21 @@ SQLITE_CONNECTION_OPTIONS = {
|
||||
},
|
||||
}
|
||||
|
||||
DATABASES = {
|
||||
"default": {
|
||||
"NAME": DATABASE_NAME,
|
||||
**SQLITE_CONNECTION_OPTIONS,
|
||||
},
|
||||
}
|
||||
if is_postgres():
|
||||
DATABASES = {
|
||||
"default": {
|
||||
"ENGINE": "django.db.backends.postgresql",
|
||||
**postgres_db_params(),
|
||||
"OPTIONS": {"connect_timeout": 10},
|
||||
},
|
||||
}
|
||||
else:
|
||||
DATABASES = {
|
||||
"default": {
|
||||
"NAME": DATABASE_NAME,
|
||||
**SQLITE_CONNECTION_OPTIONS,
|
||||
},
|
||||
}
|
||||
MIGRATION_MODULES = {"signal_webhooks": None}
|
||||
|
||||
# Django requires DEFAULT_AUTO_FIELD to subclass AutoField (BigAutoField, SmallAutoField, etc.)
|
||||
|
||||
@ -276,12 +276,20 @@ class SnapshotView(View):
|
||||
"""
|
||||
|
||||
def _fragmentless_url_query(url: str) -> Q:
|
||||
# Use a range comparison (url >= 'canonical#' AND url < 'canonical#\U0010ffff')
|
||||
# instead of LIKE/__startswith — SQLite's case-insensitive LIKE bypasses the
|
||||
# url index and forces a full-table scan over ~1M rows (~250ms). The range
|
||||
# form lets SQLite use a MULTI-INDEX OR and stays under 1ms.
|
||||
from archivebox.misc.db import is_postgres
|
||||
|
||||
canonical = without_fragment(url)
|
||||
return Q(url=canonical) | (Q(url__gte=f"{canonical}#") & Q(url__lt=f"{canonical}#\U0010ffff"))
|
||||
if not is_postgres():
|
||||
# Use a range comparison (url >= 'canonical#' AND url < 'canonical#\U0010ffff')
|
||||
# instead of LIKE/__startswith — SQLite's case-insensitive LIKE bypasses the
|
||||
# url index and forces a full-table scan over ~1M rows (~250ms). The range
|
||||
# form lets SQLite use a MULTI-INDEX OR and stays under 1ms.
|
||||
return Q(url=canonical) | (Q(url__gte=f"{canonical}#") & Q(url__lt=f"{canonical}#\U0010ffff"))
|
||||
# On postgres the range trick is unsafe: linguistic (ICU/libc) collations
|
||||
# don't compare '#'-suffixed strings bytewise, so the range can miss rows.
|
||||
# startswith compiles to LIKE 'prefix%' with wildcards escaped, which is
|
||||
# correct under any collation and uses the url pattern-ops index.
|
||||
return Q(url=canonical) | Q(url__startswith=f"{canonical}#")
|
||||
|
||||
normalized = without_fragment(path)
|
||||
if path.startswith(("http://", "https://")):
|
||||
|
||||
@ -10,19 +10,9 @@ from archivebox.uuid_compat import uuid7
|
||||
from archivebox.base_models.models import get_or_create_system_user_pk
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
initial = True
|
||||
|
||||
dependencies = [
|
||||
("auth", "0012_alter_user_first_name_max_length"),
|
||||
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
migrations.RunSQL(
|
||||
sql="""
|
||||
# Raw sqlite DDL kept EXACTLY as-is; executed only on sqlite via Django's
|
||||
# RunSQL (identical statement splitting → byte-for-byte identical behavior).
|
||||
INITIAL_SQL = """
|
||||
-- Create crawls_crawlschedule table first (circular FK will be added later)
|
||||
CREATE TABLE IF NOT EXISTS crawls_crawlschedule (
|
||||
id TEXT PRIMARY KEY NOT NULL,
|
||||
@ -75,12 +65,60 @@ class Migration(migrations.Migration):
|
||||
CREATE INDEX IF NOT EXISTS crawls_crawl_created_at_idx ON crawls_crawl(created_at);
|
||||
CREATE INDEX IF NOT EXISTS crawls_crawl_created_by_id_idx ON crawls_crawl(created_by_id);
|
||||
CREATE INDEX IF NOT EXISTS crawls_crawl_schedule_id_idx ON crawls_crawl(schedule_id);
|
||||
""",
|
||||
reverse_sql="""
|
||||
"""
|
||||
|
||||
INITIAL_REVERSE_SQL = """
|
||||
DROP TABLE IF EXISTS crawls_crawl;
|
||||
DROP TABLE IF EXISTS crawls_crawlschedule;
|
||||
""",
|
||||
),
|
||||
"""
|
||||
|
||||
|
||||
def _run_sqlite_only_sql(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_forwards(
|
||||
"crawls",
|
||||
schema_editor,
|
||||
None,
|
||||
None,
|
||||
)
|
||||
|
||||
|
||||
def _run_sqlite_only_sql_reverse(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(sql=INITIAL_SQL, reverse_sql=INITIAL_REVERSE_SQL).database_backwards(
|
||||
"crawls",
|
||||
schema_editor,
|
||||
None,
|
||||
None,
|
||||
)
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(apps, schema_editor, "crawls", ["CrawlSchedule", "Crawl"])
|
||||
|
||||
|
||||
def _pg_drop_schema(apps, schema_editor):
|
||||
from archivebox.misc.db import drop_models_on_postgres
|
||||
|
||||
drop_models_on_postgres(apps, schema_editor, "crawls", ["Crawl", "CrawlSchedule"])
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
initial = True
|
||||
|
||||
dependencies = [
|
||||
("auth", "0012_alter_user_first_name_max_length"),
|
||||
migrations.swappable_dependency(settings.AUTH_USER_MODEL),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
migrations.RunPython(_run_sqlite_only_sql, reverse_code=_run_sqlite_only_sql_reverse),
|
||||
],
|
||||
state_operations=[
|
||||
migrations.CreateModel(
|
||||
@ -174,4 +212,9 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
],
|
||||
),
|
||||
# On non-sqlite backends the raw DDL above is skipped, so the real
|
||||
# schema diverges from Django state. Rebuild both tables from the
|
||||
# post-migration state (no-op on sqlite). Runs as a top-level op so
|
||||
# ``apps`` reflects the full state including the circular template FK.
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
|
||||
]
|
||||
|
||||
@ -6,6 +6,10 @@ from django.db import migrations, connection
|
||||
|
||||
def upgrade_crawl_table_from_v086(apps, schema_editor):
|
||||
"""Upgrade crawls_crawl table from v0.8.6rc0 schema to v0.9.0 schema."""
|
||||
# sqlite-only legacy repair (sqlite_master/PRAGMA/GLOB). Postgres support
|
||||
# postdates all legacy data, so this never has work to do there.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
cursor = connection.cursor()
|
||||
|
||||
# Check if crawls_crawl table exists
|
||||
|
||||
@ -97,6 +97,12 @@ def _ensure_permissions_column(cursor):
|
||||
|
||||
|
||||
def hydrate_crawl_permissions(apps, schema_editor):
|
||||
# sqlite-only legacy hydration + generated-column repair (PRAGMA
|
||||
# table_xinfo / VIRTUAL generated column). On postgres the ``permissions``
|
||||
# column is created portably by 0013 (AddField GeneratedField) and there is
|
||||
# never legacy data to hydrate, so this is a no-op there.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
Crawl = apps.get_model("crawls", "Crawl")
|
||||
base_config = raw_base_config(apps)
|
||||
default_permissions = resolve_permissions(base_config, "public")
|
||||
|
||||
@ -7,16 +7,45 @@ import django.utils.timezone
|
||||
from archivebox.uuid_compat import uuid7
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
initial = True
|
||||
def _run_sqlite_initial_ddl(apps, schema_editor):
|
||||
# sqlite-only: raw DDL creates the real tables byte-for-byte as historically
|
||||
# shipped. On other vendors the tables are built from migration state by
|
||||
# _pg_sync_schema below (a fresh non-sqlite DB never holds legacy data here).
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(
|
||||
sql=_SQLITE_INITIAL_SQL,
|
||||
reverse_sql=_SQLITE_INITIAL_REVERSE_SQL,
|
||||
).database_forwards("machine", schema_editor, None, None)
|
||||
|
||||
dependencies = []
|
||||
|
||||
operations = [
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
migrations.RunSQL(
|
||||
sql="""
|
||||
def _reverse_sqlite_initial_ddl(apps, schema_editor):
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
migrations.RunSQL(
|
||||
sql=_SQLITE_INITIAL_SQL,
|
||||
reverse_sql=_SQLITE_INITIAL_REVERSE_SQL,
|
||||
).database_backwards("machine", schema_editor, None, None)
|
||||
|
||||
|
||||
def _pg_sync_schema(apps, schema_editor):
|
||||
from archivebox.misc.db import rebuild_models_from_migration_state
|
||||
|
||||
rebuild_models_from_migration_state(
|
||||
apps,
|
||||
schema_editor,
|
||||
"machine",
|
||||
["Machine", "NetworkInterface", "Binary"],
|
||||
)
|
||||
|
||||
|
||||
def _pg_drop_schema(apps, schema_editor):
|
||||
from archivebox.misc.db import drop_models_on_postgres
|
||||
|
||||
drop_models_on_postgres(apps, schema_editor, "machine", ["Binary", "NetworkInterface", "Machine"])
|
||||
|
||||
|
||||
_SQLITE_INITIAL_SQL = """
|
||||
-- Create machine_machine table
|
||||
CREATE TABLE IF NOT EXISTS machine_machine (
|
||||
id TEXT PRIMARY KEY NOT NULL,
|
||||
@ -98,13 +127,25 @@ class Migration(migrations.Migration):
|
||||
CREATE INDEX IF NOT EXISTS machine_binary_status_idx ON machine_binary(status);
|
||||
CREATE INDEX IF NOT EXISTS machine_binary_retry_at_idx ON machine_binary(retry_at);
|
||||
|
||||
""",
|
||||
reverse_sql="""
|
||||
"""
|
||||
|
||||
|
||||
_SQLITE_INITIAL_REVERSE_SQL = """
|
||||
DROP TABLE IF EXISTS machine_binary;
|
||||
DROP TABLE IF EXISTS machine_networkinterface;
|
||||
DROP TABLE IF EXISTS machine_machine;
|
||||
""",
|
||||
),
|
||||
"""
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
initial = True
|
||||
|
||||
dependencies = []
|
||||
|
||||
operations = [
|
||||
migrations.SeparateDatabaseAndState(
|
||||
database_operations=[
|
||||
migrations.RunPython(_run_sqlite_initial_ddl, _reverse_sqlite_initial_ddl),
|
||||
],
|
||||
state_operations=[
|
||||
migrations.CreateModel(
|
||||
@ -244,4 +285,5 @@ class Migration(migrations.Migration):
|
||||
),
|
||||
],
|
||||
),
|
||||
migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
|
||||
]
|
||||
|
||||
@ -10,6 +10,13 @@ def converge_binary_table(apps, schema_editor):
|
||||
Drop machine_installedbinary if it exists (0.8.6rc0 path).
|
||||
Create machine_binary if it doesn't exist (needed by Process model).
|
||||
"""
|
||||
# sqlite-only legacy convergence: on a fresh install machine_binary already
|
||||
# exists with the final column set (created by 0001), so every branch below
|
||||
# is a no-op schema-wise. A non-sqlite DB never carries the 0.8.x legacy
|
||||
# shapes this repairs, and its Binary table already matches migration state.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
cursor = connection.cursor()
|
||||
|
||||
# Check what tables exist
|
||||
|
||||
@ -2,9 +2,12 @@ from django.db import migrations
|
||||
|
||||
|
||||
def remove_output_dir_if_exists(apps, schema_editor):
|
||||
cursor = schema_editor.connection.cursor()
|
||||
cursor.execute("PRAGMA table_info(machine_binary)")
|
||||
columns = {row[1] for row in cursor.fetchall()}
|
||||
# On non-sqlite the RemoveField below is state-only, so the real
|
||||
# ALTER TABLE DROP COLUMN must run here too (a fresh DB has machine_binary
|
||||
# with output_dir from 0001). Portable introspection works on both backends.
|
||||
connection = schema_editor.connection
|
||||
with connection.cursor() as cursor:
|
||||
columns = {col.name for col in connection.introspection.get_table_description(cursor, "machine_binary")}
|
||||
|
||||
if "output_dir" not in columns:
|
||||
return
|
||||
|
||||
@ -2,6 +2,12 @@ from django.db import migrations
|
||||
|
||||
|
||||
def add_machine_config_if_missing(apps, schema_editor):
|
||||
# sqlite-only: uses PRAGMA introspection. On a fresh non-sqlite install the
|
||||
# config column already exists (created from 0001 migration state), so there
|
||||
# is nothing to add.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
|
||||
cursor = schema_editor.connection.cursor()
|
||||
cursor.execute("PRAGMA table_info(machine_machine)")
|
||||
columns = {row[1] for row in cursor.fetchall()}
|
||||
|
||||
@ -12,6 +12,7 @@ from pathlib import Path
|
||||
from typing import TextIO
|
||||
from typing import Any
|
||||
import fcntl
|
||||
import os
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from contextlib import contextmanager
|
||||
@ -21,6 +22,213 @@ from archivebox.config import CONSTANTS
|
||||
from archivebox.misc.util import enforce_types
|
||||
|
||||
|
||||
# ============================================================================
|
||||
# Database backend adapter (sqlite / postgresql)
|
||||
# ============================================================================
|
||||
# All sqlite-vs-postgres branching in ArchiveBox is centralized in this
|
||||
# section. Code elsewhere should call these helpers instead of checking
|
||||
# ``connection.vendor``, building ``DATABASES`` entries, or touching
|
||||
# ``CONSTANTS.DATABASE_FILE`` directly.
|
||||
|
||||
|
||||
def is_postgres() -> bool:
|
||||
"""True if DATABASE_ENGINE selects postgres (sqlite is the default)."""
|
||||
from archivebox.config.common import get_config
|
||||
|
||||
return (get_config().DATABASE_ENGINE or "sqlite").strip().lower().startswith("postgres")
|
||||
|
||||
|
||||
def postgres_db_params() -> dict[str, str]:
|
||||
"""Postgres connection params from config (NAME/USER/PASSWORD/HOST/PORT)."""
|
||||
from archivebox.config.common import get_config
|
||||
|
||||
config = get_config()
|
||||
name = config.DATABASE_NAME
|
||||
# DATABASE_NAME defaults to the sqlite file path; that default makes no
|
||||
# sense as a postgres database name, so fall back to 'archivebox'.
|
||||
if name == str(CONSTANTS.DATABASE_FILE) or name.endswith(".sqlite3"):
|
||||
name = "archivebox"
|
||||
return {
|
||||
"NAME": name,
|
||||
"USER": config.DATABASE_USER,
|
||||
"PASSWORD": config.DATABASE_PASSWORD,
|
||||
"HOST": config.DATABASE_HOST,
|
||||
"PORT": str(config.DATABASE_PORT),
|
||||
}
|
||||
|
||||
|
||||
def _psycopg_connect(dbname: str | None = None, connect_timeout: int = 5):
|
||||
import psycopg
|
||||
|
||||
params = postgres_db_params()
|
||||
return psycopg.connect(
|
||||
dbname=dbname or params["NAME"],
|
||||
user=params["USER"],
|
||||
password=params["PASSWORD"] or None,
|
||||
host=params["HOST"],
|
||||
port=params["PORT"],
|
||||
connect_timeout=connect_timeout,
|
||||
)
|
||||
|
||||
|
||||
def database_exists() -> bool:
|
||||
"""True if this collection's database has been initialized.
|
||||
|
||||
sqlite: the index.sqlite3 file exists on disk.
|
||||
postgres: the configured database is reachable and contains the
|
||||
django_migrations table. Safe to call before Django is set up.
|
||||
"""
|
||||
if not is_postgres():
|
||||
return os.path.isfile(CONSTANTS.DATABASE_FILE)
|
||||
try:
|
||||
with _psycopg_connect() as conn:
|
||||
row = conn.execute("SELECT to_regclass('django_migrations')").fetchone()
|
||||
return bool(row and row[0])
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
|
||||
def database_display_location() -> str:
|
||||
"""Human-readable location of the database (file path or postgres DSN)."""
|
||||
if not is_postgres():
|
||||
return str(CONSTANTS.DATABASE_FILE)
|
||||
params = postgres_db_params()
|
||||
return f"postgresql://{params['USER']}@{params['HOST']}:{params['PORT']}/{params['NAME']}"
|
||||
|
||||
|
||||
def ensure_database_ready() -> None:
|
||||
"""Make sure a database server is reachable before running migrations.
|
||||
|
||||
sqlite: no-op (the file is created on first connection).
|
||||
postgres: verify the server accepts connections and create the configured
|
||||
database if it does not exist yet. Raises SystemExit with a helpful
|
||||
message if the server is unreachable.
|
||||
"""
|
||||
if not is_postgres():
|
||||
return
|
||||
|
||||
import psycopg
|
||||
from rich import print as rich_print
|
||||
|
||||
params = postgres_db_params()
|
||||
try:
|
||||
with _psycopg_connect():
|
||||
return
|
||||
except psycopg.OperationalError as err:
|
||||
# 3D000 invalid_catalog_name: server is up but the database is missing
|
||||
if getattr(err, "sqlstate", None) != "3D000" and "does not exist" not in str(err):
|
||||
rich_print(f"[red][X] Error: Unable to connect to PostgreSQL at {database_display_location()}[/red]")
|
||||
rich_print(f" {err}")
|
||||
rich_print(" [violet]Hint:[/violet] Check ARCHIVEBOX_DATABASE_HOST/PORT/USER/PASSWORD and that the server is running.")
|
||||
raise SystemExit(4) from err
|
||||
|
||||
with _psycopg_connect(dbname="postgres") as conn:
|
||||
conn.autocommit = True
|
||||
safe_name = params["NAME"].replace('"', '""')
|
||||
conn.execute(f'CREATE DATABASE "{safe_name}"')
|
||||
rich_print(f" + Created PostgreSQL database {params['NAME']}")
|
||||
|
||||
|
||||
def approximate_row_counts(connection) -> dict[str, int]:
|
||||
"""Cheap per-table approximate row counts from the backend's optimizer stats.
|
||||
|
||||
sqlite: reads sqlite_stat1 (populated by ANALYZE).
|
||||
postgres: reads pg_class.reltuples (maintained by autovacuum/ANALYZE).
|
||||
Returns {} on any failure; tables never analyzed may be absent.
|
||||
"""
|
||||
counts: dict[str, int] = {}
|
||||
try:
|
||||
with connection.cursor() as cursor:
|
||||
if connection.vendor == "sqlite":
|
||||
cursor.execute("SELECT tbl, stat FROM sqlite_stat1")
|
||||
for table, stat in cursor.fetchall():
|
||||
try:
|
||||
counts[str(table)] = int(str(stat).split()[0])
|
||||
except (IndexError, TypeError, ValueError):
|
||||
continue
|
||||
elif connection.vendor == "postgresql":
|
||||
cursor.execute(
|
||||
"""
|
||||
SELECT c.relname, c.reltuples::bigint
|
||||
FROM pg_class c
|
||||
JOIN pg_namespace n ON n.oid = c.relnamespace
|
||||
WHERE c.relkind = 'r'
|
||||
AND n.nspname = current_schema()
|
||||
AND c.reltuples >= 0
|
||||
""",
|
||||
)
|
||||
counts = {str(table): int(estimate) for table, estimate in cursor.fetchall()}
|
||||
except Exception:
|
||||
return {}
|
||||
return counts
|
||||
|
||||
|
||||
def truncate_overlong_charfields(instance=None, **kwargs) -> None:
|
||||
"""Clamp a model instance's CharField values to their declared max_length.
|
||||
|
||||
SQLite never enforces VARCHAR(n) limits, so ArchiveBox has always stored
|
||||
overlong values (e.g. long crawl labels or page titles) untruncated.
|
||||
PostgreSQL enforces them and would raise DataError on save instead.
|
||||
Truncating keeps writes succeeding identically on both backends.
|
||||
|
||||
Dual-use: works as a ``pre_save`` receiver (Django passes ``instance=`` and
|
||||
``sender=`` as kwargs; registered in ``CoreConfig.ready()``) and as a plain
|
||||
``truncate_overlong_charfields(obj)`` call for ``bulk_create`` paths, which
|
||||
bypass signals.
|
||||
"""
|
||||
from django.db import models as dj_models
|
||||
|
||||
if instance is None:
|
||||
return
|
||||
for field in instance._meta.local_concrete_fields:
|
||||
if isinstance(field, dj_models.CharField) and field.max_length:
|
||||
value = getattr(instance, field.attname, None)
|
||||
if isinstance(value, str) and len(value) > field.max_length:
|
||||
setattr(instance, field.attname, value[: field.max_length])
|
||||
|
||||
|
||||
# --- migration helpers ------------------------------------------------------
|
||||
|
||||
|
||||
def rebuild_models_from_migration_state(apps, schema_editor, app_label: str, model_names: list[str]) -> None:
|
||||
"""(non-sqlite only) Drop and recreate the given models' tables from the
|
||||
current migration state.
|
||||
|
||||
ArchiveBox's historical sqlite migrations rebuild tables with raw SQL that
|
||||
intentionally diverges from Django migration state (state-only AddFields
|
||||
reconciled by later sqlite rebuilds). Postgres support postdates all of
|
||||
them, so a non-sqlite database can never contain legacy data at these
|
||||
points in history: every affected table is empty, and dropping + recreating
|
||||
it from state is always equivalent, keeping the real schema in lockstep
|
||||
with migration state at each divergence point. No-op on sqlite.
|
||||
"""
|
||||
if schema_editor.connection.vendor == "sqlite":
|
||||
return
|
||||
existing_tables = set(schema_editor.connection.introspection.table_names())
|
||||
models = [apps.get_model(app_label, model_name) for model_name in model_names]
|
||||
for model in models:
|
||||
if model._meta.db_table in existing_tables:
|
||||
schema_editor.delete_model(model)
|
||||
for model in models:
|
||||
schema_editor.create_model(model)
|
||||
|
||||
|
||||
def drop_models_on_postgres(apps, schema_editor, app_label: str, model_names: list[str]) -> None:
|
||||
"""Reverse companion to ``rebuild_models_from_migration_state``.
|
||||
|
||||
Drops the given models' tables on non-sqlite backends (in the order given,
|
||||
so callers pass reverse-dependency order). No-op on sqlite, whose reverse is
|
||||
handled by the gated ``RunSQL`` reverse_sql instead.
|
||||
"""
|
||||
if schema_editor.connection.vendor == "sqlite":
|
||||
return
|
||||
existing_tables = set(schema_editor.connection.introspection.table_names())
|
||||
for model_name in model_names:
|
||||
model = apps.get_model(app_label, model_name)
|
||||
if model._meta.db_table in existing_tables:
|
||||
schema_editor.delete_model(model)
|
||||
|
||||
|
||||
def run_db_analyze_batch(
|
||||
remaining: list[str] | None,
|
||||
*,
|
||||
@ -357,7 +565,8 @@ def migration_state(out_dir: Path = CONSTANTS.DATA_DIR) -> tuple[list[str], list
|
||||
try:
|
||||
cursor.execute("SELECT app, name FROM django_migrations")
|
||||
except Exception as err:
|
||||
if "no such table" in str(err).lower():
|
||||
msg = str(err).lower()
|
||||
if "no such table" in msg or ("relation" in msg and "does not exist" in msg):
|
||||
return set()
|
||||
raise
|
||||
return {(str(app), str(name)) for app, name in cursor.fetchall()}
|
||||
|
||||
@ -410,7 +410,8 @@ def printable_folder_status(name: str, folder: dict) -> str:
|
||||
else:
|
||||
color, symbol, note, num_files = "grey53", "-", "unused", "-"
|
||||
|
||||
if folder["path"]:
|
||||
if folder["path"] and "://" not in str(folder["path"]):
|
||||
# file-count probing only makes sense for filesystem paths, not DSNs
|
||||
if os.access(folder["path"], os.R_OK):
|
||||
try:
|
||||
num_files = (
|
||||
|
||||
@ -92,6 +92,12 @@ def _ensure_permissions_column(cursor):
|
||||
|
||||
|
||||
def hydrate_persona_permissions(apps, schema_editor):
|
||||
# sqlite-only legacy hydration + generated-column repair (PRAGMA
|
||||
# table_xinfo / VIRTUAL generated column). On postgres the ``permissions``
|
||||
# column is created portably by 0003 (AddField GeneratedField) and there is
|
||||
# never legacy data to hydrate, so this is a no-op there.
|
||||
if schema_editor.connection.vendor != "sqlite":
|
||||
return
|
||||
Persona = apps.get_model("personas", "Persona")
|
||||
base_config = raw_base_config(apps)
|
||||
default_permissions = resolve_permissions(base_config, "public")
|
||||
|
||||
@ -16,31 +16,48 @@ MAX_SEARCH_RANK_IDS = 500
|
||||
|
||||
|
||||
def escape_like_query(query: str) -> str:
|
||||
"""Escape a string for SQLite LIKE matching."""
|
||||
"""Escape a string for SQL LIKE matching (used with ESCAPE '\\')."""
|
||||
return query.replace("\\", "\\\\").replace("%", "\\%").replace("_", "\\_")
|
||||
|
||||
|
||||
def crawl_config_values_search_wave(query: str) -> Q | None:
|
||||
"""Build a Snapshot Q predicate matching values inside Crawl.config."""
|
||||
if connection.vendor != "sqlite":
|
||||
return None
|
||||
|
||||
from archivebox.crawls.models import Crawl
|
||||
|
||||
pattern = f"%{escape_like_query(query).lower()}%"
|
||||
matching_crawls = Crawl.objects.extra(
|
||||
where=[
|
||||
"""
|
||||
EXISTS (
|
||||
SELECT 1
|
||||
FROM json_tree(config)
|
||||
WHERE json_tree.atom IS NOT NULL
|
||||
AND LOWER(CAST(json_tree.atom AS TEXT)) LIKE %s ESCAPE '\\'
|
||||
)
|
||||
""",
|
||||
],
|
||||
params=[pattern],
|
||||
)
|
||||
if connection.vendor == "sqlite":
|
||||
matching_crawls = Crawl.objects.extra(
|
||||
where=[
|
||||
"""
|
||||
EXISTS (
|
||||
SELECT 1
|
||||
FROM json_tree(config)
|
||||
WHERE json_tree.atom IS NOT NULL
|
||||
AND LOWER(CAST(json_tree.atom AS TEXT)) LIKE %s ESCAPE '\\'
|
||||
)
|
||||
""",
|
||||
],
|
||||
params=[pattern],
|
||||
)
|
||||
elif connection.vendor == "postgresql":
|
||||
# Match only scalar config *values*, not keys (mirrors SQLite's
|
||||
# json_tree.atom). jsonb_path_query('$.**') walks every nested node;
|
||||
# keep the non-container leaves and compare their text form.
|
||||
matching_crawls = Crawl.objects.extra(
|
||||
where=[
|
||||
"""
|
||||
EXISTS (
|
||||
SELECT 1
|
||||
FROM jsonb_path_query(config, '$.**') AS leaf
|
||||
WHERE jsonb_typeof(leaf) NOT IN ('object', 'array')
|
||||
AND LOWER(leaf #>> '{}') LIKE %s ESCAPE '\\'
|
||||
)
|
||||
""",
|
||||
],
|
||||
params=[pattern],
|
||||
)
|
||||
else:
|
||||
return None
|
||||
return Q(crawl_id__in=matching_crawls.values("pk"))
|
||||
|
||||
|
||||
|
||||
@ -124,19 +124,32 @@ def iter_url_prefix_search_ids(prefix: str, queryset):
|
||||
table = connection.ops.quote_name(model._meta.db_table)
|
||||
pk_column = connection.ops.quote_name(model._meta.pk.column)
|
||||
url_column = connection.ops.quote_name(model._meta.get_field("url").column)
|
||||
upper_bound = url_prefix_upper_bound(prefix)
|
||||
raw_ids = []
|
||||
|
||||
if connection.vendor == "sqlite":
|
||||
# Bytewise range comparison uses the plain url btree index directly.
|
||||
where_clause = f"{url_column} >= %s AND {url_column} < %s"
|
||||
where_params = [prefix, url_prefix_upper_bound(prefix)]
|
||||
else:
|
||||
# Range comparisons are collation-dependent on postgres (linguistic
|
||||
# collations don't compare bytewise), so use LIKE with escaped
|
||||
# wildcards instead — correct under any collation and able to use the
|
||||
# url pattern-ops index.
|
||||
from archivebox.search.query import escape_like_query
|
||||
|
||||
where_clause = f"{url_column} LIKE %s ESCAPE '\\'"
|
||||
where_params = [f"{escape_like_query(prefix)}%"]
|
||||
|
||||
with connection.cursor() as cursor:
|
||||
cursor.execute(
|
||||
f"""
|
||||
SELECT {pk_column}
|
||||
FROM {table}
|
||||
WHERE {url_column} >= %s AND {url_column} < %s
|
||||
WHERE {where_clause}
|
||||
ORDER BY {url_column}
|
||||
LIMIT %s
|
||||
""",
|
||||
[prefix, upper_bound, URL_PREFIX_SEARCH_LIMIT],
|
||||
[*where_params, URL_PREFIX_SEARCH_LIMIT],
|
||||
)
|
||||
raw_ids = [str(row[0]).replace("-", "") for row in cursor.fetchall()]
|
||||
|
||||
|
||||
228
archivebox/tests/test_postgres_backend.py
Normal file
228
archivebox/tests/test_postgres_backend.py
Normal file
@ -0,0 +1,228 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
End-to-end tests for the PostgreSQL database backend (DATABASE_ENGINE=postgres).
|
||||
|
||||
Spins up a real throwaway PostgreSQL cluster (initdb + pg_ctl) for the module,
|
||||
then exercises real archivebox CLI flows against it: init, status, re-init,
|
||||
add --index-only, list, remove, and a full schema-vs-models parity check.
|
||||
|
||||
Requires PostgreSQL server binaries (initdb/pg_ctl) to be installed.
|
||||
"""
|
||||
|
||||
import os
|
||||
import shlex
|
||||
import shutil
|
||||
import socket
|
||||
import subprocess
|
||||
import tempfile
|
||||
import uuid
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from .conftest import cli_env, run_archivebox_cmd, run_queued_crawls
|
||||
|
||||
|
||||
def _find_pg_bindir() -> Path:
|
||||
"""Locate PostgreSQL server binaries (initdb) on this machine."""
|
||||
initdb_on_path = shutil.which("initdb")
|
||||
if initdb_on_path:
|
||||
return Path(initdb_on_path).resolve().parent
|
||||
candidates = []
|
||||
for base in (Path("/usr/lib/postgresql"), Path("/opt/homebrew/opt"), Path("/usr/local/opt"), Path("/opt/homebrew/Cellar/postgresql")):
|
||||
if base.is_dir():
|
||||
for sub in sorted(base.iterdir(), reverse=True):
|
||||
initdb = sub / "bin" / "initdb"
|
||||
if initdb.is_file():
|
||||
candidates.append(initdb.parent)
|
||||
assert candidates, "PostgreSQL server binaries (initdb) are required for test_postgres_backend tests"
|
||||
return candidates[0]
|
||||
|
||||
|
||||
def _free_port() -> int:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
|
||||
sock.bind(("127.0.0.1", 0))
|
||||
return sock.getsockname()[1]
|
||||
|
||||
|
||||
PG_TEST_USER = "abxtest"
|
||||
|
||||
|
||||
@pytest.fixture(scope="module")
|
||||
def pg_server():
|
||||
"""A real throwaway PostgreSQL server for the whole test module."""
|
||||
bindir = _find_pg_bindir()
|
||||
workdir = Path(tempfile.mkdtemp(prefix="abx-pg-test-"))
|
||||
datadir = workdir / "data"
|
||||
logfile = workdir / "server.log"
|
||||
port = _free_port()
|
||||
|
||||
# postgres refuses to run as root; when the suite runs as root (e.g. in a
|
||||
# dev container), run the server as the system postgres user instead.
|
||||
run_as_postgres_user = os.geteuid() == 0
|
||||
if run_as_postgres_user:
|
||||
workdir.chmod(0o755)
|
||||
shutil.chown(workdir, user="postgres")
|
||||
|
||||
def pg_cmd(args: list[str]) -> subprocess.CompletedProcess:
|
||||
if run_as_postgres_user:
|
||||
full = ["su", "postgres", "-s", "/bin/sh", "-c", shlex.join(args)]
|
||||
else:
|
||||
full = args
|
||||
return subprocess.run(full, capture_output=True, text=True, check=False, timeout=120)
|
||||
|
||||
result = pg_cmd([str(bindir / "initdb"), "-D", str(datadir), "-E", "UTF8", "-A", "trust", "-U", PG_TEST_USER])
|
||||
assert result.returncode == 0, f"initdb failed: {result.stderr}"
|
||||
|
||||
server_opts = f"-p {port} -k {workdir} -c listen_addresses=127.0.0.1 -c fsync=off -c synchronous_commit=off"
|
||||
result = pg_cmd([str(bindir / "pg_ctl"), "-D", str(datadir), "-l", str(logfile), "-o", server_opts, "-w", "start"])
|
||||
assert result.returncode == 0, f"pg_ctl start failed: {result.stderr}\n{logfile.read_text() if logfile.exists() else ''}"
|
||||
|
||||
try:
|
||||
yield {"host": "127.0.0.1", "port": port, "user": PG_TEST_USER}
|
||||
finally:
|
||||
pg_cmd([str(bindir / "pg_ctl"), "-D", str(datadir), "-m", "immediate", "stop"])
|
||||
shutil.rmtree(workdir, ignore_errors=True)
|
||||
|
||||
|
||||
def pg_cli_env(pg_server: dict, dbname: str, **extra) -> dict:
|
||||
env = cli_env(**extra)
|
||||
env.update(
|
||||
{
|
||||
"ARCHIVEBOX_DATABASE_ENGINE": "postgres",
|
||||
"ARCHIVEBOX_DATABASE_HOST": pg_server["host"],
|
||||
"ARCHIVEBOX_DATABASE_PORT": str(pg_server["port"]),
|
||||
"ARCHIVEBOX_DATABASE_USER": pg_server["user"],
|
||||
"ARCHIVEBOX_DATABASE_NAME": dbname,
|
||||
},
|
||||
)
|
||||
return env
|
||||
|
||||
|
||||
def unique_dbname() -> str:
|
||||
return f"abx_test_{uuid.uuid4().hex[:12]}"
|
||||
|
||||
|
||||
def pg_query(pg_server: dict, dbname: str, query: str) -> list[tuple]:
|
||||
import psycopg
|
||||
|
||||
with psycopg.connect(host=pg_server["host"], port=pg_server["port"], user=pg_server["user"], dbname=dbname) as conn:
|
||||
return conn.execute(query).fetchall()
|
||||
|
||||
|
||||
def test_init_creates_postgres_schema(pg_server, tmp_path):
|
||||
"""Fresh init against postgres should create the database + full schema, and no sqlite file."""
|
||||
dbname = unique_dbname()
|
||||
env = pg_cli_env(pg_server, dbname)
|
||||
|
||||
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
|
||||
assert result.returncode == 0, f"Init failed: {result.stderr}\n{result.stdout}"
|
||||
|
||||
assert not (tmp_path / "index.sqlite3").exists(), "sqlite file should not be created when using postgres"
|
||||
assert (tmp_path / "archive").is_dir(), "Archive dir not created"
|
||||
|
||||
(applied_migrations,) = pg_query(pg_server, dbname, "SELECT COUNT(*) FROM django_migrations")[0]
|
||||
assert applied_migrations > 50, f"Expected all migrations applied, got {applied_migrations}"
|
||||
|
||||
for table in ("core_snapshot", "core_archiveresult", "core_tag", "crawls_crawl", "machine_machine", "api_apitoken", "personas_persona"):
|
||||
(regclass,) = pg_query(pg_server, dbname, f"SELECT to_regclass('{table}')")[0]
|
||||
assert regclass == table, f"Table {table} missing from postgres schema"
|
||||
|
||||
|
||||
def test_postgres_schema_matches_models(pg_server, tmp_path):
|
||||
"""Every model column must exist in postgres and vice versa (no state/schema drift)."""
|
||||
dbname = unique_dbname()
|
||||
env = pg_cli_env(pg_server, dbname)
|
||||
|
||||
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
|
||||
assert result.returncode == 0, f"Init failed: {result.stderr}"
|
||||
|
||||
parity_script = (
|
||||
"from django.apps import apps\n"
|
||||
"from django.db import connection\n"
|
||||
"problems = []\n"
|
||||
"with connection.cursor() as cursor:\n"
|
||||
" tables = set(connection.introspection.table_names(cursor))\n"
|
||||
" for model in apps.get_models(include_auto_created=True):\n"
|
||||
" meta = model._meta\n"
|
||||
" if not meta.managed or meta.proxy:\n"
|
||||
" continue\n"
|
||||
" if meta.db_table not in tables:\n"
|
||||
" problems.append(f'missing table {meta.db_table}')\n"
|
||||
" continue\n"
|
||||
" db_cols = {col.name for col in connection.introspection.get_table_description(cursor, meta.db_table)}\n"
|
||||
" model_cols = {field.column for field in meta.local_concrete_fields}\n"
|
||||
" for col in sorted(model_cols - db_cols):\n"
|
||||
" problems.append(f'{meta.db_table}: missing column {col}')\n"
|
||||
" for col in sorted(db_cols - model_cols):\n"
|
||||
" problems.append(f'{meta.db_table}: extra column {col}')\n"
|
||||
"print('SCHEMA_PROBLEMS=' + repr(sorted(problems)))\n"
|
||||
)
|
||||
result = run_archivebox_cmd(["manage", "shell", "-c", parity_script], cwd=tmp_path, env=env, timeout=120)
|
||||
assert result.returncode == 0, f"manage shell failed: {result.stderr}"
|
||||
assert "SCHEMA_PROBLEMS=[]" in result.stdout, f"Postgres schema diverges from models:\n{result.stdout}\n{result.stderr}"
|
||||
|
||||
result = run_archivebox_cmd(["manage", "makemigrations", "--check", "--dry-run"], cwd=tmp_path, env=env, timeout=120)
|
||||
assert result.returncode == 0, f"Model state does not match migrations: {result.stdout}\n{result.stderr}"
|
||||
|
||||
|
||||
def test_status_and_reinit_on_postgres(pg_server, tmp_path):
|
||||
"""status works against postgres, and a second init takes the 'verify existing' path."""
|
||||
dbname = unique_dbname()
|
||||
env = pg_cli_env(pg_server, dbname)
|
||||
|
||||
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
|
||||
assert result.returncode == 0, f"Init failed: {result.stderr}"
|
||||
assert "Initializing a new ArchiveBox" in result.stdout
|
||||
|
||||
result = run_archivebox_cmd(["status"], cwd=tmp_path, env=env, timeout=120)
|
||||
assert result.returncode == 0, f"Status failed: {result.stderr}"
|
||||
assert f"postgresql://{PG_TEST_USER}@" in result.stdout, f"status should show the postgres DSN:\n{result.stdout}"
|
||||
|
||||
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
|
||||
assert result.returncode == 0, f"Re-init failed: {result.stderr}"
|
||||
assert "Verifying and updating existing ArchiveBox collection" in result.stdout
|
||||
|
||||
|
||||
def test_add_list_remove_on_postgres(pg_server, tmp_path):
|
||||
"""Real add/list/remove CLI flows store and retrieve rows from postgres."""
|
||||
dbname = unique_dbname()
|
||||
env = pg_cli_env(pg_server, dbname, disable_extractors=True)
|
||||
test_url = "https://example.com/abx-postgres-test"
|
||||
|
||||
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=env, timeout=300)
|
||||
assert result.returncode == 0, f"Init failed: {result.stderr}"
|
||||
|
||||
result = run_archivebox_cmd(["add", "--index-only", test_url], cwd=tmp_path, env=env, timeout=300)
|
||||
assert result.returncode == 0, f"Add failed: {result.stderr}\n{result.stdout}"
|
||||
run_queued_crawls(tmp_path, env)
|
||||
|
||||
(snapshot_count,) = pg_query(pg_server, dbname, "SELECT COUNT(*) FROM core_snapshot")[0]
|
||||
assert snapshot_count >= 1, "Snapshot row not written to postgres"
|
||||
(crawl_count,) = pg_query(pg_server, dbname, "SELECT COUNT(*) FROM crawls_crawl")[0]
|
||||
assert crawl_count >= 1, "Crawl row not written to postgres"
|
||||
|
||||
result = run_archivebox_cmd(["list"], cwd=tmp_path, env=env, timeout=120)
|
||||
assert result.returncode == 0, f"List failed: {result.stderr}"
|
||||
assert "example.com/abx-postgres-test" in result.stdout, f"Added URL missing from list output:\n{result.stdout}"
|
||||
|
||||
result = run_archivebox_cmd(["search", "abx-postgres-test"], cwd=tmp_path, env=env, timeout=120)
|
||||
assert result.returncode == 0, f"Search failed: {result.stderr}"
|
||||
assert "example.com/abx-postgres-test" in result.stdout, f"Added URL missing from search output:\n{result.stdout}"
|
||||
|
||||
result = run_archivebox_cmd(
|
||||
["remove", "--yes", "--delete", "--filter-type=exact", test_url],
|
||||
cwd=tmp_path,
|
||||
env=env,
|
||||
timeout=120,
|
||||
)
|
||||
assert result.returncode == 0, f"Remove failed: {result.stderr}\n{result.stdout}"
|
||||
(snapshot_count,) = pg_query(pg_server, dbname, f"SELECT COUNT(*) FROM core_snapshot WHERE url = '{test_url}'")[0]
|
||||
assert snapshot_count == 0, "Snapshot row should be deleted from postgres"
|
||||
|
||||
|
||||
def test_sqlite_remains_the_default(tmp_path):
|
||||
"""Without DATABASE_ENGINE config, init keeps using the sqlite file backend."""
|
||||
result = run_archivebox_cmd(["init", "--quick"], cwd=tmp_path, env=cli_env(), timeout=300)
|
||||
assert result.returncode == 0, f"Init failed: {result.stderr}"
|
||||
assert (tmp_path / "index.sqlite3").exists(), "sqlite file should be created by default"
|
||||
202
bin/benchmark_db_backends.py
Executable file
202
bin/benchmark_db_backends.py
Executable file
@ -0,0 +1,202 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Benchmark ArchiveBox hot-path index queries at large row counts.
|
||||
|
||||
Run from inside an initialized (empty) collection directory, with the same
|
||||
ARCHIVEBOX_DATABASE_* env vars the collection was initialized with:
|
||||
|
||||
cd /path/to/collection
|
||||
uv run --project /path/to/ArchiveBox python /path/to/ArchiveBox/bin/benchmark_db_backends.py --rows 1000000
|
||||
|
||||
Seeds N snapshots (+1 archiveresult each) via bulk_create, runs ANALYZE, then
|
||||
times the hot queries used by the admin UI, snapshot detail views, URL prefix
|
||||
search, and the worker queue/claim paths. Works on both sqlite and postgres —
|
||||
use it to compare backends or to catch performance regressions.
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import statistics
|
||||
import sys
|
||||
import time
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
|
||||
def seed(rows: int, batch_size: int = 20_000) -> None:
|
||||
from django.db import transaction
|
||||
|
||||
from archivebox.base_models.models import get_or_create_system_user_pk
|
||||
from archivebox.core.models import ArchiveResult, Snapshot
|
||||
from archivebox.crawls.models import Crawl
|
||||
from archivebox.uuid_compat import uuid7
|
||||
|
||||
user_pk = get_or_create_system_user_pk()
|
||||
crawl = Crawl.objects.create(urls="https://example.com/", created_by_id=user_pk, status="sealed", label="benchmark seed")
|
||||
|
||||
existing = Snapshot.objects.count()
|
||||
if existing >= rows:
|
||||
print(f"already seeded ({existing} snapshots)", flush=True)
|
||||
return
|
||||
|
||||
base_time = datetime(2020, 1, 1, tzinfo=timezone.utc)
|
||||
started = time.monotonic()
|
||||
for start in range(existing, rows, batch_size):
|
||||
end = min(start + batch_size, rows)
|
||||
snapshots = []
|
||||
for i in range(start, end):
|
||||
created = base_time + timedelta(seconds=i)
|
||||
snapshots.append(
|
||||
Snapshot(
|
||||
id=uuid7(),
|
||||
url=f"https://site{i % 1000}.example.org/path/{i}" + ("#section" if i % 7 == 0 else ""),
|
||||
timestamp=f"{1577836800 + i}.{i}",
|
||||
title=f"Benchmark page {i}",
|
||||
crawl=crawl,
|
||||
bookmarked_at=created,
|
||||
created_at=created,
|
||||
modified_at=created,
|
||||
downloaded_at=created if i % 10 else None,
|
||||
status="sealed" if i % 20 else "queued",
|
||||
retry_at=None if i % 20 else created,
|
||||
fs_version="0.9.0",
|
||||
config={},
|
||||
depth=0,
|
||||
),
|
||||
)
|
||||
with transaction.atomic():
|
||||
Snapshot.objects.bulk_create(snapshots, batch_size=batch_size)
|
||||
ArchiveResult.objects.bulk_create(
|
||||
[
|
||||
ArchiveResult(
|
||||
id=uuid7(),
|
||||
snapshot=snapshot,
|
||||
plugin="wget",
|
||||
hook_name="on_Snapshot__06_wget.py",
|
||||
status="succeeded",
|
||||
created_at=snapshot.created_at,
|
||||
modified_at=snapshot.created_at,
|
||||
start_ts=snapshot.created_at,
|
||||
end_ts=snapshot.created_at,
|
||||
output_str="benchmark",
|
||||
output_files={},
|
||||
)
|
||||
for snapshot in snapshots
|
||||
],
|
||||
batch_size=batch_size,
|
||||
)
|
||||
if (end // batch_size) % 5 == 0 or end == rows:
|
||||
rate = (end - existing) / max(time.monotonic() - started, 0.001)
|
||||
print(f" seeded {end}/{rows} snapshots ({rate:,.0f} rows/s)", flush=True)
|
||||
|
||||
|
||||
def analyze_tables() -> None:
|
||||
from django.db import connection
|
||||
|
||||
with connection.cursor() as cursor:
|
||||
for table in ("core_snapshot", "core_archiveresult"):
|
||||
cursor.execute(f"ANALYZE {table}")
|
||||
|
||||
|
||||
def timed(func, repeat: int = 5) -> tuple[float, object]:
|
||||
times = []
|
||||
result = None
|
||||
for _ in range(repeat):
|
||||
started = time.perf_counter()
|
||||
result = func()
|
||||
times.append((time.perf_counter() - started) * 1000)
|
||||
return statistics.median(times), result
|
||||
|
||||
|
||||
def run_benchmarks(rows: int) -> dict[str, float]:
|
||||
from django.db import connection
|
||||
from django.db.models import Count, Q
|
||||
from django.utils import timezone as dj_timezone
|
||||
|
||||
from archivebox.core.models import ArchiveResult, Snapshot
|
||||
from archivebox.misc.db import approximate_row_counts
|
||||
from archivebox.search.views import iter_url_prefix_search_ids
|
||||
|
||||
now = dj_timezone.now()
|
||||
results: dict[str, float] = {}
|
||||
target_i = rows // 2
|
||||
target_url = f"https://site{target_i % 1000}.example.org/path/{target_i}"
|
||||
|
||||
def fragmentless_q(url: str) -> Q:
|
||||
if connection.vendor == "sqlite":
|
||||
return Q(url=url) | (Q(url__gte=f"{url}#") & Q(url__lt=f"{url}#\U0010ffff"))
|
||||
return Q(url=url) | Q(url__startswith=f"{url}#")
|
||||
|
||||
benchmarks = {
|
||||
"exact_count": lambda: Snapshot.objects.count(),
|
||||
"approximate_row_counts": lambda: approximate_row_counts(connection),
|
||||
"admin_list_page": lambda: list(
|
||||
Snapshot.objects.order_by("-bookmarked_at").values("id", "url", "title", "status", "bookmarked_at")[:40],
|
||||
),
|
||||
"admin_list_page_offset_10k": lambda: list(Snapshot.objects.order_by("-bookmarked_at").values("id", "url", "title")[10_000:10_040]),
|
||||
"snapshot_detail_by_url": lambda: list(Snapshot.objects.filter(fragmentless_q(target_url))[:10]),
|
||||
"snapshot_detail_archiveresults": lambda: list(
|
||||
ArchiveResult.objects.filter(snapshot__url=target_url).order_by("start_ts").values("id", "plugin", "status")[:100],
|
||||
),
|
||||
"url_prefix_search": lambda: list(iter_url_prefix_search_ids("https://site500.example.org/", Snapshot.objects.all())),
|
||||
"worker_queue_scan": lambda: list(
|
||||
Snapshot.objects.filter(status="queued", retry_at__lte=now)
|
||||
.order_by("retry_at", "created_at")
|
||||
.values_list("id", flat=True)[:100],
|
||||
),
|
||||
"status_facet_counts": lambda: dict(Snapshot.objects.values_list("status").annotate(n=Count("id")).values_list("status", "n")),
|
||||
"tag_join_filter": lambda: list(Snapshot.objects.filter(title__icontains="page 4242").values("id")[:20]),
|
||||
}
|
||||
|
||||
for name, func in benchmarks.items():
|
||||
median_ms, _ = timed(func)
|
||||
results[name] = round(median_ms, 2)
|
||||
print(f" {name:35s} {median_ms:10.2f} ms", flush=True)
|
||||
|
||||
def claim_one() -> int:
|
||||
snapshot = Snapshot.objects.filter(status="queued").order_by("retry_at").first()
|
||||
if snapshot is None:
|
||||
return 0
|
||||
return Snapshot.objects.filter(pk=snapshot.pk, retry_at=snapshot.retry_at).update(
|
||||
retry_at=now + timedelta(seconds=60),
|
||||
modified_at=now,
|
||||
)
|
||||
|
||||
median_ms, _ = timed(claim_one)
|
||||
results["worker_cas_claim"] = round(median_ms, 2)
|
||||
print(f" {'worker_cas_claim':35s} {median_ms:10.2f} ms", flush=True)
|
||||
return results
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description=__doc__)
|
||||
parser.add_argument("--rows", type=int, default=1_000_000)
|
||||
parser.add_argument("--seed-only", action="store_true")
|
||||
parser.add_argument("--json", dest="json_path", default=None, help="also write results to this JSON file")
|
||||
args = parser.parse_args()
|
||||
|
||||
from archivebox.config.django import setup_django
|
||||
from archivebox.misc.db import database_exists
|
||||
|
||||
setup_django()
|
||||
assert database_exists(), "run archivebox init in this directory first"
|
||||
from django.db import connection
|
||||
|
||||
print(f"backend: {connection.vendor}, target rows: {args.rows}", flush=True)
|
||||
print("seeding...", flush=True)
|
||||
seed(args.rows)
|
||||
if args.seed_only:
|
||||
return
|
||||
print("running ANALYZE...", flush=True)
|
||||
analyze_tables()
|
||||
print("benchmarks (median of 5):", flush=True)
|
||||
results = run_benchmarks(args.rows)
|
||||
results["_backend"] = connection.vendor
|
||||
results["_rows"] = args.rows
|
||||
if args.json_path:
|
||||
with open(args.json_path, "w", encoding="utf-8") as f:
|
||||
json.dump(results, f, indent=2)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
|
||||
main()
|
||||
@ -645,9 +645,9 @@ Skip the startup check that verifies [`TMP_DIR`](#tmp_dir) can host unix-domain
|
||||
|
||||
## Database Settings
|
||||
|
||||
*Options for tuning the SQLite index database that backs ArchiveBox's snapshot, tag, and crawl metadata.*
|
||||
*Options for choosing and tuning the index database that backs ArchiveBox's snapshot, tag, and crawl metadata.*
|
||||
|
||||
ArchiveBox stores all of its index metadata in a single SQLite database file (`index.sqlite3` inside your data directory). The defaults are tuned for nearly all users — the knobs below mostly govern **lock-contention behavior**, which matters when multiple workers touch the database concurrently (e.g. supervised orchestrators, parallel `archivebox add` runs, container restarts that race against an in-flight write, or long-running web/admin processes alongside CLI commands).
|
||||
ArchiveBox stores all of its index metadata in a single SQLite database file (`index.sqlite3` inside your data directory) by default, or optionally in a PostgreSQL database (see [`DATABASE_ENGINE`](#database_engine)). The defaults are tuned for nearly all users — the knobs below mostly govern **lock-contention behavior**, which matters when multiple workers touch the database concurrently (e.g. supervised orchestrators, parallel `archivebox add` runs, container restarts that race against an in-flight write, or long-running web/admin processes alongside CLI commands).
|
||||
|
||||
> [!NOTE]
|
||||
> These are advanced operator tuning options. If you are not actively diagnosing `database is locked` errors or planning a non-default storage layout, you can safely leave everything in this section at its default.
|
||||
@ -657,9 +657,59 @@ ArchiveBox stores all of its index metadata in a single SQLite database file (`i
|
||||
- https://www.sqlite.org/wal.html
|
||||
- https://www.sqlite.org/pragma.html
|
||||
|
||||
---
|
||||
<a id="database_engine"></a>
|
||||
<a id="archivebox_database_engine"></a>
|
||||
#### `DATABASE_ENGINE`
|
||||
**Possible Values:** [`sqlite`]/`postgres`
|
||||
Which database backend to use for the main index. Settable as `ARCHIVEBOX_DATABASE_ENGINE` or under `[DATABASE_CONFIG]` in `ArchiveBox.conf`.
|
||||
|
||||
The default `sqlite` keeps everything in a single `index.sqlite3` file inside the data directory and requires no external services. Set this to `postgres` to store the index in a PostgreSQL database instead — useful for large collections with many concurrent writers, or when the data directory lives on a filesystem where SQLite performs poorly (e.g. network mounts).
|
||||
|
||||
With `postgres`, connection details come from the options below. `archivebox init` will create the configured database automatically if the server is reachable and the database does not exist yet. Only the *index* moves to PostgreSQL — snapshot output files stay in `./archive/` in the data directory, and plugin-owned sidecar databases (e.g. the `search.sqlite3` full-text index) are unaffected.
|
||||
|
||||
```ini
|
||||
# example ArchiveBox.conf
|
||||
[DATABASE_CONFIG]
|
||||
DATABASE_ENGINE = postgres
|
||||
DATABASE_NAME = archivebox
|
||||
DATABASE_HOST = 127.0.0.1
|
||||
DATABASE_PORT = 5432
|
||||
DATABASE_USER = archivebox
|
||||
DATABASE_PASSWORD = s3cret
|
||||
```
|
||||
|
||||
> [!WARNING]
|
||||
> Pick a backend when you first run `archivebox init` and stick with it. There is no built-in tool (yet) to move an existing collection's index between SQLite and PostgreSQL.
|
||||
|
||||
*Related options:*
|
||||
[`DATABASE_NAME`](#database_name), [`DATABASE_HOST`](#database_host)
|
||||
|
||||
---
|
||||
<a id="database_host"></a>
|
||||
<a id="archivebox_database_host"></a>
|
||||
<a id="database_port"></a>
|
||||
<a id="archivebox_database_port"></a>
|
||||
<a id="database_user"></a>
|
||||
<a id="archivebox_database_user"></a>
|
||||
<a id="database_password"></a>
|
||||
<a id="archivebox_database_password"></a>
|
||||
#### `DATABASE_HOST` / `DATABASE_PORT` / `DATABASE_USER` / `DATABASE_PASSWORD`
|
||||
**Possible Values:** [`127.0.0.1`] / [`5432`] / [`archivebox`] / [empty]
|
||||
PostgreSQL connection settings, used only when [`DATABASE_ENGINE`](#database_engine)`=postgres`. Settable as `ARCHIVEBOX_DATABASE_HOST`, `ARCHIVEBOX_DATABASE_PORT`, `ARCHIVEBOX_DATABASE_USER`, and `ARCHIVEBOX_DATABASE_PASSWORD`.
|
||||
|
||||
`DATABASE_HOST` may also be a path to a directory containing a PostgreSQL unix socket (e.g. `/var/run/postgresql`).
|
||||
|
||||
---
|
||||
<a id="database_name"></a>
|
||||
<a id="archivebox_database_name"></a>
|
||||
#### `DATABASE_NAME`
|
||||
**Possible Values:** [`index.sqlite3`] / `archivebox` / ...
|
||||
The main index database. Settable as `ARCHIVEBOX_DATABASE_NAME`.
|
||||
|
||||
With the default [`DATABASE_ENGINE`](#database_engine)`=sqlite`, this is the path to the SQLite index file inside the data directory (`index.sqlite3`). With `DATABASE_ENGINE=postgres`, it is the name of the PostgreSQL database instead (default: `archivebox`), created automatically by `archivebox init` if it does not exist.
|
||||
|
||||
---
|
||||
#### `SQLITE_JOURNAL_MODE`
|
||||
**Possible Values:** [`WAL`]/`DELETE`/`TRUNCATE`/`PERSIST`/`MEMORY`/`OFF`
|
||||
SQLite [journal mode](https://www.sqlite.org/pragma.html#pragma_journal_mode), applied via `PRAGMA journal_mode = ...` on every new connection. Settable as `ARCHIVEBOX_SQLITE_JOURNAL_MODE`.
|
||||
|
||||
@ -92,6 +92,7 @@ version = 2
|
||||
"7890998169008ad0-1" = "illustration"
|
||||
|
||||
# docs/Configuration.md
|
||||
"48b48da014bd91dd-1" = "illustration"
|
||||
"129eb9ca56f17500-1" = "illustration"
|
||||
"798793d9d6b2e12e-1" = "run"
|
||||
"3fecc495abdc20ef-1" = "illustration"
|
||||
|
||||
@ -41,6 +41,7 @@ dependencies = [
|
||||
### Django libraries
|
||||
"setuptools>=74.1.0", # for: django 5 on python >=3.12, distutils is no longer in stdlib but django 5.1 expects distutils (TODO: check if this can be removed eventually)
|
||||
"django>=6.0",
|
||||
"psycopg[binary]>=3.2", # for: PostgreSQL database backend (ARCHIVEBOX_DATABASE_ENGINE=postgres)
|
||||
"daphne>=4.2.1", # ASGI server for Django (no channels needed - websockets not used)
|
||||
"django-ninja>=1.5.1",
|
||||
"django-extensions>=3.2.3",
|
||||
|
||||
47
uv.lock
47
uv.lock
@ -13,11 +13,11 @@ supported-markers = [
|
||||
]
|
||||
|
||||
[options]
|
||||
exclude-newer = "2026-07-18T21:19:18.486818Z"
|
||||
exclude-newer = "2026-07-19T10:29:25.327679866Z"
|
||||
exclude-newer-span = "P5D"
|
||||
|
||||
[options.exclude-newer-package]
|
||||
abxbus = { timestamp = "2026-07-23T21:19:17.486835Z", span = "PT1S" }
|
||||
abxbus = { timestamp = "2026-07-24T10:29:24.328016284Z", span = "PT1S" }
|
||||
abx-plugins = "2100-01-01T00:00:00Z"
|
||||
abx-dl = "2100-01-01T00:00:00Z"
|
||||
abxpkg = "2100-01-01T00:00:00Z"
|
||||
@ -149,6 +149,7 @@ dependencies = [
|
||||
{ name = "ipython", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
{ name = "platformdirs", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
{ name = "psutil", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
{ name = "psycopg", extra = ["binary"], marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
{ name = "py-machineid", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
{ name = "pydantic", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
{ name = "pydantic-settings", marker = "sys_platform == 'darwin' or sys_platform == 'linux'" },
|
||||
@ -251,6 +252,7 @@ requires-dist = [
|
||||
{ name = "ipython", specifier = ">=8.27.0" },
|
||||
{ name = "platformdirs", specifier = ">=4.3.6" },
|
||||
{ name = "psutil", specifier = ">=6.0.0" },
|
||||
{ name = "psycopg", extras = ["binary"], specifier = ">=3.2" },
|
||||
{ name = "py-machineid", specifier = ">=0.6.0" },
|
||||
{ name = "pydantic", specifier = ">=2.8.0" },
|
||||
{ name = "pydantic-settings", specifier = ">=2.5.2" },
|
||||
@ -1695,6 +1697,47 @@ wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/04/78/0acd37ca84ce3ddffaa92ef0f571e073faa6d8ff1f0559ab1272188ea2be/psutil-7.2.2-cp36-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:b58fabe35e80b264a4e3bb23e6b96f9e45a3df7fb7eed419ac0e5947c61e47cc", size = 148266, upload-time = "2026-01-28T18:15:31.597Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "psycopg"
|
||||
version = "3.3.4"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/db/2f/cb91e5502ec9de1de6f1b76cfbf69531932725361168bb06963620c77e2e/psycopg-3.3.4.tar.gz", hash = "sha256:e21207764952cff81b6b8bdacad9a3939f2793367fdac2987b3aac36a651b5bc", size = 165799, upload-time = "2026-05-01T23:31:55.179Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/5c/e0/7b3dee031daae7743609ce3c746565d4a3ed7c2c186479eb48e34e838c64/psycopg-3.3.4-py3-none-any.whl", hash = "sha256:b6bbc25ccf05c8fad3b061d9db2ef0909a555171b84b07f29458a447253d679a", size = 213001, upload-time = "2026-05-01T23:20:50.816Z" },
|
||||
]
|
||||
|
||||
[package.optional-dependencies]
|
||||
binary = [
|
||||
{ name = "psycopg-binary", marker = "(implementation_name != 'pypy' and sys_platform == 'darwin') or (implementation_name != 'pypy' and sys_platform == 'linux')" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "psycopg-binary"
|
||||
version = "3.3.4"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/09/43/13e9c406fbbf354580476e248a16b64802a376873ebe6339e30bb655572d/psycopg_binary-3.3.4-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:fbd1d4ed566895ad2d3bf4ddfd8bae90026930ddf29df3b9d91d32c8c47866a7", size = 4590377, upload-time = "2026-05-01T23:29:18.782Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/22/be/2923cd7c3683e7afdecf4f10796a18de02f5c5ddc0969aa2ad0a8cdd3bbd/psycopg_binary-3.3.4-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:75a9067e236f9b9ae3535b66fe99bddb33d39c0de10112e49b9ab11eee53dc31", size = 4669023, upload-time = "2026-05-01T23:29:25.884Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/96/a0/2c913d6fe13d6a8bd13597d36739bf47af063ad9399e402cfecab16f3c1e/psycopg_binary-3.3.4-cp313-cp313-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:b56b603ebcea8aa10b46228b8410ba7f13e7c2ee54389d4d9be0927fd8ce2a70", size = 5467423, upload-time = "2026-05-01T23:29:33.416Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/e7/38/205d10bc1ad0df4a21c5c51659126bd3ea0ef98fcad1e852f78c249bb9c3/psycopg_binary-3.3.4-cp313-cp313-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:c677c4ad433cb7150c8cd304a0769ae3bcfbe5ea0676eb53faa7b1443b16d0d3", size = 5151137, upload-time = "2026-05-01T23:29:42.013Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/36/fc/f0381ddcd45eff3bb70dbca6823a996048d7f507b2ec3fc92c6fabc0fe87/psycopg_binary-3.3.4-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:26df2717e59c0473e4465a97dfb1b7afebaa479277870fd5784d1436470db47c", size = 6736671, upload-time = "2026-05-01T23:29:51.626Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/95/40/fa545ae152c24327651e5624e4902121e808270be36c10b12e9939be09bc/psycopg_binary-3.3.4-cp313-cp313-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:1dc1f79fd16bb1f3f4421417a514607539f17804d95c7ed617265369d1981cae", size = 4979601, upload-time = "2026-05-01T23:29:56.961Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/86/e4/2f8a47ee97f90cd2b933d0463081d35631ff419de2b8c984a5f369857de0/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:136f199a407b5348b9b857c504aff60c77622a28482e7195839ce1b51238c4cc", size = 4510513, upload-time = "2026-05-01T23:30:07.243Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/0e/0e/94e842ff4a7f98ed162580ca2e8b8864b28c1e0350f2443f8ee47f821167/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_ppc64le.whl", hash = "sha256:b6f5a29e9c775b9f12a1a717aa7a2c80f9e1db6f27ba44a5b59c80ac61d2ffcf", size = 4187243, upload-time = "2026-05-01T23:30:15.352Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/d0/83/fc6c174b672e29b7de996ea77b6cbddf46c891751c3355f6974292baa6b4/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_riscv64.whl", hash = "sha256:ee17a2cf4943cde261adfad1bbc5bf38d6b3776d7afff74c7cabcbeaeb08c260", size = 3927347, upload-time = "2026-05-01T23:30:21.186Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/e9/65/768364d4a97a15b1a7f47ba52688c1686f22941d8332a8398cefc468e25f/psycopg_binary-3.3.4-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:5c4ab71be17bdca30cb34c34c4e1496e2f5d6f20c199c12bad226070b22ef9bf", size = 4236393, upload-time = "2026-05-01T23:30:26.211Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/48/a6/828c9185701dab71b234c2a76c38a08b098ebfec5020716b4e93807492b5/psycopg_binary-3.3.4-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:28b7398fdd19db3232c884fb24550bdfe951221f510e195e233299e4c9b78f97", size = 4607292, upload-time = "2026-05-01T23:30:38.962Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/92/58/5b40dbc9d839045c9dae956960e4fb6d20bcabe6c59a2aa34fc3a371913f/psycopg_binary-3.3.4-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:1fbaa292a3c8bb61b45df1ad3da1908ccee7cb889db9425e3557d9e34e2a4829", size = 4687023, upload-time = "2026-05-01T23:30:47.227Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/85/a9/793f0ac107a9003b48441d0d1f9f616d96e0f37458dd8dc12528ceff55fb/psycopg_binary-3.3.4-cp314-cp314-manylinux2014_ppc64le.manylinux_2_17_ppc64le.whl", hash = "sha256:94596f9e7633ee3f6440711d43bb70aa31cc0a46a900ab8b4201a366ace5c9e7", size = 5486985, upload-time = "2026-05-01T23:30:55.517Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/8f/26/42e8533497e2592334f68ec529cf5f840f7fa4e99575a4bb61aa184dbfbf/psycopg_binary-3.3.4-cp314-cp314-manylinux2014_x86_64.manylinux_2_17_x86_64.whl", hash = "sha256:8c0056529e68dbe9184cd4019a1f3d8f3a4ead2f6fc7a5afcf27d3314edd1277", size = 5168745, upload-time = "2026-05-01T23:31:01.904Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/15/af/b7151776cc08d5935d45c833ec818a9beb417cf7c08239af1aafbdae78ee/psycopg_binary-3.3.4-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:2c09aad7051326e7603c14e50636db9c01f78272dc54b3accff03d46370461e6", size = 6761486, upload-time = "2026-05-01T23:31:14.511Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/d0/ed/c92533b9124712d592cbf1cd6c76da933a2e0acea81dfe1fbe7e735f0cff/psycopg_binary-3.3.4-cp314-cp314-manylinux_2_38_riscv64.manylinux_2_39_riscv64.whl", hash = "sha256:514404ed543efd620c85602b747df2a23cf1241b4067199e1a66f2d2757aaa41", size = 4997427, upload-time = "2026-05-01T23:31:20.901Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a2/23/ccadfd0de416aa188356daa199453af24087b042e296088706d190ae0295/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:46893c26858be12cc49ca4226ed6a60b4bfccadd946b3bebb783a60b38788228", size = 4533549, upload-time = "2026-05-01T23:31:26.204Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/fd/a0/c8f43cee36386f7bc891ab41a9d31ea07cf9826038e732da79f26b1e5f34/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_ppc64le.whl", hash = "sha256:df1d567fc430f6df15c9fcf67d87685fc49bdb325adc0db5af1adfb2f44eb5c9", size = 4210256, upload-time = "2026-05-01T23:31:33.884Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/4e/2c/c1547871be3790676e8868b38655496422f94f0978dfb66b74bdba2f1676/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_riscv64.whl", hash = "sha256:6b9016b1714da4dd5ecaaa75b82098aa5a0b87854ce9b092e21c27c4ae23e014", size = 3946204, upload-time = "2026-05-01T23:31:39.626Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c4/b1/f6670f00fa7ea601584623f6c11602ab92117d83eaff885e0210f6de7418/psycopg_binary-3.3.4-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:47c656a8a7ba6eb0cff1801a4caaa9c8bdc12d03080e273aff1c8ac39971a77e", size = 4255811, upload-time = "2026-05-01T23:31:44.986Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "ptyprocess"
|
||||
version = "0.7.0"
|
||||
|
||||
Loading…
Reference in New Issue
Block a user