diff --git a/.github/workflows/test-parallel.yml b/.github/workflows/test-parallel.yml
index 0f6d3403..ac997d2b 100644
--- a/.github/workflows/test-parallel.yml
+++ b/.github/workflows/test-parallel.yml
@@ -318,6 +318,7 @@ jobs:
>> "$GITHUB_ENV"
- name: Install PostgreSQL server binaries
+ if: contains(toJson(matrix.test.paths), 'test_postgres_backend')
run: |
set -Eeuo pipefail
if ! ls /usr/lib/postgresql/*/bin/initdb >/dev/null 2>&1 && ! command -v initdb >/dev/null 2>&1; then
diff --git a/archivebox/api/migrations/0001_initial.py b/archivebox/api/migrations/0001_initial.py
index b3a2aab3..f2667e52 100644
--- a/archivebox/api/migrations/0001_initial.py
+++ b/archivebox/api/migrations/0001_initial.py
@@ -96,6 +96,12 @@ def _pg_sync_schema(apps, schema_editor):
rebuild_models_from_migration_state(apps, schema_editor, "api", ["APIToken", "OutboundWebhook"])
+def _pg_drop_schema(apps, schema_editor):
+ from archivebox.misc.db import drop_models_on_postgres
+
+ drop_models_on_postgres(apps, schema_editor, "api", ["OutboundWebhook", "APIToken"])
+
+
class Migration(migrations.Migration):
initial = True
@@ -271,5 +277,5 @@ class Migration(migrations.Migration):
# On non-sqlite backends the raw DDL above is skipped, so the real
# schema diverges from Django state. Rebuild both tables (including the
# unique constraint) from the post-migration state (no-op on sqlite).
- migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
+ migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
]
diff --git a/archivebox/core/models.py b/archivebox/core/models.py
index 04fd18e4..c710574a 100755
--- a/archivebox/core/models.py
+++ b/archivebox/core/models.py
@@ -183,9 +183,14 @@ class SnapshotQuerySet(models.QuerySet):
for crawl_id, permissions in Crawl.objects.filter(pk__in=missing_crawl_ids).values_list("pk", "permissions")
}
+ from archivebox.misc.db import truncate_overlong_charfields
+
for obj in objs:
if isinstance(obj, self.model):
obj.ensure_permissions_config(crawl_permissions=crawl_permissions_by_id.get(str(obj.crawl_id)))
+ # bulk_create bypasses pre_save, so clamp CharFields here too
+ # (e.g. page titles) to stay within postgres VARCHAR limits.
+ truncate_overlong_charfields(obj)
return super().bulk_create(objs, *args, **kwargs)
def paged_iterator(self, chunk_size: int = 500):
diff --git a/archivebox/core/settings.py b/archivebox/core/settings.py
index 80619bcd..7f5239bd 100644
--- a/archivebox/core/settings.py
+++ b/archivebox/core/settings.py
@@ -15,9 +15,9 @@ from archivebox.config.constants import CONSTANTS
from archivebox.config.common import get_config
from archivebox.core.routes_util import get_api_base_url, get_admin_base_url, get_base_url, normalize_base_url
-# All sqlite-vs-postgres connection logic lives in archivebox.misc.db;
-# DATABASE_ENGINE config selects the backend (sqlite by default).
-from archivebox.misc.db import get_database_settings, get_sqlite_connection_options
+# DATABASE_ENGINE config selects the backend (sqlite by default); the
+# sqlite-vs-postgres helpers live in archivebox.misc.db.
+from archivebox.misc.db import is_postgres, postgres_db_params
from .settings_logging import SETTINGS_LOGGING
@@ -220,11 +220,50 @@ DATABASE_NAME = CONFIG.DATABASE_NAME
SQLITE_JOURNAL_MODE = CONFIG.SQLITE_JOURNAL_MODE
SQLITE_MMAP_SIZE = CONFIG.SQLITE_MMAP_SIZE
-SQLITE_CONNECTION_OPTIONS = get_sqlite_connection_options()
-
-DATABASES = {
- "default": get_database_settings(),
+SQLITE_CONNECTION_OPTIONS = {
+ "ENGINE": "archivebox.core.sqlite_backend",
+ "TIME_ZONE": CONSTANTS.TIMEZONE,
+ "OPTIONS": {
+ # https://gcollazo.com/optimal-sqlite-settings-for-django/
+ # https://litestream.io/tips/#busy-timeout
+ # https://docs.djangoproject.com/en/5.1/ref/databases/#setting-pragma-options
+ "timeout": CONFIG.SQLITE_BUSY_TIMEOUT / 1000,
+ "check_same_thread": False,
+ # Keep SQLite on Django's default deferred transaction mode. BEGIN
+ # IMMEDIATE grabs the write lock as soon as atomic() opens, which is
+ # exactly what hurts ArchiveBox on large collections where Python code
+ # may do filesystem work before the actual row write. Deferred BEGIN
+ # keeps writes statement-scoped unless a caller explicitly opens a
+ # transaction around multiple writes.
+ "transaction_mode": None,
+ "init_command": (
+ "PRAGMA foreign_keys=ON;"
+ f"PRAGMA busy_timeout = {CONFIG.SQLITE_BUSY_TIMEOUT};"
+ f"PRAGMA journal_mode = {SQLITE_JOURNAL_MODE};"
+ "PRAGMA synchronous = NORMAL;"
+ "PRAGMA temp_store = MEMORY;"
+ f"PRAGMA mmap_size = {SQLITE_MMAP_SIZE};"
+ "PRAGMA journal_size_limit = 67108864;"
+ "PRAGMA cache_size = 2000;"
+ ),
+ },
}
+
+if is_postgres():
+ DATABASES = {
+ "default": {
+ "ENGINE": "django.db.backends.postgresql",
+ **postgres_db_params(),
+ "OPTIONS": {"connect_timeout": 10},
+ },
+ }
+else:
+ DATABASES = {
+ "default": {
+ "NAME": DATABASE_NAME,
+ **SQLITE_CONNECTION_OPTIONS,
+ },
+ }
MIGRATION_MODULES = {"signal_webhooks": None}
# Django requires DEFAULT_AUTO_FIELD to subclass AutoField (BigAutoField, SmallAutoField, etc.)
diff --git a/archivebox/crawls/migrations/0001_initial.py b/archivebox/crawls/migrations/0001_initial.py
index 6a079d30..f7e97fee 100644
--- a/archivebox/crawls/migrations/0001_initial.py
+++ b/archivebox/crawls/migrations/0001_initial.py
@@ -101,6 +101,12 @@ def _pg_sync_schema(apps, schema_editor):
rebuild_models_from_migration_state(apps, schema_editor, "crawls", ["CrawlSchedule", "Crawl"])
+def _pg_drop_schema(apps, schema_editor):
+ from archivebox.misc.db import drop_models_on_postgres
+
+ drop_models_on_postgres(apps, schema_editor, "crawls", ["Crawl", "CrawlSchedule"])
+
+
class Migration(migrations.Migration):
initial = True
@@ -210,5 +216,5 @@ class Migration(migrations.Migration):
# schema diverges from Django state. Rebuild both tables from the
# post-migration state (no-op on sqlite). Runs as a top-level op so
# ``apps`` reflects the full state including the circular template FK.
- migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
+ migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
]
diff --git a/archivebox/machine/migrations/0001_initial.py b/archivebox/machine/migrations/0001_initial.py
index cea929bb..7d688ea2 100644
--- a/archivebox/machine/migrations/0001_initial.py
+++ b/archivebox/machine/migrations/0001_initial.py
@@ -39,6 +39,12 @@ def _pg_sync_schema(apps, schema_editor):
)
+def _pg_drop_schema(apps, schema_editor):
+ from archivebox.misc.db import drop_models_on_postgres
+
+ drop_models_on_postgres(apps, schema_editor, "machine", ["Binary", "NetworkInterface", "Machine"])
+
+
_SQLITE_INITIAL_SQL = """
-- Create machine_machine table
CREATE TABLE IF NOT EXISTS machine_machine (
@@ -279,5 +285,5 @@ class Migration(migrations.Migration):
),
],
),
- migrations.RunPython(_pg_sync_schema, reverse_code=migrations.RunPython.noop),
+ migrations.RunPython(_pg_sync_schema, reverse_code=_pg_drop_schema),
]
diff --git a/archivebox/machine/migrations/0011_remove_binary_output_dir.py b/archivebox/machine/migrations/0011_remove_binary_output_dir.py
index 92f0e935..820637bf 100644
--- a/archivebox/machine/migrations/0011_remove_binary_output_dir.py
+++ b/archivebox/machine/migrations/0011_remove_binary_output_dir.py
@@ -2,17 +2,12 @@ from django.db import migrations
def remove_output_dir_if_exists(apps, schema_editor):
- if schema_editor.connection.vendor == "sqlite":
- cursor = schema_editor.connection.cursor()
- cursor.execute("PRAGMA table_info(machine_binary)")
- columns = {row[1] for row in cursor.fetchall()}
- else:
- # Portable introspection: on non-sqlite the RemoveField below is
- # state-only, so the real ALTER TABLE DROP COLUMN must run here too.
- # A fresh non-sqlite DB has machine_binary with output_dir from 0001.
- from archivebox.misc.db import migration_table_columns
-
- columns = migration_table_columns(schema_editor.connection, "machine_binary")
+ # On non-sqlite the RemoveField below is state-only, so the real
+ # ALTER TABLE DROP COLUMN must run here too (a fresh DB has machine_binary
+ # with output_dir from 0001). Portable introspection works on both backends.
+ connection = schema_editor.connection
+ with connection.cursor() as cursor:
+ columns = {col.name for col in connection.introspection.get_table_description(cursor, "machine_binary")}
if "output_dir" not in columns:
return
diff --git a/archivebox/misc/db.py b/archivebox/misc/db.py
index f154b8fe..8d30de0b 100644
--- a/archivebox/misc/db.py
+++ b/archivebox/misc/db.py
@@ -57,57 +57,6 @@ def postgres_db_params() -> dict[str, str]:
}
-def get_sqlite_connection_options() -> dict[str, Any]:
- """ENGINE + OPTIONS for the sqlite backend (without NAME)."""
- from archivebox.config.common import get_config
-
- config = get_config()
- return {
- "ENGINE": "archivebox.core.sqlite_backend",
- "TIME_ZONE": CONSTANTS.TIMEZONE,
- "OPTIONS": {
- # https://gcollazo.com/optimal-sqlite-settings-for-django/
- # https://litestream.io/tips/#busy-timeout
- # https://docs.djangoproject.com/en/5.1/ref/databases/#setting-pragma-options
- "timeout": config.SQLITE_BUSY_TIMEOUT / 1000,
- "check_same_thread": False,
- # Keep SQLite on Django's default deferred transaction mode. BEGIN
- # IMMEDIATE grabs the write lock as soon as atomic() opens, which is
- # exactly what hurts ArchiveBox on large collections where Python code
- # may do filesystem work before the actual row write. Deferred BEGIN
- # keeps writes statement-scoped unless a caller explicitly opens a
- # transaction around multiple writes.
- "transaction_mode": None,
- "init_command": (
- "PRAGMA foreign_keys=ON;"
- f"PRAGMA busy_timeout = {config.SQLITE_BUSY_TIMEOUT};"
- f"PRAGMA journal_mode = {config.SQLITE_JOURNAL_MODE};"
- "PRAGMA synchronous = NORMAL;"
- "PRAGMA temp_store = MEMORY;"
- f"PRAGMA mmap_size = {config.SQLITE_MMAP_SIZE};"
- "PRAGMA journal_size_limit = 67108864;"
- "PRAGMA cache_size = 2000;"
- ),
- },
- }
-
-
-def get_database_settings() -> dict[str, Any]:
- """The DATABASES['default'] dict for Django settings, per configured backend."""
- from archivebox.config.common import get_config
-
- if is_postgres():
- return {
- "ENGINE": "django.db.backends.postgresql",
- **postgres_db_params(),
- "OPTIONS": {"connect_timeout": 10},
- }
- return {
- "NAME": get_config().DATABASE_NAME,
- **get_sqlite_connection_options(),
- }
-
-
def _psycopg_connect(dbname: str | None = None, connect_timeout: int = 5):
import psycopg
@@ -214,17 +163,23 @@ def approximate_row_counts(connection) -> dict[str, int]:
return counts
-def truncate_overlong_charfields(sender, instance, **kwargs) -> None:
- """pre_save receiver: clamp CharField values to their declared max_length.
+def truncate_overlong_charfields(instance=None, **kwargs) -> None:
+ """Clamp a model instance's CharField values to their declared max_length.
SQLite never enforces VARCHAR(n) limits, so ArchiveBox has always stored
overlong values (e.g. long crawl labels or page titles) untruncated.
PostgreSQL enforces them and would raise DataError on save instead.
- Truncating at save time keeps writes succeeding identically on both
- backends. Registered once for all models in CoreConfig.ready().
+ Truncating keeps writes succeeding identically on both backends.
+
+ Dual-use: works as a ``pre_save`` receiver (Django passes ``instance=`` and
+ ``sender=`` as kwargs; registered in ``CoreConfig.ready()``) and as a plain
+ ``truncate_overlong_charfields(obj)`` call for ``bulk_create`` paths, which
+ bypass signals.
"""
from django.db import models as dj_models
+ if instance is None:
+ return
for field in instance._meta.local_concrete_fields:
if isinstance(field, dj_models.CharField) and field.max_length:
value = getattr(instance, field.attname, None)
@@ -235,12 +190,6 @@ def truncate_overlong_charfields(sender, instance, **kwargs) -> None:
# --- migration helpers ------------------------------------------------------
-def migration_table_columns(connection, table_name: str) -> set[str]:
- """Portable column-name introspection usable from inside migrations."""
- with connection.cursor() as cursor:
- return {col.name for col in connection.introspection.get_table_description(cursor, table_name)}
-
-
def rebuild_models_from_migration_state(apps, schema_editor, app_label: str, model_names: list[str]) -> None:
"""(non-sqlite only) Drop and recreate the given models' tables from the
current migration state.
@@ -264,6 +213,22 @@ def rebuild_models_from_migration_state(apps, schema_editor, app_label: str, mod
schema_editor.create_model(model)
+def drop_models_on_postgres(apps, schema_editor, app_label: str, model_names: list[str]) -> None:
+ """Reverse companion to ``rebuild_models_from_migration_state``.
+
+ Drops the given models' tables on non-sqlite backends (in the order given,
+ so callers pass reverse-dependency order). No-op on sqlite, whose reverse is
+ handled by the gated ``RunSQL`` reverse_sql instead.
+ """
+ if schema_editor.connection.vendor == "sqlite":
+ return
+ existing_tables = set(schema_editor.connection.introspection.table_names())
+ for model_name in model_names:
+ model = apps.get_model(app_label, model_name)
+ if model._meta.db_table in existing_tables:
+ schema_editor.delete_model(model)
+
+
def run_db_analyze_batch(
remaining: list[str] | None,
*,
diff --git a/archivebox/search/query.py b/archivebox/search/query.py
index 2142faa4..86eb7745 100644
--- a/archivebox/search/query.py
+++ b/archivebox/search/query.py
@@ -40,10 +40,20 @@ def crawl_config_values_search_wave(query: str) -> Q | None:
params=[pattern],
)
elif connection.vendor == "postgresql":
- # jsonb has no json_tree() equivalent; matching the serialized jsonb text
- # covers every nested value (keys can also match, which only widens recall).
+ # Match only scalar config *values*, not keys (mirrors SQLite's
+ # json_tree.atom). jsonb_path_query('$.**') walks every nested node;
+ # keep the non-container leaves and compare their text form.
matching_crawls = Crawl.objects.extra(
- where=["LOWER(config::text) LIKE %s ESCAPE '\\'"],
+ where=[
+ """
+ EXISTS (
+ SELECT 1
+ FROM jsonb_path_query(config, '$.**') AS leaf
+ WHERE jsonb_typeof(leaf) NOT IN ('object', 'array')
+ AND LOWER(leaf #>> '{}') LIKE %s ESCAPE '\\'
+ )
+ """,
+ ],
params=[pattern],
)
else:
diff --git a/docs/Configuration.md b/docs/Configuration.md
index 5a996bba..8a78f06b 100644
--- a/docs/Configuration.md
+++ b/docs/Configuration.md
@@ -719,11 +719,18 @@ DATABASE_PASSWORD = s3cret
**Possible Values:** [`127.0.0.1`] / [`5432`] / [`archivebox`] / [empty]
PostgreSQL connection settings, used only when [`DATABASE_ENGINE`](#database_engine)`=postgres`. Settable as `ARCHIVEBOX_DATABASE_HOST`, `ARCHIVEBOX_DATABASE_PORT`, `ARCHIVEBOX_DATABASE_USER`, and `ARCHIVEBOX_DATABASE_PASSWORD`.
-`DATABASE_HOST` may also be a path to a directory containing a PostgreSQL unix socket (e.g. `/var/run/postgresql`). When `DATABASE_ENGINE=postgres`, [`DATABASE_NAME`](#database_name) is the name of the PostgreSQL database (default: `archivebox`) instead of a file path.
+`DATABASE_HOST` may also be a path to a directory containing a PostgreSQL unix socket (e.g. `/var/run/postgresql`).
---
+#### `DATABASE_NAME`
+**Possible Values:** [`index.sqlite3`] / `archivebox` / ...
+The main index database. Settable as `ARCHIVEBOX_DATABASE_NAME`.
+
+With the default [`DATABASE_ENGINE`](#database_engine)`=sqlite`, this is the path to the SQLite index file inside the data directory (`index.sqlite3`). With `DATABASE_ENGINE=postgres`, it is the name of the PostgreSQL database instead (default: `archivebox`), created automatically by `archivebox init` if it does not exist.
+
+---
#### `SQLITE_JOURNAL_MODE`
**Possible Values:** [`WAL`]/`DELETE`/`TRUNCATE`/`PERSIST`/`MEMORY`/`OFF`
SQLite [journal mode](https://www.sqlite.org/pragma.html#pragma_journal_mode), applied via `PRAGMA journal_mode = ...` on every new connection. Settable as `ARCHIVEBOX_SQLITE_JOURNAL_MODE`.