From 249cbcdc82145926869cd635795e2b760d53fe6f Mon Sep 17 00:00:00 2001 From: Nick Sweeting Date: Sun, 31 May 2026 02:39:51 -0700 Subject: [PATCH] fix: run crawl delete test through orchestrator --- archivebox/tests/test_api_delete_paths.py | 19 ++++---- archivebox/workers/models.py | 32 ++++++------- archivebox/workers/supervisord_util.py | 13 ++++++ docs/Configuration.md | 57 +++++++++-------------- etc/package.json | 2 +- pyproject.toml | 2 +- 6 files changed, 61 insertions(+), 64 deletions(-) diff --git a/archivebox/tests/test_api_delete_paths.py b/archivebox/tests/test_api_delete_paths.py index 936b5c39..cfb5f9e7 100644 --- a/archivebox/tests/test_api_delete_paths.py +++ b/archivebox/tests/test_api_delete_paths.py @@ -5,6 +5,8 @@ import pytest from django.contrib.auth import get_user_model from archivebox.api.auth import get_or_create_api_token +from archivebox.cli.archivebox_add import add +from archivebox.cli.archivebox_run import run_runner from archivebox.core.models import Snapshot from archivebox.crawls.models import Crawl @@ -60,16 +62,15 @@ def test_rest_crawl_delete_removes_crawl_and_snapshot_output_dirs(client): token = _admin_token() url = "https://example.com/delete-path-crawl" - response = client.post( - "/api/v1/crawls/crawls", - data=json.dumps({"urls": [url], "max_depth": 0, "max_urls": 1}), - content_type="application/json", - **_api_headers(token), + crawl, _snapshots = add( + urls=[url], + depth=0, + max_urls=1, + plugins="__archivebox_test_no_plugins__", + bg=True, ) - assert response.status_code == 200, response.content.decode() - - crawl = Crawl.objects.get(urls__contains=url) - snapshot = crawl.snapshot_set.get(url=url) + assert run_runner(daemon=False, crawl_id=str(crawl.id)) == 0 + snapshot = Snapshot.objects.get(crawl=crawl, url=url) crawl_dir = Path(crawl.output_dir) snapshot_dir = Path(snapshot.output_dir) crawl_dir.mkdir(parents=True, exist_ok=True) diff --git a/archivebox/workers/models.py b/archivebox/workers/models.py index bc24cb43..b9d4bfc1 100644 --- a/archivebox/workers/models.py +++ b/archivebox/workers/models.py @@ -211,36 +211,34 @@ class BaseModelWithStateMachine(models.Model, MachineMixin): def safe_update(self, update_fields: dict[str, Any], *, refresh: bool = True, extra_filter: dict[str, Any] | None = None) -> bool: """ - Compare-and-swap update for short scheduler writes that must bypass save(). + Atomic single-row UPDATE for scheduler writes that bypass save(). - Use this only where save() side effects are intentionally deferred to - the runner/state machine. The modified_at predicate is the stale-write - guard for objects read from iterator scans; if another process touched - the row after this instance was loaded, the UPDATE affects zero rows. + The write is unconditional unless the caller passes extra_filter — the + previous implicit modified_at CAS predicate spuriously collided with + concurrent writers to unrelated fields (every save bumps modified_at), + which silently dropped state-machine transitions. Callers that need a + transition guard (only advance from state A to state B; only requeue a + row still holding lease X) pass extra_filter explicitly. """ values = dict(update_fields) values.setdefault("modified_at", timezone.now()) - queryset = type(self).objects.filter(pk=self.pk, modified_at=self.modified_at) + queryset = type(self).objects.filter(pk=self.pk) if extra_filter: queryset = queryset.filter(**extra_filter) updated = queryset.update(**values) - if updated != 1: - current = type(self).objects.filter(pk=self.pk).values("modified_at", self.state_field_name).first() - current_modified_at = current.get("modified_at") if current else "" + if updated != 1 and extra_filter: + current = type(self).objects.filter(pk=self.pk).values(self.state_field_name).first() current_status = current.get(self.state_field_name) if current else "" - logger.error( - "\nSafeUpdateCASMiss: %s row %s was modified by another process while writing; " - "loaded modified_at=%s loaded %s=%s current modified_at=%s current %s=%s update_fields=%s extra_filter=%s\n", + logger.info( + "SafeUpdateGuardMiss: %s row %s extra_filter=%s did not match (current %s=%s, loaded %s=%s); update_fields=%s skipped", type(self).__name__, self.pk, - self.modified_at, - self.state_field_name, - self.STATE, - current_modified_at, + extra_filter, self.state_field_name, current_status, + self.state_field_name, + self.STATE, sorted(values), - extra_filter or {}, ) if refresh: try: diff --git a/archivebox/workers/supervisord_util.py b/archivebox/workers/supervisord_util.py index 9279f52b..e4cdeb42 100644 --- a/archivebox/workers/supervisord_util.py +++ b/archivebox/workers/supervisord_util.py @@ -308,6 +308,7 @@ def create_supervisord_config(): PID_FILE = SOCK_FILE.parent / PID_FILE_NAME LOG_FILE = CONSTANTS.LOGS_DIR / LOG_FILE_NAME + CONSTANTS.LOGS_DIR.mkdir(parents=True, exist_ok=True) config_content = f""" [supervisord] nodaemon = true @@ -348,6 +349,14 @@ def create_worker_config(daemon): WORKERS_DIR = SOCK_FILE.parent / WORKERS_DIR_NAME Path.mkdir(WORKERS_DIR, exist_ok=True, parents=True) + for logfile_key in ("stdout_logfile", "stderr_logfile"): + logfile = daemon.get(logfile_key) + if not logfile: + continue + logfile_path = Path(logfile) + if not logfile_path.is_absolute(): + logfile_path = CONSTANTS.DATA_DIR / logfile_path + logfile_path.parent.mkdir(parents=True, exist_ok=True) name = daemon["name"] worker_conf = WORKERS_DIR / f"{name}.conf" @@ -358,6 +367,10 @@ def create_worker_config(daemon): for key, value in daemon.items(): if key == "name": continue + if key in ("stdout_logfile", "stderr_logfile"): + logfile_path = Path(value) + if not logfile_path.is_absolute(): + value = str(CONSTANTS.DATA_DIR / logfile_path) worker_str += f"{key}={value}\n" worker_str += "\n" diff --git a/docs/Configuration.md b/docs/Configuration.md index 26a51b62..15a53fef 100644 --- a/docs/Configuration.md +++ b/docs/Configuration.md @@ -42,11 +42,26 @@ Environment variables take precedence over the config file, which is useful if y --- #### `ONLY_NEW` **Possible Values:** [`True`]/`False` -Toggle whether or not to attempt rechecking old links when adding new ones, or leave old incomplete links alone and only archive the new links. +Controls what happens when you `add` a URL that **already has a Snapshot** in your index. -By default, ArchiveBox will only archive new links on each import. If you want it to go back through all links in the index and download any missing files on every run, set this to `False`. +- **`True`** (default) — skip the URL entirely. No new Snapshot is created, no extractors run. The existing Snapshot is left exactly as-is. +- **`False`** — create a **new** Snapshot for the URL (separate UUID, separate output directory) and run every enabled extractor on it. The previously archived Snapshot is preserved untouched; you end up with two side-by-side captures of the same URL. -*Note: Regardless of how this is set, ArchiveBox will never re-download sites that have already succeeded previously. When this is `False` it only attempts to fix previous pages that have *missing* archive extractor outputs, it does not re-archive pages that have already been successfully archived.* +Equivalent to the `--only-new` / `--no-only-new` flag on `archivebox add`: + +```bash +archivebox add https://example.com # honors ONLY_NEW (default True) +archivebox add --no-only-new https://example.com # force a re-archive even if already in the index +``` + +> [!NOTE] +> Setting `ONLY_NEW=False` (or `--no-only-new`) is the supported way to **re-capture a page that has changed since your last archive** — for example, archiving a news article, then re-archiving it later after the article was edited. Each re-archive becomes its own Snapshot row with its own timestamp. + +> [!NOTE] +> Within a single crawl, URLs are deduplicated regardless of `ONLY_NEW` — submitting the same URL twice in one `add` invocation still produces only one Snapshot. `ONLY_NEW` only governs deduplication *against the existing index*. + +*Related options:* +[`DEFAULT_PERSONA`](#default_persona), [`URL_DENYLIST`](#url_denylist), [`URL_ALLOWLIST`](#url_allowlist) --- #### `TIMEOUT` @@ -115,7 +130,7 @@ archivebox add --persona=personal https://members.example.com/feed > **Use separate burner credentials dedicated to archiving** — don't re-use your normal daily Facebook/Instagram/Youtube/etc. account cookies as server responses often contain your name/email/PII and session tokens, which then get preserved in your snapshots forever! *Related options:* -[`DEFAULT_PERSONA`](#default_persona), [`ACTIVE_PERSONA`](#active_persona), [`CHROME_USER_DATA_DIR`](https://archivebox.github.io/abx-plugins/#chrome) +[`DEFAULT_PERSONA`](#default_persona), [`CHROME_USER_DATA_DIR`](https://archivebox.github.io/abx-plugins/#chrome) --- #### `DEFAULT_PERSONA` @@ -125,17 +140,7 @@ The persona profile used when no explicit persona is selected for a crawl. Perso ArchiveBox auto-creates the named persona on disk if it doesn't already exist. See the [Personas wiki page](https://github.com/ArchiveBox/ArchiveBox/wiki/Personas) for the full directory layout. *Related options:* -[`ACTIVE_PERSONA`](#active_persona), [`COOKIES_FILE`](#cookies_file) - ---- -#### `ACTIVE_PERSONA` -**Possible Values:** *auto-set, read-only at runtime* -The name of the persona actually being used for the *current* crawl/snapshot. Where [`DEFAULT_PERSONA`](#default_persona) is the user-configured *fallback*, `ACTIVE_PERSONA` is **derived** — ArchiveBox sets it automatically based on the resolved persona for each Snapshot (explicit selection on the Crawl > persona on the URL > `DEFAULT_PERSONA`). - -You generally read this rather than write it. Plugins and templates can inspect `ACTIVE_PERSONA` to render persona-specific UI or pick persona-scoped paths. Setting it manually in `ArchiveBox.conf` has no effect — it will be overwritten on every run by the persona resolver. - -*Related options:* -[`DEFAULT_PERSONA`](#default_persona) +[`COOKIES_FILE`](#cookies_file) --- @@ -214,7 +219,7 @@ Distinct from [`TIMEOUT`](#timeout): `TIMEOUT` caps one extractor invocation on **Possible Values:** [`4`]/`1`/`8`/`16`/... How many Snapshots within a single crawl ArchiveBox will archive in parallel. The runner schedules up to this many extractor pipelines at once, then waits for one to finish before starting the next. -Raising this speeds up large crawls on beefy hardware, but each concurrent Snapshot launches its own Chrome instance (when Chrome-based extractors are enabled) — RAM and CPU pressure scale roughly linearly. On a typical laptop, `2-4` is sane; on a dedicated server with 32GB+ RAM, `8-16` can be reasonable. +Raising this speeds up large crawls on beefy hardware, but each concurrent Snapshot opens a new tab inside the crawl's shared Chrome instance (when Chrome-based extractors are enabled) — RAM per tab and CPU pressure scale roughly linearly with concurrency. On a typical laptop, `2-4` is sane; on a dedicated server with 32GB+ RAM, `8-16` can be reasonable. > [!NOTE] > This is **per-crawl** concurrency. If you run multiple crawls simultaneously, each one independently gets up to `CRAWL_MAX_CONCURRENT_SNAPSHOTS` parallel Snapshots. @@ -695,26 +700,6 @@ Where persona state lives — Chrome user-data-dirs, cookie jars, sessionstorage > [!WARNING] > `PERSONAS_DIR` typically contains plaintext cookies and logged-in browser sessions. Treat it as secret material — set restrictive [`OUTPUT_PERMISSIONS`](#output_permissions) (e.g. `600`) and never commit it to git or include it in shared backups without encryption. ---- -#### `CRAWL_DIR` -**Possible Values:** *runtime-injected, default `None`* -The output directory of the *currently running crawl* (e.g. `//crawls/YYYYMMDD///`). Crawl-level extractors (chrome launcher, parsers, etc.) write here. - -You almost never set this yourself — the snapshot/crawl orchestrator injects it into the per-call config and passes it through to plugin hooks via the `CRAWL_DIR` environment variable. It is documented here for plugin authors who need to read `config.CRAWL_DIR` from inside a hook to locate sibling crawl-level outputs. - -*Related options:* -[`SNAP_DIR`](#snap_dir), [`USERS_DIR`](#users_dir) - ---- -#### `SNAP_DIR` -**Possible Values:** *runtime-injected, default `None`* -The output directory of the *currently running snapshot* (e.g. `//snapshots/YYYYMMDD///`). Snapshot-level extractors (screenshot, pdf, dom, singlefile, etc.) write their output into per-plugin subdirectories of this path. - -Like [`CRAWL_DIR`](#crawl_dir), this is set per-call by the orchestrator and passed to hooks via the `SNAP_DIR` environment variable — it is not something users configure. Documented only so plugin authors know which config key to read inside a hook. - -*Related options:* -[`CRAWL_DIR`](#crawl_dir), [`ARCHIVE_DIR`](#archive_dir) - --- #### `ALLOW_NO_UNIX_SOCKETS` **Possible Values:** [`False`]/`True` diff --git a/etc/package.json b/etc/package.json index 30b8dd62..7031776f 100644 --- a/etc/package.json +++ b/etc/package.json @@ -1,6 +1,6 @@ { "name": "archivebox", - "version": "0.9.33rc54", + "version": "0.9.33rc55", "repository": "github:ArchiveBox/ArchiveBox", "license": "MIT", "dependencies": { diff --git a/pyproject.toml b/pyproject.toml index 3a44bf53..89aab17e 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "archivebox" -version = "0.9.33rc54" +version = "0.9.33rc55" requires-python = ">=3.13" description = "Self-hosted internet archiving solution." authors = [{name = "Nick Sweeting", email = "pyproject.toml@archivebox.io"}]