mirror of
https://github.com/ArchiveBox/ArchiveBox.git
synced 2026-09-14 02:56:11 +05:00
Stabilize frozen config CLI test flows
This commit is contained in:
parent
7062f9f2e5
commit
2bca869e32
@ -66,6 +66,8 @@ class AddCommandSchema(Schema):
|
||||
parser: str = "auto"
|
||||
plugins: str = ""
|
||||
only_new: bool | None = None
|
||||
update: bool = False
|
||||
overwrite: bool = False
|
||||
index_only: bool = False
|
||||
|
||||
|
||||
@ -90,6 +92,8 @@ class ScheduleCommandSchema(Schema):
|
||||
tag: str = ""
|
||||
depth: int = 0
|
||||
only_new: bool | None = None
|
||||
update: bool = False
|
||||
overwrite: bool = False
|
||||
clear: bool = False
|
||||
|
||||
|
||||
@ -120,6 +124,8 @@ def cli_add(request: HttpRequest, args: AddCommandSchema):
|
||||
config_overrides: dict[str, object] = {}
|
||||
if args.only_new is not None:
|
||||
config_overrides["ONLY_NEW"] = bool(args.only_new)
|
||||
if args.update or args.overwrite:
|
||||
config_overrides["ONLY_NEW"] = False
|
||||
crawl, snapshots = add(
|
||||
urls=args.urls,
|
||||
snapshot_ids=args.snapshot_ids,
|
||||
@ -189,6 +195,8 @@ def cli_schedule(request: HttpRequest, args: ScheduleCommandSchema):
|
||||
config_overrides: dict[str, object] = {}
|
||||
if args.only_new is not None:
|
||||
config_overrides["ONLY_NEW"] = bool(args.only_new)
|
||||
if args.update or args.overwrite:
|
||||
config_overrides["ONLY_NEW"] = False
|
||||
result = schedule(
|
||||
import_path=args.import_path,
|
||||
add=args.add,
|
||||
|
||||
@ -122,7 +122,8 @@ def create_crawl(request: HttpRequest, data: CrawlCreateSchema):
|
||||
|
||||
tags = normalize_tag_list(data.tags, data.tags_str)
|
||||
config = dict(data.config or {})
|
||||
config.setdefault("PERMISSIONS", str(get_config(user=request.user).PERMISSIONS))
|
||||
request_user = request.user if request.user.is_authenticated else None
|
||||
config.setdefault("PERMISSIONS", str(get_config(user=request_user).PERMISSIONS))
|
||||
crawl = Crawl.objects.create(
|
||||
urls="\n".join(urls),
|
||||
max_depth=data.max_depth,
|
||||
|
||||
@ -185,7 +185,7 @@ def add(
|
||||
label=f"{USER}@{HOSTNAME} $ {cmd_str} [{timestamp}]",
|
||||
created_by_id=created_by_id,
|
||||
status=Crawl.StatusChoices.QUEUED,
|
||||
retry_at=None if index_only else timezone.now(),
|
||||
retry_at=None if (index_only or bg) else timezone.now(),
|
||||
config=crawl_config,
|
||||
)
|
||||
|
||||
@ -198,15 +198,8 @@ def add(
|
||||
# Discovered URLs become child Snapshots (depth+1)
|
||||
|
||||
if index_only:
|
||||
# ``--index-only`` means "add the URLs to the index without archiving
|
||||
# them now". That only holds if we actually materialize the Snapshot
|
||||
# rows here — otherwise the CLI returns success with nothing in the
|
||||
# index, which broke ``test_add_url_after_init`` & friends. Create
|
||||
# the Snapshots synchronously (the same step the runner would do)
|
||||
# but skip starting any worker so extractors don't run.
|
||||
crawl.create_snapshots_from_urls()
|
||||
print("[yellow]\\[*] Index-only mode - URLs indexed, runner not started[/yellow]")
|
||||
return crawl, crawl.snapshot_set.all()
|
||||
print("[yellow]\\[*] Index-only mode - URLs queued, runner not started[/yellow]")
|
||||
return crawl, crawl.snapshot_set.none()
|
||||
|
||||
# 5. Start the crawl runner to process the queue
|
||||
# The runner will:
|
||||
@ -330,6 +323,8 @@ def add(
|
||||
"Pass --no-only-new to force re-archive of URLs that already exist.",
|
||||
)
|
||||
@click.option("--index-only", is_flag=True, help="Just add the URLs to the index without archiving them now")
|
||||
@click.option("--overwrite", is_flag=True, help="Re-archive URLs even if they already exist (alias for --no-only-new)")
|
||||
@click.option("--update", is_flag=True, help="Re-archive URLs even if they already exist (alias for --no-only-new)")
|
||||
@click.option("--bg", is_flag=True, help="Run archiving in background (queue work and return immediately)")
|
||||
@click.argument("urls", nargs=-1, type=click.Path())
|
||||
@docstring(add.__doc__)
|
||||
@ -360,7 +355,11 @@ def main(**kwargs):
|
||||
|
||||
# Translate --only-new/--no-only-new into a crawl config override.
|
||||
# add() takes config overrides as a dict; no per-flag kwargs.
|
||||
overwrite = kwargs.pop("overwrite", False)
|
||||
update = kwargs.pop("update", False)
|
||||
only_new = kwargs.pop("only_new", None)
|
||||
if overwrite or update:
|
||||
only_new = False
|
||||
if only_new is not None:
|
||||
kwargs["config"] = {"ONLY_NEW": bool(only_new)}
|
||||
|
||||
|
||||
@ -73,6 +73,8 @@ def init(force: bool = False, quick: bool = False, install: bool = False) -> Non
|
||||
config.ARCHIVE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
config.USERS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
Path(CONSTANTS.LOGS_DIR).mkdir(exist_ok=True)
|
||||
for path in (Path(CONSTANTS.SOURCES_DIR), config.ARCHIVE_DIR, config.USERS_DIR, Path(CONSTANTS.LOGS_DIR)):
|
||||
path.chmod(int(config.OUTPUT_PERMISSIONS, base=8) | 0o111)
|
||||
|
||||
print(f" + {_display_data_path(CONSTANTS.CONFIG_FILE, DATA_DIR)}...")
|
||||
|
||||
|
||||
@ -28,6 +28,11 @@ def install(binaries: tuple[str, ...] = (), binproviders: str = "*", dry_run: bo
|
||||
|
||||
config = get_config()
|
||||
archive_dir = config.ARCHIVE_DIR
|
||||
|
||||
if dry_run:
|
||||
print("[dim]Dry run - would detect ArchiveBox dependencies and run the abx-dl install flow[/dim]")
|
||||
return
|
||||
|
||||
if not (os.access(archive_dir, os.R_OK) and archive_dir.is_dir()):
|
||||
init() # must init full index because we need a db to store Binary entries in
|
||||
|
||||
@ -47,10 +52,6 @@ def install(binaries: tuple[str, ...] = (), binproviders: str = "*", dry_run: bo
|
||||
print(f" DATA_DIR will be owned by [blue]{ARCHIVEBOX_USER}:{ARCHIVEBOX_GROUP}[/blue].")
|
||||
print()
|
||||
|
||||
if dry_run:
|
||||
print("[dim]Dry run - would run the abx-dl install flow[/dim]")
|
||||
return
|
||||
|
||||
# Set up Django
|
||||
from archivebox.config.django import setup_django
|
||||
|
||||
|
||||
@ -1,3 +1,5 @@
|
||||
from __future__ import annotations
|
||||
|
||||
__package__ = "archivebox.config"
|
||||
|
||||
import json
|
||||
@ -13,7 +15,7 @@ from typing import Any, ClassVar, cast
|
||||
from pathlib import Path
|
||||
|
||||
from rich.console import Console
|
||||
from pydantic import BaseModel, Field, create_model, field_validator, model_validator
|
||||
from pydantic import BaseModel, Field, PrivateAttr, create_model, field_validator, model_validator
|
||||
from pydantic_settings import SettingsConfigDict
|
||||
from abx_plugins.plugins.base.utils import BASE_CONFIG_PATH, build_config_model, resolve_plugin_configs
|
||||
|
||||
@ -64,6 +66,18 @@ def permissions_from_legacy_public_flags(raw_config: Mapping[str, object]) -> st
|
||||
|
||||
_SENSITIVE_CONFIG_KEY_NEEDLES = ("TOKEN", "SECRET", "API_KEY", "APIKEY", "PASSWORD")
|
||||
SENSITIVE_CONFIG_VALUE_REDACTED = "********"
|
||||
_SCOPE_CRAWL_FROZEN = "crawl_frozen"
|
||||
_SCOPE_CRAWL_EXECUTION = "crawl_execution"
|
||||
_SCOPE_SERVER = "server"
|
||||
|
||||
|
||||
|
||||
def _plugin_sensitive_config_keys() -> set[str]:
|
||||
sensitive_keys: set[str] = set()
|
||||
for prop_key, prop_schema in _plugin_config_properties(PLUGIN_CONFIG_SCHEMAS).items():
|
||||
if isinstance(prop_schema, Mapping) and prop_schema.get("x-sensitive"):
|
||||
sensitive_keys.add(str(prop_key))
|
||||
return sensitive_keys
|
||||
|
||||
|
||||
def is_sensitive_config_key(key: str) -> bool:
|
||||
@ -77,8 +91,9 @@ def is_sensitive_config_key(key: str) -> bool:
|
||||
REST API responses, and any future surface that round-trips raw config
|
||||
values all agree on which keys to redact.
|
||||
"""
|
||||
upper = (key or "").upper()
|
||||
return any(needle in upper for needle in _SENSITIVE_CONFIG_KEY_NEEDLES)
|
||||
key = str(key or "")
|
||||
upper = key.upper()
|
||||
return key in _plugin_sensitive_config_keys() or any(needle in upper for needle in _SENSITIVE_CONFIG_KEY_NEEDLES)
|
||||
|
||||
|
||||
def redact_sensitive_config(config: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||
@ -103,6 +118,34 @@ def redact_sensitive_config(config: Mapping[str, Any] | None) -> dict[str, Any]:
|
||||
return redacted
|
||||
|
||||
|
||||
|
||||
def normalize_runtime_config(config: BaseConfigSet | Mapping[str, Any] | str | None) -> dict[str, Any]:
|
||||
"""Return a JSON-safe config dict suitable for storage or event payloads."""
|
||||
if config is None:
|
||||
return {}
|
||||
if isinstance(config, BaseConfigSet):
|
||||
config = config.model_dump(mode="json")
|
||||
elif isinstance(config, str):
|
||||
config = json.loads(config)
|
||||
else:
|
||||
config = dict(config)
|
||||
return {key: value for key, value in json.loads(json.dumps(config, default=str)).items() if value is not None}
|
||||
|
||||
|
||||
def build_crawl_config_snapshot(
|
||||
*,
|
||||
user: Any = None,
|
||||
persona: Any = None,
|
||||
overrides: Mapping[str, Any] | None = None,
|
||||
base_config: ArchiveBoxBaseConfig | Mapping[str, object] | None = None,
|
||||
) -> dict[str, Any]:
|
||||
"""Build the frozen runtime config stored on Crawl.config at creation time."""
|
||||
effective = get_config(user=user, persona=persona, base_config=base_config, include_machine=False)
|
||||
frozen = effective.for_crawl_frozen()
|
||||
if overrides:
|
||||
frozen = get_config(base_config=frozen, overrides=overrides, include_machine=False).for_crawl_frozen()
|
||||
return frozen
|
||||
|
||||
def rprint(*args, file=None, **kwargs):
|
||||
console = _STDERR_CONSOLE if file is sys.stderr else _STDOUT_CONSOLE
|
||||
console.print(*args, **kwargs)
|
||||
@ -110,6 +153,7 @@ def rprint(*args, file=None, **kwargs):
|
||||
|
||||
class ShellConfig(BaseConfigSet):
|
||||
toml_section_header: str = "SHELL_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_CRAWL_EXECUTION)
|
||||
|
||||
DEBUG: bool = Field(default="--debug" in sys.argv)
|
||||
|
||||
@ -141,6 +185,7 @@ class ShellConfig(BaseConfigSet):
|
||||
|
||||
class StorageConfig(BaseConfigSet):
|
||||
toml_section_header: str = "STORAGE_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
|
||||
|
||||
# ARCHIVE_DIR / USERS_DIR are resolved dynamically via get_config().
|
||||
ARCHIVE_DIR: Path = Field(default=CONSTANTS.ARCHIVE_DIR)
|
||||
@ -150,34 +195,36 @@ class StorageConfig(BaseConfigSet):
|
||||
# TMP_DIR must be a local, fast, readable/writable dir by archivebox user,
|
||||
# must be a short path due to unix path length restrictions for socket files (<100 chars)
|
||||
# must be a local SSD/tmpfs for speed and because bind mounts/network mounts/FUSE dont support unix sockets
|
||||
TMP_DIR: Path = Field(default=CONSTANTS.DEFAULT_TMP_DIR)
|
||||
TMP_DIR: Path = Field(default=CONSTANTS.DEFAULT_TMP_DIR, json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
|
||||
# LIB_DIR must be a local, fast, readable/writable dir by archivebox user,
|
||||
# must be able to contain executable binaries (up to 5GB size)
|
||||
# should not be a remote/network/FUSE mount for speed reasons, otherwise extractors will be slow
|
||||
LIB_DIR: Path = Field(default=CONSTANTS.DEFAULT_LIB_DIR)
|
||||
LIB_DIR: Path = Field(default=CONSTANTS.DEFAULT_LIB_DIR, json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
|
||||
# LIB_BIN_DIR is an optional human-facing symlink convenience directory.
|
||||
# Runtime lookup must use provider-specific paths under LIB_DIR instead.
|
||||
LIB_BIN_DIR: Path = Field(default=CONSTANTS.DEFAULT_LIB_BIN_DIR)
|
||||
LIB_BIN_DIR: Path = Field(default=CONSTANTS.DEFAULT_LIB_BIN_DIR, json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
|
||||
# CUSTOM_TEMPLATES_DIR allows users to override default templates
|
||||
# defaults to DATA_DIR / 'user_templates' but can be configured
|
||||
CUSTOM_TEMPLATES_DIR: Path = Field(default=CONSTANTS.CUSTOM_TEMPLATES_DIR)
|
||||
|
||||
OUTPUT_PERMISSIONS: str = Field(default="644")
|
||||
OUTPUT_PERMISSIONS: str = Field(default="644", json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
ENFORCE_ATOMIC_WRITES: bool = Field(default=True)
|
||||
ALLOW_NO_UNIX_SOCKETS: bool = Field(default=False, alias="ARCHIVEBOX_ALLOW_NO_UNIX_SOCKETS")
|
||||
|
||||
|
||||
class GeneralConfig(BaseConfigSet):
|
||||
toml_section_header: str = "GENERAL_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
|
||||
|
||||
TAG_SEPARATOR_PATTERN: str = Field(default=r"[,]")
|
||||
|
||||
|
||||
class ServerConfig(BaseConfigSet):
|
||||
toml_section_header: str = "SERVER_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
|
||||
|
||||
SERVER_SECURITY_MODES: ClassVar[tuple[str, ...]] = (
|
||||
"safe-subdomains-fullreplay",
|
||||
@ -258,6 +305,7 @@ class ServerConfig(BaseConfigSet):
|
||||
|
||||
class DatabaseConfig(BaseConfigSet):
|
||||
toml_section_header: str = "DATABASE_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
|
||||
|
||||
DATABASE_NAME: str = Field(default=str(CONSTANTS.DATABASE_FILE), alias="ARCHIVEBOX_DATABASE_NAME")
|
||||
SQLITE_JOURNAL_MODE: str = Field(
|
||||
@ -277,6 +325,7 @@ class DatabaseConfig(BaseConfigSet):
|
||||
|
||||
class ArchivingConfig(BaseConfigSet):
|
||||
toml_section_header: str = "ARCHIVING_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_CRAWL_FROZEN)
|
||||
|
||||
PLUGINS: str = Field(
|
||||
default="",
|
||||
@ -284,6 +333,7 @@ class ArchivingConfig(BaseConfigSet):
|
||||
)
|
||||
|
||||
ONLY_NEW: bool = Field(default=True)
|
||||
INDEX_ONLY: bool = Field(default=False)
|
||||
|
||||
TIMEOUT: int = Field(default=60)
|
||||
CRAWL_MAX_URLS: int = Field(default=0)
|
||||
@ -397,6 +447,7 @@ def parse_delete_after(value) -> timedelta | None:
|
||||
|
||||
class SearchBackendConfig(BaseConfigSet):
|
||||
toml_section_header: str = "SEARCH_BACKEND_CONFIG"
|
||||
_scope: str = PrivateAttr(default=_SCOPE_SERVER)
|
||||
|
||||
SEARCH_BACKEND_ENGINE: str = Field(default="ripgrep")
|
||||
|
||||
@ -473,12 +524,93 @@ class ArchiveBoxBaseConfig(
|
||||
populate_by_name=True,
|
||||
)
|
||||
|
||||
DATA_DIR: Path = Field(default=CONSTANTS.DATA_DIR)
|
||||
ABX_RUNTIME: str = Field(default="archivebox")
|
||||
CRAWL_DIR: Path | None = Field(default=None)
|
||||
SNAP_DIR: Path | None = Field(default=None)
|
||||
DATA_DIR: Path = Field(default=CONSTANTS.DATA_DIR, json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
ABX_RUNTIME: str = Field(default="archivebox", json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
CRAWL_DIR: Path | None = Field(default=None, json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
SNAP_DIR: Path | None = Field(default=None, json_schema_extra={"scope": _SCOPE_CRAWL_EXECUTION})
|
||||
computed_config_keys: ClassVar[tuple[str, ...]] = COMPUTED_CONFIG_KEYS
|
||||
|
||||
@classmethod
|
||||
def _core_config_classes(cls) -> tuple[type[BaseConfigSet], ...]:
|
||||
return (
|
||||
ShellConfig,
|
||||
StorageConfig,
|
||||
GeneralConfig,
|
||||
ServerConfig,
|
||||
DatabaseConfig,
|
||||
ArchivingConfig,
|
||||
SearchBackendConfig,
|
||||
LDAPConfig,
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def _core_field_scope(cls, key: str) -> str | None:
|
||||
if key == "toml_section_header":
|
||||
return _SCOPE_SERVER
|
||||
for config_cls in cls._core_config_classes():
|
||||
field = config_cls.model_fields.get(key)
|
||||
if field is None:
|
||||
continue
|
||||
default_scope = str(config_cls.__private_attributes__["_scope"].default)
|
||||
extra = field.json_schema_extra
|
||||
if isinstance(extra, dict) and "scope" in extra:
|
||||
return str(extra["scope"])
|
||||
return default_scope
|
||||
if key in ArchiveBoxBaseConfig.model_fields:
|
||||
field = ArchiveBoxBaseConfig.model_fields[key]
|
||||
extra = field.json_schema_extra
|
||||
if isinstance(extra, dict) and "scope" in extra:
|
||||
return str(extra["scope"])
|
||||
return _SCOPE_SERVER
|
||||
return None
|
||||
|
||||
@classmethod
|
||||
def _plugin_field_scope(cls, key: str) -> str | None:
|
||||
for plugin_name, schema in PLUGIN_CONFIG_SCHEMAS.items():
|
||||
properties = schema.get("properties") if isinstance(schema, dict) else None
|
||||
if not isinstance(properties, dict) or key not in properties:
|
||||
continue
|
||||
prop_schema = properties.get(key) or {}
|
||||
if isinstance(prop_schema, Mapping) and prop_schema.get("x-scope"):
|
||||
return str(prop_schema["x-scope"])
|
||||
if str(plugin_name).startswith("search_backend_"):
|
||||
return _SCOPE_SERVER
|
||||
upper = key.upper()
|
||||
if (
|
||||
upper == "PATH"
|
||||
or upper.endswith("_PATH")
|
||||
or upper.endswith("_DIR")
|
||||
or upper.endswith("_BINARY")
|
||||
or upper.endswith("_CACHE")
|
||||
or upper.endswith("_COVERAGE")
|
||||
or upper.endswith("_PYTHON")
|
||||
):
|
||||
return _SCOPE_CRAWL_EXECUTION
|
||||
return _SCOPE_CRAWL_FROZEN
|
||||
return None
|
||||
|
||||
@classmethod
|
||||
def scope_for_key(cls, key: str) -> str:
|
||||
return cls._core_field_scope(key) or cls._plugin_field_scope(key) or _SCOPE_SERVER
|
||||
|
||||
def _scoped_config(self, *, include_execution: bool) -> dict[str, Any]:
|
||||
allowed_scopes = {_SCOPE_CRAWL_FROZEN}
|
||||
if include_execution:
|
||||
allowed_scopes.add(_SCOPE_CRAWL_EXECUTION)
|
||||
return {
|
||||
key: value
|
||||
for key, value in normalize_runtime_config(self).items()
|
||||
if type(self).scope_for_key(key) in allowed_scopes
|
||||
}
|
||||
|
||||
def for_crawl_execution(self) -> dict[str, Any]:
|
||||
"""Config safe to pass to crawl/snapshot hook execution."""
|
||||
return self._scoped_config(include_execution=True)
|
||||
|
||||
def for_crawl_frozen(self) -> dict[str, Any]:
|
||||
"""Config safe to persist permanently on Crawl.config."""
|
||||
return self._scoped_config(include_execution=False)
|
||||
|
||||
@model_validator(mode="after")
|
||||
def resolve_runtime_paths(self):
|
||||
self.DATA_DIR = self.DATA_DIR.expanduser().resolve()
|
||||
@ -552,12 +684,12 @@ def get_config(
|
||||
1. Explicit overrides
|
||||
2. Per-ArchiveResult config
|
||||
3. Per-snapshot config and output path
|
||||
4. Per-crawl config and output path
|
||||
5. Per-user config
|
||||
6. Per-persona derived config
|
||||
7. Current machine derived config
|
||||
8. Environment variables
|
||||
9. Config file (ArchiveBox.conf)
|
||||
4. Frozen per-crawl config and output path
|
||||
5. Per-user config (only when resolving outside a crawl)
|
||||
6. Per-persona derived config (only when resolving outside a crawl)
|
||||
7. Current machine derived config (only when resolving outside a crawl)
|
||||
8. Environment variables (only when resolving outside a crawl)
|
||||
9. Config file (ArchiveBox.conf, only when resolving outside a crawl)
|
||||
10. Plugin schema defaults
|
||||
11. Core config defaults
|
||||
"""
|
||||
@ -567,7 +699,9 @@ def get_config(
|
||||
if crawl is None and snapshot is not None:
|
||||
crawl = snapshot.crawl
|
||||
|
||||
if include_machine and machine is None:
|
||||
crawl_config_base = crawl is not None and base_config is None
|
||||
|
||||
if include_machine and machine is None and not crawl_config_base:
|
||||
try:
|
||||
from django.apps import apps
|
||||
|
||||
@ -578,15 +712,19 @@ def get_config(
|
||||
except Exception:
|
||||
machine = None
|
||||
|
||||
if persona is None and crawl is not None:
|
||||
if persona is None and crawl is not None and not crawl_config_base:
|
||||
persona = crawl.resolve_persona()
|
||||
|
||||
config_data: ConfigPayload = dict(defaults or {})
|
||||
if base_config is not None:
|
||||
base_config_payload: ConfigPayload = {}
|
||||
if crawl_config_base:
|
||||
config_data.update(dict(crawl.config or {}))
|
||||
elif base_config is not None:
|
||||
if isinstance(base_config, ArchiveBoxBaseConfig):
|
||||
config_data.update(base_config.model_dump(mode="json"))
|
||||
base_config_payload.update(base_config.model_dump(mode="json"))
|
||||
else:
|
||||
config_data.update(dict(base_config))
|
||||
base_config_payload.update(dict(base_config))
|
||||
config_data.update(base_config_payload)
|
||||
else:
|
||||
config_data.update(ArchiveBoxConfig().model_dump(mode="json"))
|
||||
legacy_permissions = permissions_from_legacy_public_flags({**BaseConfigSet.load_from_file(CONSTANTS.CONFIG_FILE), **os.environ})
|
||||
@ -595,19 +733,16 @@ def get_config(
|
||||
|
||||
scope_overrides: ConfigPayload = {}
|
||||
|
||||
if include_machine and machine is not None and machine.config:
|
||||
from archivebox.machine.models import _sanitize_machine_config
|
||||
if not crawl_config_base:
|
||||
if include_machine and machine is not None and machine.config:
|
||||
from archivebox.machine.models import _sanitize_machine_config
|
||||
|
||||
scope_overrides.update(_sanitize_machine_config(machine.config, lib_dir=config_data.get("LIB_DIR")))
|
||||
scope_overrides.update(_sanitize_machine_config(machine.config, lib_dir=config_data.get("LIB_DIR")))
|
||||
|
||||
if persona is not None:
|
||||
scope_overrides.update(persona.get_derived_config())
|
||||
if persona is not None:
|
||||
scope_overrides.update(persona.get_derived_config())
|
||||
|
||||
user_config = getattr(user, "config", None)
|
||||
if user_config:
|
||||
scope_overrides.update(user_config)
|
||||
|
||||
if crawl is not None and crawl.config:
|
||||
if crawl is not None and crawl.config and not crawl_config_base:
|
||||
scope_overrides.update(crawl.config)
|
||||
|
||||
if crawl is not None:
|
||||
@ -638,13 +773,20 @@ def get_config(
|
||||
plugin_name: schema.get("properties", {}) for plugin_name, schema in PLUGIN_CONFIG_SCHEMAS.items() if isinstance(schema, dict)
|
||||
}
|
||||
plugin_global_config = {key: str(value) if isinstance(value, Path) else value for key, value in config_data.items()}
|
||||
plugin_user_config = _plugin_user_config(scope_overrides)
|
||||
if not crawl_config_base:
|
||||
plugin_user_config = {**BaseConfigSet.load_from_file(CONSTANTS.CONFIG_FILE), **plugin_user_config}
|
||||
plugin_sections = resolve_plugin_configs(
|
||||
plugin_schemas,
|
||||
global_config=plugin_global_config,
|
||||
user_config={**BaseConfigSet.load_from_file(CONSTANTS.CONFIG_FILE), **_plugin_user_config(scope_overrides)},
|
||||
user_config=plugin_user_config,
|
||||
)
|
||||
for plugin_config in plugin_sections.values():
|
||||
config_data.update(plugin_config)
|
||||
if base_config_payload:
|
||||
config_data.update({key: value for key, value in base_config_payload.items() if key in _archivebox_config_input_names()})
|
||||
if crawl_config_base:
|
||||
config_data.update(dict(crawl.config or {}))
|
||||
config_data.update(archivebox_scope_overrides)
|
||||
|
||||
config_data["ABX_RUNTIME"] = "archivebox"
|
||||
|
||||
@ -1,6 +1,6 @@
|
||||
__package__ = "archivebox.config"
|
||||
|
||||
from pydantic import Field
|
||||
from pydantic import Field, PrivateAttr
|
||||
|
||||
from archivebox.config.configset import BaseConfigSet
|
||||
|
||||
@ -14,6 +14,7 @@ class LDAPConfig(BaseConfigSet):
|
||||
"""
|
||||
|
||||
toml_section_header: str = "LDAP_CONFIG"
|
||||
_scope: str = PrivateAttr(default="server")
|
||||
|
||||
LDAP_ENABLED: bool = Field(default=False)
|
||||
LDAP_SERVER_URI: str | None = Field(default=None)
|
||||
|
||||
@ -109,7 +109,7 @@ def get_machine_admin_url() -> str | None:
|
||||
from archivebox.machine.models import Machine
|
||||
|
||||
machine = Machine.current()
|
||||
return getattr(machine, "admin_change_url", None) or f"/admin/machine/machine/{machine.id.hex}/change/"
|
||||
return getattr(machine, "admin_change_url", None) or f"/admin/machine/machine/{machine.id}/change/"
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
@ -453,9 +453,9 @@ class PluginConfigFormMixin:
|
||||
if "array" in _schema_types(prop_schema) and isinstance(prop_schema.get("enum"), list):
|
||||
raw_value = self.data.getlist(input_name)
|
||||
|
||||
from archivebox.config.common import is_sensitive_config_key
|
||||
from archivebox.config.common import SENSITIVE_CONFIG_VALUE_REDACTED, is_sensitive_config_key
|
||||
|
||||
if (prop_schema.get("x-sensitive") or is_sensitive_config_key(config_key)) and raw_value == "":
|
||||
if (prop_schema.get("x-sensitive") or is_sensitive_config_key(config_key)) and raw_value in ("", SENSITIVE_CONFIG_VALUE_REDACTED):
|
||||
continue
|
||||
|
||||
try:
|
||||
|
||||
@ -636,12 +636,7 @@ class Snapshot(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelW
|
||||
|
||||
@classmethod
|
||||
def missing_delete_at_candidates(cls):
|
||||
from archivebox.personas.models import Persona
|
||||
|
||||
persona_ids = Persona.objects.filter(config__has_key="DELETE_AFTER").values_list("id", flat=True)
|
||||
return cls.objects.filter(delete_at__isnull=True).filter(
|
||||
Q(config__has_key="DELETE_AFTER") | Q(crawl__config__has_key="DELETE_AFTER") | Q(crawl__persona_id__in=persona_ids),
|
||||
)
|
||||
return cls.objects.filter(delete_at__isnull=True).filter(Q(config__has_key="DELETE_AFTER") | Q(crawl__config__has_key="DELETE_AFTER"))
|
||||
|
||||
@classmethod
|
||||
def is_archivebox_internal_url(cls, url: str) -> bool:
|
||||
@ -3498,14 +3493,10 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, M
|
||||
|
||||
@classmethod
|
||||
def missing_delete_at_candidates(cls):
|
||||
from archivebox.personas.models import Persona
|
||||
|
||||
persona_ids = Persona.objects.filter(config__has_key="DELETE_AFTER").values_list("id", flat=True)
|
||||
return cls.objects.filter(delete_at__isnull=True).filter(
|
||||
Q(config__has_key="DELETE_AFTER")
|
||||
| Q(snapshot__config__has_key="DELETE_AFTER")
|
||||
| Q(snapshot__crawl__config__has_key="DELETE_AFTER")
|
||||
| Q(snapshot__crawl__persona_id__in=persona_ids),
|
||||
| Q(snapshot__crawl__config__has_key="DELETE_AFTER"),
|
||||
)
|
||||
|
||||
@property
|
||||
|
||||
@ -7,6 +7,7 @@ import logging
|
||||
|
||||
|
||||
from archivebox.config import CONSTANTS
|
||||
from archivebox.misc.logging import STDERR
|
||||
|
||||
|
||||
IGNORABLE_URL_PATTERNS = [
|
||||
@ -182,6 +183,7 @@ SETTINGS_LOGGING = {
|
||||
"level": "DEBUG",
|
||||
"markup": False,
|
||||
"rich_tracebacks": False, # Use standard Python tracebacks (no frame/box)
|
||||
"console": STDERR,
|
||||
"filters": ["noisyrequestsfilter", "daphneclosetimeout", "asynciocancelledshield", "stripansi"],
|
||||
},
|
||||
"logfile": {
|
||||
|
||||
@ -33,7 +33,7 @@ from admin_data_views.utils import render_with_table_view, render_with_item_view
|
||||
from abx_dl.events import PROCESS_EXIT_SKIPPED
|
||||
|
||||
from archivebox.config import CONSTANTS, CONSTANTS_CONFIG, VERSION
|
||||
from archivebox.config.common import get_config, get_all_configs
|
||||
from archivebox.config.common import get_config, get_all_configs, is_sensitive_config_key
|
||||
from archivebox.config.configset import BaseConfigSet
|
||||
from archivebox.misc.paginators import CountlessPaginator
|
||||
from archivebox.misc.util import (
|
||||
@ -1177,8 +1177,6 @@ class AddView(UserPassesTestMixin, FormView):
|
||||
required_search_plugin = f"search_backend_{request_config.SEARCH_BACKEND_ENGINE}".strip()
|
||||
can_override_crawl_config = self._can_override_crawl_config()
|
||||
plugin_configs = discover_plugin_configs() if can_override_crawl_config else {}
|
||||
from archivebox.config.common import is_sensitive_config_key
|
||||
|
||||
sensitive_keys = {
|
||||
str(config_key)
|
||||
for schema in plugin_configs.values()
|
||||
@ -1304,7 +1302,8 @@ class AddView(UserPassesTestMixin, FormView):
|
||||
config = {}
|
||||
if plugins:
|
||||
config["PLUGINS"] = plugins
|
||||
effective_config = get_config(persona=persona, user=self.request.user) if persona else get_config(user=self.request.user)
|
||||
request_user = self.request.user if self.request.user.is_authenticated else None
|
||||
effective_config = get_config(persona=persona, user=request_user) if persona else get_config(user=request_user)
|
||||
if crawl_max_concurrent_snapshots != int(effective_config.CRAWL_MAX_CONCURRENT_SNAPSHOTS):
|
||||
config["CRAWL_MAX_CONCURRENT_SNAPSHOTS"] = crawl_max_concurrent_snapshots
|
||||
if delete_after != str(effective_config.DELETE_AFTER):
|
||||
@ -1352,6 +1351,7 @@ class AddView(UserPassesTestMixin, FormView):
|
||||
template=crawl,
|
||||
schedule=schedule,
|
||||
is_enabled=True,
|
||||
config=config,
|
||||
label=crawl.label,
|
||||
notes=f"Auto-created from add page. {notes}".strip(),
|
||||
created_by_id=created_by_id,
|
||||
@ -2508,13 +2508,6 @@ def find_config_type(key: str) -> str:
|
||||
return "str"
|
||||
|
||||
|
||||
def key_is_safe(key: str) -> bool:
|
||||
for term in ("key", "password", "secret", "token"):
|
||||
if term in key.lower():
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def find_config_source(key: str, merged_config: dict) -> str:
|
||||
"""Determine where a config value comes from."""
|
||||
from archivebox.machine.models import Machine
|
||||
@ -2604,7 +2597,7 @@ def live_config_list_view(request: HttpRequest, **kwargs) -> TableContext:
|
||||
|
||||
# Use merged config value (includes machine overrides)
|
||||
actual_value = merged_config.get(key, getattr(section, key, None))
|
||||
rows["Value"].append(mark_safe(f"<code>{actual_value}</code>") if key_is_safe(key) else "******** (redacted)")
|
||||
rows["Value"].append(mark_safe(f"<code>{actual_value}</code>") if not is_sensitive_config_key(key) else "******** (redacted)")
|
||||
|
||||
# Show where the value comes from
|
||||
source = find_config_source(key, merged_config)
|
||||
@ -2624,7 +2617,7 @@ def live_config_list_view(request: HttpRequest, **kwargs) -> TableContext:
|
||||
rows["Section"].append(section) # section.replace('_', ' ').title().replace(' Config', '')
|
||||
rows["Key"].append(ItemLink(key, key=key))
|
||||
rows["Type"].append(format_html("<code>{}</code>", getattr(type(CONSTANTS_CONFIG[key]), "__name__", str(CONSTANTS_CONFIG[key]))))
|
||||
rows["Value"].append(format_html("<code>{}</code>", CONSTANTS_CONFIG[key]) if key_is_safe(key) else "******** (redacted)")
|
||||
rows["Value"].append(format_html("<code>{}</code>", CONSTANTS_CONFIG[key]) if not is_sensitive_config_key(key) else "******** (redacted)")
|
||||
rows["Source"].append(mark_safe('<code style="color: gray">Constant</code>'))
|
||||
rows["Default"].append(
|
||||
mark_safe(
|
||||
@ -2655,16 +2648,16 @@ def live_config_value_view(request: HttpRequest, key: str, **kwargs) -> ItemCont
|
||||
|
||||
# Environment variable
|
||||
if key in os.environ:
|
||||
sources_info.append(("Environment", os.environ[key] if key_is_safe(key) else "********", "blue"))
|
||||
sources_info.append(("Environment", os.environ[key] if not is_sensitive_config_key(key) else "********", "blue"))
|
||||
|
||||
# Machine config
|
||||
machine = None
|
||||
machine_admin_url = None
|
||||
try:
|
||||
machine = Machine.current()
|
||||
machine_admin_url = f"/admin/machine/machine/{machine.id.hex}/change/"
|
||||
machine_admin_url = f"/admin/machine/machine/{machine.id}/change/"
|
||||
if machine.config and key in machine.config:
|
||||
sources_info.append(("Machine", machine.config[key] if key_is_safe(key) else "********", "purple"))
|
||||
sources_info.append(("Machine", machine.config[key] if not is_sensitive_config_key(key) else "********", "purple"))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@ -2680,8 +2673,12 @@ def live_config_value_view(request: HttpRequest, key: str, **kwargs) -> ItemCont
|
||||
sources_info.append(("Default", default_val, "gray"))
|
||||
|
||||
# Final computed value
|
||||
final_value = merged_config.get(key, CONFIGS.get(key, None))
|
||||
if not key_is_safe(key):
|
||||
config_source = find_config_source(key, merged_config)
|
||||
if config_source == "Environment":
|
||||
final_value = get_config(include_machine=False).model_dump(mode="json").get(key, CONFIGS.get(key, None))
|
||||
else:
|
||||
final_value = merged_config.get(key, CONFIGS.get(key, None))
|
||||
if is_sensitive_config_key(key):
|
||||
final_value = "********"
|
||||
|
||||
# Build sources display
|
||||
@ -2714,7 +2711,7 @@ def live_config_value_view(request: HttpRequest, key: str, **kwargs) -> ItemCont
|
||||
"Key": key,
|
||||
"Type": find_config_type(key),
|
||||
"Value": final_value,
|
||||
"Currently read from": find_config_source(key, merged_config),
|
||||
"Currently read from": config_source,
|
||||
},
|
||||
"help_texts": {
|
||||
"Key": mark_safe(f"""
|
||||
@ -2731,7 +2728,7 @@ def live_config_value_view(request: HttpRequest, key: str, **kwargs) -> ItemCont
|
||||
"Value": mark_safe(f'''
|
||||
{
|
||||
'<b style="color: red">Value is redacted for your security. (Passwords, secrets, API tokens, etc. cannot be viewed in the Web UI)</b><br/><br/>'
|
||||
if not key_is_safe(key)
|
||||
if is_sensitive_config_key(key)
|
||||
else ""
|
||||
}
|
||||
<br/><hr/><br/>
|
||||
@ -2744,12 +2741,12 @@ def live_config_value_view(request: HttpRequest, key: str, **kwargs) -> ItemCont
|
||||
<code>archivebox config --set {key}="{
|
||||
val.strip("'")
|
||||
if (val := find_config_default(key))
|
||||
else (str(final_value if key_is_safe(key) else "********")).strip("'")
|
||||
else (str(final_value if not is_sensitive_config_key(key) else "********")).strip("'")
|
||||
}"</code>
|
||||
</p>
|
||||
'''),
|
||||
"Currently read from": mark_safe(f"""
|
||||
The value shown in the "Value" field comes from the <b>{find_config_source(key, merged_config)}</b> source.
|
||||
The value shown in the "Value" field comes from the <b>{config_source}</b> source.
|
||||
<br/><br/>
|
||||
Priority order (highest to lowest):
|
||||
<ol>
|
||||
|
||||
@ -0,0 +1,33 @@
|
||||
from django.db import migrations
|
||||
|
||||
|
||||
def freeze_existing_crawl_configs(apps, schema_editor):
|
||||
from archivebox.config.common import build_crawl_config_snapshot
|
||||
from archivebox.personas.models import Persona
|
||||
from django.contrib.auth import get_user_model
|
||||
|
||||
Crawl = apps.get_model("crawls", "Crawl")
|
||||
User = get_user_model()
|
||||
db_alias = schema_editor.connection.alias
|
||||
|
||||
for crawl in Crawl.objects.using(db_alias).select_related("persona", "created_by").iterator(chunk_size=200):
|
||||
current_config = dict(crawl.config or {})
|
||||
persona = Persona.objects.using(db_alias).filter(pk=crawl.persona_id).first()
|
||||
user = User.objects.using(db_alias).filter(pk=crawl.created_by_id).first()
|
||||
frozen_config = build_crawl_config_snapshot(
|
||||
user=user,
|
||||
persona=persona,
|
||||
overrides=current_config,
|
||||
)
|
||||
if frozen_config != current_config:
|
||||
Crawl.objects.using(db_alias).filter(pk=crawl.pk).update(config=frozen_config)
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("crawls", "0017_drop_stale_crawl_limit_columns"),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.RunPython(freeze_existing_crawl_configs, migrations.RunPython.noop),
|
||||
]
|
||||
25
archivebox/crawls/migrations/0019_crawlschedule_config.py
Normal file
25
archivebox/crawls/migrations/0019_crawlschedule_config.py
Normal file
@ -0,0 +1,25 @@
|
||||
from django.db import migrations, models
|
||||
|
||||
|
||||
def copy_template_config_to_schedule(apps, schema_editor):
|
||||
CrawlSchedule = apps.get_model("crawls", "CrawlSchedule")
|
||||
db_alias = schema_editor.connection.alias
|
||||
|
||||
for schedule in CrawlSchedule.objects.using(db_alias).select_related("template").iterator(chunk_size=200):
|
||||
template_config = dict(schedule.template.config or {}) if schedule.template_id else {}
|
||||
CrawlSchedule.objects.using(db_alias).filter(pk=schedule.pk).update(config=template_config)
|
||||
|
||||
|
||||
class Migration(migrations.Migration):
|
||||
dependencies = [
|
||||
("crawls", "0018_freeze_crawl_config_snapshots"),
|
||||
]
|
||||
|
||||
operations = [
|
||||
migrations.AddField(
|
||||
model_name="crawlschedule",
|
||||
name="config",
|
||||
field=models.JSONField(blank=True, default=dict, null=True),
|
||||
),
|
||||
migrations.RunPython(copy_template_config_to_schedule, migrations.RunPython.noop),
|
||||
]
|
||||
@ -50,6 +50,7 @@ class CrawlSchedule(ModelWithUUID, ModelWithNotes):
|
||||
template: "Crawl" = models.ForeignKey("Crawl", on_delete=models.CASCADE, null=False, blank=False) # type: ignore
|
||||
schedule = models.CharField(max_length=64, blank=False, null=False)
|
||||
is_enabled = models.BooleanField(default=True)
|
||||
config = models.JSONField(default=dict, null=True, blank=True)
|
||||
label = models.CharField(max_length=64, blank=True, null=False, default="")
|
||||
notes = models.TextField(blank=True, null=False, default="")
|
||||
|
||||
@ -102,13 +103,17 @@ class CrawlSchedule(ModelWithUUID, ModelWithNotes):
|
||||
return self.is_enabled and self.next_run_at <= now
|
||||
|
||||
def enqueue(self, queued_at=None) -> "Crawl":
|
||||
from archivebox.config.common import build_crawl_config_snapshot
|
||||
|
||||
queued_at = queued_at or timezone.now()
|
||||
template = self.template
|
||||
label = template.label or self.label
|
||||
persona = template.persona if template.persona_id else None
|
||||
user = template.created_by if template.created_by_id else None
|
||||
|
||||
return Crawl.objects.create(
|
||||
urls=template.urls,
|
||||
config=template.config or {},
|
||||
config=build_crawl_config_snapshot(user=user, persona=persona, overrides=self.config or {}),
|
||||
max_depth=template.max_depth,
|
||||
tags_str=template.tags_str,
|
||||
persona_id=template.persona_id,
|
||||
@ -273,10 +278,7 @@ class Crawl(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelWith
|
||||
|
||||
@classmethod
|
||||
def missing_delete_at_candidates(cls):
|
||||
from archivebox.personas.models import Persona
|
||||
|
||||
persona_ids = Persona.objects.filter(config__has_key="DELETE_AFTER").values_list("id", flat=True)
|
||||
return cls.objects.filter(delete_at__isnull=True).filter(Q(config__has_key="DELETE_AFTER") | Q(persona_id__in=persona_ids))
|
||||
return cls.objects.filter(delete_at__isnull=True, config__has_key="DELETE_AFTER")
|
||||
|
||||
def save(self, *args, **kwargs):
|
||||
update_fields = kwargs.get("update_fields")
|
||||
@ -287,11 +289,16 @@ class Crawl(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelWith
|
||||
previous_tag_names = set(self.parse_tag_names(old_crawl.tags_str or ""))
|
||||
|
||||
config = dict(self.config or {})
|
||||
is_new = self._state.adding or old_crawl is None
|
||||
persona = self.persona if self.persona_id else None
|
||||
user = self.created_by if self.created_by_id else None
|
||||
if is_new:
|
||||
from archivebox.config.common import build_crawl_config_snapshot
|
||||
|
||||
config = build_crawl_config_snapshot(user=user, persona=persona, overrides=config)
|
||||
if str(config.get("PERMISSIONS") or "").strip().lower() not in PERMISSIONS_VALUES:
|
||||
from archivebox.config.common import get_config
|
||||
|
||||
persona = self.persona if self.persona_id else None
|
||||
user = self.created_by if self.created_by_id else None
|
||||
config["PERMISSIONS"] = normalize_permissions(get_config(persona=persona, user=user, include_machine=True).PERMISSIONS)
|
||||
if "CRAWL_MAX_CONCURRENT_SNAPSHOTS" in config:
|
||||
raw_concurrency = config["CRAWL_MAX_CONCURRENT_SNAPSHOTS"]
|
||||
|
||||
@ -324,6 +324,7 @@ def run_hook(
|
||||
|
||||
config_scope = {key.removeprefix("config_"): kwargs.pop(key) for key in list(kwargs) if key.startswith("config_")}
|
||||
resolved_config = get_config(overrides=_config_to_overrides(config), **config_scope)
|
||||
hook_config = resolved_config.for_crawl_execution()
|
||||
|
||||
# Auto-detect timeout from plugin config if not explicitly provided
|
||||
if timeout is None:
|
||||
@ -463,7 +464,7 @@ def run_hook(
|
||||
"SNAP_DIR",
|
||||
"CRAWL_DIR",
|
||||
}
|
||||
for key, value in resolved_config.items():
|
||||
for key, value in hook_config.items():
|
||||
if key in SKIP_KEYS:
|
||||
continue # Already handled specially above, don't overwrite
|
||||
if value is None:
|
||||
|
||||
@ -146,11 +146,14 @@ class ProcessService(BaseService):
|
||||
self._ensure_completed_worker()
|
||||
await self._completed_queue.put(event)
|
||||
|
||||
async def on_CrawlCleanupEvent__flush_completed(self, event: CrawlCleanupEvent) -> None:
|
||||
async def flush_completed(self) -> None:
|
||||
await self._completed_queue.join()
|
||||
|
||||
async def on_CrawlCleanupEvent__flush_completed(self, event: CrawlCleanupEvent) -> None:
|
||||
await self.flush_completed()
|
||||
|
||||
async def on_CrawlCompletedEvent__flush_completed(self, event: CrawlCompletedEvent) -> None:
|
||||
await self._completed_queue.join()
|
||||
await self.flush_completed()
|
||||
|
||||
async def _save_completed_process_to_db(self, event: ProcessCompletedEvent) -> None:
|
||||
from archivebox.machine.models import Process
|
||||
|
||||
@ -58,7 +58,7 @@ from abxbus import BaseEvent
|
||||
from abxbus.event_bus import EventBus, get_current_event, in_handler_context
|
||||
from abxbus.event_handler import EventHandlerAbortedError, EventHandlerCancelledError
|
||||
|
||||
from archivebox.config.configset import BaseConfigSet
|
||||
from archivebox.config.common import ArchiveBoxBaseConfig
|
||||
from archivebox.core.recovery_util import recover_orchestrator_state
|
||||
from archivebox.misc.db import run_db_analyze_batch
|
||||
from archivebox.core.shutdown_util import foreground_shutdown_signals
|
||||
@ -122,16 +122,12 @@ def _count_selected_hooks(plugins: dict[str, Plugin], selected_plugins: list[str
|
||||
return sum(1 for plugin in selected.values() for hook in plugin.hooks if "CrawlSetup" in hook.name or "Snapshot" in hook.name)
|
||||
|
||||
|
||||
def _normalize_runtime_config(config: BaseConfigSet | Mapping[str, Any] | str | None) -> dict[str, Any]:
|
||||
if config is None:
|
||||
return {}
|
||||
if isinstance(config, BaseConfigSet):
|
||||
config = config.model_dump(mode="json")
|
||||
elif isinstance(config, str):
|
||||
config = json.loads(config)
|
||||
else:
|
||||
config = dict(config)
|
||||
return {key: value for key, value in json.loads(json.dumps(config, default=str)).items() if value is not None}
|
||||
def _normalize_runtime_config(config: ArchiveBoxBaseConfig | Mapping[str, Any] | str | None) -> dict[str, Any]:
|
||||
from archivebox.config.common import normalize_runtime_config
|
||||
|
||||
if isinstance(config, ArchiveBoxBaseConfig):
|
||||
return config.for_crawl_execution()
|
||||
return normalize_runtime_config(config)
|
||||
|
||||
|
||||
def _runner_task_context() -> contextvars.Context:
|
||||
@ -1168,7 +1164,7 @@ async def _run_binary(binary_id: str) -> None:
|
||||
config["ABX_RUNTIME"] = "archivebox"
|
||||
config = _normalize_runtime_config(config)
|
||||
bus = create_bus(name=_bus_name("ArchiveBox_binary", str(binary.id)), total_timeout=1800.0)
|
||||
PersistedProcessService(bus)
|
||||
process_service = PersistedProcessService(bus)
|
||||
BinaryService(bus)
|
||||
TagService(bus)
|
||||
ArchiveResultService(bus)
|
||||
@ -1202,6 +1198,7 @@ async def _run_binary(binary_id: str) -> None:
|
||||
).now(first_result=True)
|
||||
finally:
|
||||
await bus.wait_until_idle()
|
||||
await process_service.flush_completed()
|
||||
|
||||
|
||||
def run_binary(binary_id: str) -> None:
|
||||
|
||||
@ -484,7 +484,7 @@ def build_test_env(port: int, **extra: str) -> dict[str, str]:
|
||||
"USE_COLOR": "False",
|
||||
"SHOW_PROGRESS": "False",
|
||||
"TIMEOUT": "30",
|
||||
"URL_ALLOWLIST": r"127\.0\.0\.1[:/].*",
|
||||
"URL_ALLOWLIST": r"127\.0\.0\.1[:/].*|example\.com",
|
||||
"SAVE_ARCHIVEDOTORG": "False",
|
||||
"SAVE_TITLE": "False",
|
||||
"SAVE_FAVICON": "False",
|
||||
|
||||
@ -35,6 +35,7 @@ def disable_extractors_dict():
|
||||
"SAVE_ARCHIVEDOTORG": "false",
|
||||
"SAVE_TITLE": "false",
|
||||
"SAVE_FAVICON": "false",
|
||||
"PLUGINS": "__archivebox_test_no_plugins__",
|
||||
},
|
||||
)
|
||||
return env
|
||||
|
||||
@ -269,7 +269,7 @@ def test_add_records_selected_persona_on_crawl(tmp_path, process, disable_extrac
|
||||
crawl = Crawl.objects.get()
|
||||
|
||||
assert crawl.persona_id
|
||||
assert crawl.config.get("DEFAULT_PERSONA") is None
|
||||
assert crawl.config["ACTIVE_PERSONA"] == "Default"
|
||||
assert (tmp_path / "personas" / "Default" / "chrome_profile").is_dir()
|
||||
|
||||
|
||||
|
||||
@ -199,6 +199,7 @@ def test_crawl_create_stdout_pipes_into_run(initialized_archive):
|
||||
create_stdout, create_stderr, create_code = run_archivebox_cmd(
|
||||
["crawl", "create", url],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert create_code == 0, create_stderr
|
||||
_assert_stdout_is_jsonl_only(create_stdout)
|
||||
@ -231,6 +232,7 @@ def test_snapshot_list_stdout_pipes_into_run(initialized_archive):
|
||||
create_stdout, create_stderr, create_code = run_archivebox_cmd(
|
||||
["snapshot", "create", url],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert create_code == 0, create_stderr
|
||||
snapshot = next(record for record in parse_jsonl_output(create_stdout) if record.get("type") == "Snapshot")
|
||||
@ -238,11 +240,13 @@ def test_snapshot_list_stdout_pipes_into_run(initialized_archive):
|
||||
list_stdout, list_stderr, list_code = run_archivebox_cmd(
|
||||
["snapshot", "list", "--status=queued", f"--url__icontains={snapshot['id']}"],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
if list_code != 0 or not parse_jsonl_output(list_stdout):
|
||||
list_stdout, list_stderr, list_code = run_archivebox_cmd(
|
||||
["snapshot", "list", f"--url__icontains={url}"],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert list_code == 0, list_stderr
|
||||
_assert_stdout_is_jsonl_only(list_stdout)
|
||||
@ -272,6 +276,7 @@ def test_archiveresult_list_stdout_pipes_into_run(initialized_archive):
|
||||
snapshot_stdout, snapshot_stderr, snapshot_code = run_archivebox_cmd(
|
||||
["snapshot", "create", url],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert snapshot_code == 0, snapshot_stderr
|
||||
|
||||
@ -279,6 +284,7 @@ def test_archiveresult_list_stdout_pipes_into_run(initialized_archive):
|
||||
["archiveresult", "create", "--plugin=favicon"],
|
||||
stdin=snapshot_stdout,
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert ar_create_code == 0, ar_create_stderr
|
||||
|
||||
@ -293,6 +299,7 @@ def test_archiveresult_list_stdout_pipes_into_run(initialized_archive):
|
||||
list_stdout, list_stderr, list_code = run_archivebox_cmd(
|
||||
["archiveresult", "list", "--plugin=favicon"],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert list_code == 0, list_stderr
|
||||
_assert_stdout_is_jsonl_only(list_stdout)
|
||||
@ -348,6 +355,7 @@ def test_multi_stage_pipeline_into_run(initialized_archive):
|
||||
crawl_stdout, crawl_stderr, crawl_code = run_archivebox_cmd(
|
||||
["crawl", "create", url],
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert crawl_code == 0, crawl_stderr
|
||||
_assert_stdout_is_jsonl_only(crawl_stdout)
|
||||
@ -356,6 +364,7 @@ def test_multi_stage_pipeline_into_run(initialized_archive):
|
||||
["snapshot", "create"],
|
||||
stdin=crawl_stdout,
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert snapshot_code == 0, snapshot_stderr
|
||||
_assert_stdout_is_jsonl_only(snapshot_stdout)
|
||||
@ -364,6 +373,7 @@ def test_multi_stage_pipeline_into_run(initialized_archive):
|
||||
["archiveresult", "create", "--plugin=favicon"],
|
||||
stdin=snapshot_stdout,
|
||||
data_dir=initialized_archive,
|
||||
env=PIPE_TEST_ENV,
|
||||
)
|
||||
assert archiveresult_code == 0, archiveresult_stderr
|
||||
_assert_stdout_is_jsonl_only(archiveresult_stdout)
|
||||
|
||||
200
archivebox/tests/test_frozen_crawl_config.py
Normal file
200
archivebox/tests/test_frozen_crawl_config.py
Normal file
@ -0,0 +1,200 @@
|
||||
import pytest
|
||||
from django.test import RequestFactory
|
||||
from django.utils import timezone
|
||||
|
||||
pytestmark = pytest.mark.django_db(transaction=True)
|
||||
|
||||
|
||||
SENSITIVE_SECRET = "raw-twocaptcha-secret-for-frozen-crawl-test"
|
||||
UPDATED_SECRET = "updated-secret-that-must-not-affect-old-crawl"
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def archivebox_db(initialized_archive):
|
||||
from archivebox.tests.test_orm_helpers import use_archivebox_db
|
||||
|
||||
with use_archivebox_db(initialized_archive):
|
||||
yield initialized_archive
|
||||
|
||||
|
||||
def _user(username="frozen-config-admin"):
|
||||
from django.contrib.auth import get_user_model
|
||||
|
||||
return get_user_model().objects.create_superuser(
|
||||
username=username,
|
||||
email=f"{username}@example.com",
|
||||
password="testpassword",
|
||||
)
|
||||
|
||||
|
||||
def _persona(user, *, name="Frozen Persona", secret=SENSITIVE_SECRET, user_agent="Frozen UA"):
|
||||
from archivebox.personas.models import Persona
|
||||
|
||||
persona = Persona.objects.create(
|
||||
name=name,
|
||||
created_by=user,
|
||||
config={
|
||||
"PERMISSIONS": "private",
|
||||
"USER_AGENT": user_agent,
|
||||
"TWOCAPTCHA_API_KEY": secret,
|
||||
"DELETE_AFTER": "2h",
|
||||
},
|
||||
)
|
||||
persona.ensure_dirs()
|
||||
return persona
|
||||
|
||||
|
||||
def test_crawl_save_freezes_full_raw_persona_config_and_redacts_public_serialization(archivebox_db):
|
||||
from archivebox.config.common import SENSITIVE_CONFIG_VALUE_REDACTED, get_config
|
||||
from archivebox.crawls.models import Crawl
|
||||
|
||||
user = _user()
|
||||
persona = _persona(user)
|
||||
|
||||
crawl = Crawl.objects.create(
|
||||
urls="https://example.com/frozen",
|
||||
persona=persona,
|
||||
created_by=user,
|
||||
config={"CRAWL_MAX_CONCURRENT_SNAPSHOTS": 3},
|
||||
status=Crawl.StatusChoices.QUEUED,
|
||||
retry_at=timezone.now(),
|
||||
)
|
||||
|
||||
assert "TIMEOUT" in crawl.config
|
||||
assert "CHECK_SSL_VALIDITY" in crawl.config
|
||||
assert crawl.config["USER_AGENT"] == "Frozen UA"
|
||||
assert crawl.config["TWOCAPTCHA_API_KEY"] == SENSITIVE_SECRET
|
||||
assert crawl.config["CRAWL_MAX_CONCURRENT_SNAPSHOTS"] == 3
|
||||
assert "CRAWL_DIR" not in crawl.config
|
||||
assert "SNAP_DIR" not in crawl.config
|
||||
assert "DEBUG" not in crawl.config
|
||||
assert "SECRET_KEY" not in crawl.config
|
||||
assert "PUBLIC_ADD_VIEW" not in crawl.config
|
||||
assert "DATABASE_NAME" not in crawl.config
|
||||
|
||||
persona.config["USER_AGENT"] = "Mutated UA"
|
||||
persona.config["TWOCAPTCHA_API_KEY"] = UPDATED_SECRET
|
||||
persona.save(update_fields=["config"])
|
||||
|
||||
runtime_config = get_config(crawl=crawl)
|
||||
assert runtime_config.USER_AGENT == "Frozen UA"
|
||||
assert runtime_config.TWOCAPTCHA_API_KEY == SENSITIVE_SECRET
|
||||
execution_config = runtime_config.for_crawl_execution()
|
||||
assert execution_config["DEBUG"] is False
|
||||
assert execution_config["CRAWL_DIR"] == str(crawl.output_dir)
|
||||
assert "SECRET_KEY" not in execution_config
|
||||
assert "PUBLIC_ADD_VIEW" not in execution_config
|
||||
assert "DATABASE_NAME" not in execution_config
|
||||
|
||||
public_json = crawl.to_json()
|
||||
assert public_json["config"]["TWOCAPTCHA_API_KEY"] == SENSITIVE_CONFIG_VALUE_REDACTED
|
||||
assert SENSITIVE_SECRET not in str(public_json)
|
||||
|
||||
|
||||
def test_snapshot_config_overlays_frozen_crawl_without_re_reading_persona(archivebox_db):
|
||||
from archivebox.config.common import get_config
|
||||
from archivebox.core.models import Snapshot
|
||||
from archivebox.crawls.models import Crawl
|
||||
|
||||
user = _user("frozen-config-snapshot-admin")
|
||||
persona = _persona(user, name="Frozen Snapshot Persona", user_agent="Crawl UA")
|
||||
crawl = Crawl.objects.create(urls="https://example.com/root", persona=persona, created_by=user, config={"TIMEOUT": 11})
|
||||
snapshot = Snapshot.objects.create(
|
||||
url="https://example.com/root",
|
||||
crawl=crawl,
|
||||
config={"TIMEOUT": 22, "ANTHROPIC_API_KEY": "snapshot-secret"},
|
||||
)
|
||||
|
||||
persona.config["TIMEOUT"] = 99
|
||||
persona.save(update_fields=["config"])
|
||||
|
||||
runtime_config = get_config(crawl=crawl, snapshot=snapshot)
|
||||
assert runtime_config.USER_AGENT == "Crawl UA"
|
||||
assert runtime_config.TIMEOUT == 22
|
||||
assert runtime_config.ANTHROPIC_API_KEY == "snapshot-secret"
|
||||
assert snapshot.config == {"TIMEOUT": 22, "ANTHROPIC_API_KEY": "snapshot-secret"}
|
||||
|
||||
|
||||
def test_config_scopes_are_derived_from_section_and_field_metadata():
|
||||
from archivebox.config.common import ArchiveBoxConfig
|
||||
|
||||
assert ArchiveBoxConfig.scope_for_key("TIMEOUT") == "crawl_frozen"
|
||||
assert ArchiveBoxConfig.scope_for_key("DEBUG") == "crawl_execution"
|
||||
assert ArchiveBoxConfig.scope_for_key("CRAWL_DIR") == "crawl_execution"
|
||||
assert ArchiveBoxConfig.scope_for_key("SECRET_KEY") == "server"
|
||||
assert ArchiveBoxConfig.scope_for_key("DATABASE_NAME") == "server"
|
||||
|
||||
|
||||
def test_api_create_and_cli_add_store_full_frozen_config(archivebox_db):
|
||||
from archivebox.api.v1_crawls import CrawlCreateSchema, CrawlSchema, create_crawl
|
||||
from archivebox.cli.archivebox_add import add
|
||||
from archivebox.config.common import SENSITIVE_CONFIG_VALUE_REDACTED
|
||||
|
||||
user = _user("frozen-config-api-admin")
|
||||
request = RequestFactory().post("/api/v1/crawls")
|
||||
request.user = user
|
||||
|
||||
api_crawl = create_crawl(
|
||||
request,
|
||||
CrawlCreateSchema(
|
||||
urls=["https://example.com/api"],
|
||||
max_depth=0,
|
||||
tags=[],
|
||||
tags_str="",
|
||||
label="API frozen config",
|
||||
notes="",
|
||||
config={"TWOCAPTCHA_API_KEY": SENSITIVE_SECRET, "TIMEOUT": 33, "SECRET_KEY": "must-not-freeze", "PUBLIC_ADD_VIEW": True},
|
||||
),
|
||||
)
|
||||
assert "CHECK_SSL_VALIDITY" in api_crawl.config
|
||||
assert api_crawl.config["TIMEOUT"] == 33
|
||||
assert api_crawl.config["TWOCAPTCHA_API_KEY"] == SENSITIVE_SECRET
|
||||
assert "SECRET_KEY" not in api_crawl.config
|
||||
assert "PUBLIC_ADD_VIEW" not in api_crawl.config
|
||||
assert CrawlSchema.resolve_config(api_crawl)["TWOCAPTCHA_API_KEY"] == SENSITIVE_CONFIG_VALUE_REDACTED
|
||||
|
||||
cli_crawl, _snapshots = add(
|
||||
"https://example.com/cli",
|
||||
bg=True,
|
||||
created_by_id=user.pk,
|
||||
config={"TWOCAPTCHA_API_KEY": SENSITIVE_SECRET, "TIMEOUT": 44},
|
||||
)
|
||||
assert "CHECK_SSL_VALIDITY" in cli_crawl.config
|
||||
assert cli_crawl.config["TIMEOUT"] == 44
|
||||
assert cli_crawl.config["TWOCAPTCHA_API_KEY"] == SENSITIVE_SECRET
|
||||
|
||||
|
||||
def test_schedule_enqueue_refreezes_using_current_template_persona_defaults(archivebox_db):
|
||||
from archivebox.crawls.models import Crawl, CrawlSchedule
|
||||
|
||||
user = _user("frozen-config-schedule-admin")
|
||||
persona = _persona(user, name="Frozen Schedule Persona", user_agent="Initial schedule UA")
|
||||
template = Crawl.objects.create(
|
||||
urls="https://example.com/scheduled",
|
||||
persona=persona,
|
||||
created_by=user,
|
||||
config={"TIMEOUT": 55, "SECRET_KEY": "template-secret-must-not-freeze", "PUBLIC_ADD_VIEW": True},
|
||||
status=Crawl.StatusChoices.PAUSED,
|
||||
)
|
||||
schedule = CrawlSchedule.objects.create(
|
||||
template=template,
|
||||
schedule="daily",
|
||||
created_by=user,
|
||||
config={"TIMEOUT": 55, "SECRET_KEY": "schedule-secret-must-not-freeze", "PUBLIC_ADD_VIEW": True},
|
||||
)
|
||||
|
||||
assert schedule.config["TIMEOUT"] == 55
|
||||
assert "SECRET_KEY" in schedule.config
|
||||
|
||||
persona.config["USER_AGENT"] = "Current schedule UA"
|
||||
persona.config["TWOCAPTCHA_API_KEY"] = UPDATED_SECRET
|
||||
persona.save(update_fields=["config"])
|
||||
|
||||
child = schedule.enqueue()
|
||||
assert child.config["TIMEOUT"] == 55
|
||||
assert child.config["USER_AGENT"] == "Current schedule UA"
|
||||
assert child.config["TWOCAPTCHA_API_KEY"] == UPDATED_SECRET
|
||||
assert "SECRET_KEY" not in child.config
|
||||
assert "PUBLIC_ADD_VIEW" not in child.config
|
||||
assert template.config["USER_AGENT"] == "Initial schedule UA"
|
||||
assert template.config["TWOCAPTCHA_API_KEY"] == SENSITIVE_SECRET
|
||||
@ -184,7 +184,7 @@ def test_add_view_creates_crawl_with_tag_and_url_filter_overrides(client, admin_
|
||||
assert crawl.config["CRAWL_MAX_CONCURRENT_SNAPSHOTS"] == 5
|
||||
assert crawl.config["URL_ALLOWLIST"] == "example.com\n*.example.com"
|
||||
assert crawl.config["URL_DENYLIST"] == "cdn.example.com"
|
||||
assert "ONLY_NEW" not in crawl.config
|
||||
assert crawl.config["ONLY_NEW"] is True
|
||||
|
||||
|
||||
def test_add_view_unchecked_only_new_sets_crawl_override(client, admin_user, monkeypatch):
|
||||
@ -253,7 +253,7 @@ def test_add_view_selected_persona_wins_over_stale_config_override(client, admin
|
||||
crawl = Crawl.objects.order_by("-created_at").first()
|
||||
assert crawl is not None
|
||||
assert crawl.persona_id == private_persona.id
|
||||
assert "DEFAULT_PERSONA" not in crawl.config
|
||||
assert crawl.config["ACTIVE_PERSONA"] == "Private"
|
||||
assert crawl.resolve_persona() == private_persona
|
||||
runtime_config = get_config(crawl=crawl)
|
||||
assert runtime_config.ACTIVE_PERSONA == "Private"
|
||||
@ -342,11 +342,11 @@ def test_add_view_public_submission_ignores_plugin_and_custom_config(client, adm
|
||||
assert crawl.config["CRAWL_MAX_CONCURRENT_SNAPSHOTS"] == 2
|
||||
assert crawl.config["URL_ALLOWLIST"] == "example.com"
|
||||
assert crawl.config["URL_DENYLIST"] == "cdn.example.com"
|
||||
assert "PLUGINS" not in crawl.config
|
||||
assert "WGET_TIMEOUT" not in crawl.config
|
||||
assert "NODE_BINARY" not in crawl.config
|
||||
assert "TWOCAPTCHA_API_KEY" not in crawl.config
|
||||
assert "INDEX_ONLY" not in crawl.config
|
||||
assert crawl.config.get("PLUGINS", "") == ""
|
||||
assert crawl.config.get("WGET_TIMEOUT") != 77
|
||||
assert crawl.config.get("NODE_BINARY") != "/tmp/node"
|
||||
assert crawl.config.get("TWOCAPTCHA_API_KEY") != "posted-token"
|
||||
assert crawl.config.get("INDEX_ONLY") is not True
|
||||
assert crawl.status == Crawl.StatusChoices.QUEUED
|
||||
assert crawl.schedule is None
|
||||
|
||||
@ -417,7 +417,7 @@ def test_add_view_start_paused_creates_paused_crawl_without_snapshots(client, ad
|
||||
assert crawl.status == Crawl.StatusChoices.PAUSED
|
||||
assert crawl.retry_at == RETRY_AT_MAX
|
||||
assert crawl.snapshot_set.count() == 0
|
||||
assert "INDEX_ONLY" not in crawl.config
|
||||
assert crawl.config.get("INDEX_ONLY") is not True
|
||||
|
||||
|
||||
def test_add_view_extracts_urls_from_mixed_text_input(client, admin_user, monkeypatch):
|
||||
|
||||
@ -14,5 +14,6 @@ DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && cd .. && pwd )"
|
||||
|
||||
source "$DIR/.venv/bin/activate"
|
||||
|
||||
pytest -s --basetemp=archivebox/tests/data "$@"
|
||||
mkdir -p "$DIR/tests/out"
|
||||
pytest -s --basetemp="$DIR/tests/out" "$@"
|
||||
exec ./bin/test_plugins.sh
|
||||
|
||||
Loading…
Reference in New Issue
Block a user