From 032c20bdf54d58a41bd9abe2a4039d436e6d98c9 Mon Sep 17 00:00:00 2001 From: Nick Sweeting Date: Tue, 1 Sep 2026 15:06:46 -0700 Subject: [PATCH] cleanup: delete obsolete model lifecycle helpers --- archivebox/core/models.py | 62 ------------------------------------- archivebox/crawls/models.py | 37 ---------------------- 2 files changed, 99 deletions(-) diff --git a/archivebox/core/models.py b/archivebox/core/models.py index 44d1f4db..d99f2843 100644 --- a/archivebox/core/models.py +++ b/archivebox/core/models.py @@ -972,7 +972,6 @@ class Snapshot(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelW @property def binary_set(self): """Get all Binary objects used by processes related to this snapshot.""" - from archivebox.machine.models import Binary return Binary.objects.filter(process_set__archiveresult__snapshot_id=self.id).distinct() @@ -4276,12 +4275,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes): self.refresh_from_db() return True - @property - def plugin_module(self) -> Any | None: - # Hook scripts are now used instead of Python plugin modules - # The plugin name maps to hooks in abx_plugins/plugins/{plugin}/ - return None - @staticmethod def _normalize_output_files(raw_output_files: Any) -> dict[str, dict[str, Any]]: from abx_dl.output_files import OutputManifest @@ -4301,14 +4294,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes): def output_file_paths(self) -> list[str]: return list(self.output_file_map().keys()) - def output_file_count(self) -> int: - return len(self.output_file_paths()) - - def output_size_from_files(self) -> int: - from abx_dl.output_files import OutputManifest - - return OutputManifest.from_value(self.output_files).total_size - def update_output_metadata_from_filesystem(self, snapshot_dir: Path | None = None, save: bool = True) -> bool: from abx_dl.output_files import OutputManifest, output_file_from_path @@ -4369,9 +4354,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes): self.save(update_fields=["output_files", "output_size", "output_mimetypes", "modified_at"]) return True - def output_exists(self) -> bool: - return os.path.exists(Path(self.snapshot_dir) / self.plugin) - @staticmethod def _looks_like_output_path(raw_output: str | None, plugin_name: str | None = None) -> bool: value = str(raw_output or "").strip() @@ -4633,50 +4615,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes): process = self.process_record return process.timeout if process else 120 - def save_search_index(self): - pass - - def _set_binary_from_cmd(self, cmd: list) -> None: - """ - Find Binary for command and set binary FK. - - Tries matching by absolute path first, then by binary name. - Only matches binaries on the current machine. - """ - if not cmd: - return - - from archivebox.machine.models import Machine - - bin_path_or_name = cmd[0] if isinstance(cmd, list) else cmd - machine = Machine.current() - - # Try matching by absolute path first - binary = Binary.objects.filter( - abspath=bin_path_or_name, - machine=machine, - ).first() - - if binary: - process = self.process_record - if process: - process.binary = binary - process.save() - return - - # Fallback: match by binary name - bin_name = Path(bin_path_or_name).name - binary = Binary.objects.filter( - name=bin_name, - machine=machine, - ).first() - - if binary: - process = self.process_record - if process: - process.binary = binary - process.save() - def _url_passes_filters(self, url: str) -> bool: """Check if URL passes URL_ALLOWLIST and URL_DENYLIST config filters. diff --git a/archivebox/crawls/models.py b/archivebox/crawls/models.py index 490fcfed..f858ec3e 100755 --- a/archivebox/crawls/models.py +++ b/archivebox/crawls/models.py @@ -1293,43 +1293,6 @@ class Crawl(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelWith return created_snapshots - def install_declared_binaries(self, binary_names: set[str], machine=None) -> None: - """Install crawl-declared binaries through their unified lifecycle.""" - from archivebox.crawls.locks import binary_lifecycle_lock - from archivebox.machine.models import Binary, Machine - - if not binary_names: - return - - machine = machine or Machine.current() - binaries = Binary.objects.filter(machine=machine, name__in=binary_names).order_by("name") - for binary in binaries: - with binary_lifecycle_lock(str(binary.id)): - binary.refresh_from_db() - if binary.status == Binary.StatusChoices.INSTALLED: - continue - binary.update_and_requeue(retry_at=timezone.now()) - binary.refresh_from_db() - binary.install_claimed(lock_seconds=600) - - unresolved_binaries = list( - Binary.objects.filter( - machine=machine, - name__in=binary_names, - ) - .exclude( - status=Binary.StatusChoices.INSTALLED, - ) - .order_by("name"), - ) - if unresolved_binaries: - binary_details = ", ".join( - f"{binary.name} (status={binary.status}, retry_at={binary.retry_at})" for binary in unresolved_binaries - ) - raise RuntimeError( - f"Crawl dependencies failed to install before continuing: {binary_details}", - ) - def is_finished(self) -> bool: """Check if crawl is finished (all snapshots sealed or no snapshots exist).""" from archivebox.core.models import Snapshot