cleanup: delete obsolete model lifecycle helpers

This commit is contained in:
Nick Sweeting 2026-09-01 15:06:46 -07:00
parent d42b530caf
commit 032c20bdf5
No known key found for this signature in database
2 changed files with 0 additions and 99 deletions

View File

@ -972,7 +972,6 @@ class Snapshot(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelW
@property
def binary_set(self):
"""Get all Binary objects used by processes related to this snapshot."""
from archivebox.machine.models import Binary
return Binary.objects.filter(process_set__archiveresult__snapshot_id=self.id).distinct()
@ -4276,12 +4275,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes):
self.refresh_from_db()
return True
@property
def plugin_module(self) -> Any | None:
# Hook scripts are now used instead of Python plugin modules
# The plugin name maps to hooks in abx_plugins/plugins/{plugin}/
return None
@staticmethod
def _normalize_output_files(raw_output_files: Any) -> dict[str, dict[str, Any]]:
from abx_dl.output_files import OutputManifest
@ -4301,14 +4294,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes):
def output_file_paths(self) -> list[str]:
return list(self.output_file_map().keys())
def output_file_count(self) -> int:
return len(self.output_file_paths())
def output_size_from_files(self) -> int:
from abx_dl.output_files import OutputManifest
return OutputManifest.from_value(self.output_files).total_size
def update_output_metadata_from_filesystem(self, snapshot_dir: Path | None = None, save: bool = True) -> bool:
from abx_dl.output_files import OutputManifest, output_file_from_path
@ -4369,9 +4354,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes):
self.save(update_fields=["output_files", "output_size", "output_mimetypes", "modified_at"])
return True
def output_exists(self) -> bool:
return os.path.exists(Path(self.snapshot_dir) / self.plugin)
@staticmethod
def _looks_like_output_path(raw_output: str | None, plugin_name: str | None = None) -> bool:
value = str(raw_output or "").strip()
@ -4633,50 +4615,6 @@ class ArchiveResult(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithNotes):
process = self.process_record
return process.timeout if process else 120
def save_search_index(self):
pass
def _set_binary_from_cmd(self, cmd: list) -> None:
"""
Find Binary for command and set binary FK.
Tries matching by absolute path first, then by binary name.
Only matches binaries on the current machine.
"""
if not cmd:
return
from archivebox.machine.models import Machine
bin_path_or_name = cmd[0] if isinstance(cmd, list) else cmd
machine = Machine.current()
# Try matching by absolute path first
binary = Binary.objects.filter(
abspath=bin_path_or_name,
machine=machine,
).first()
if binary:
process = self.process_record
if process:
process.binary = binary
process.save()
return
# Fallback: match by binary name
bin_name = Path(bin_path_or_name).name
binary = Binary.objects.filter(
name=bin_name,
machine=machine,
).first()
if binary:
process = self.process_record
if process:
process.binary = binary
process.save()
def _url_passes_filters(self, url: str) -> bool:
"""Check if URL passes URL_ALLOWLIST and URL_DENYLIST config filters.

View File

@ -1293,43 +1293,6 @@ class Crawl(ModelWithDeleteAfter, ModelWithOutputDir, ModelWithConfig, ModelWith
return created_snapshots
def install_declared_binaries(self, binary_names: set[str], machine=None) -> None:
"""Install crawl-declared binaries through their unified lifecycle."""
from archivebox.crawls.locks import binary_lifecycle_lock
from archivebox.machine.models import Binary, Machine
if not binary_names:
return
machine = machine or Machine.current()
binaries = Binary.objects.filter(machine=machine, name__in=binary_names).order_by("name")
for binary in binaries:
with binary_lifecycle_lock(str(binary.id)):
binary.refresh_from_db()
if binary.status == Binary.StatusChoices.INSTALLED:
continue
binary.update_and_requeue(retry_at=timezone.now())
binary.refresh_from_db()
binary.install_claimed(lock_seconds=600)
unresolved_binaries = list(
Binary.objects.filter(
machine=machine,
name__in=binary_names,
)
.exclude(
status=Binary.StatusChoices.INSTALLED,
)
.order_by("name"),
)
if unresolved_binaries:
binary_details = ", ".join(
f"{binary.name} (status={binary.status}, retry_at={binary.retry_at})" for binary in unresolved_binaries
)
raise RuntimeError(
f"Crawl dependencies failed to install before continuing: {binary_details}",
)
def is_finished(self) -> bool:
"""Check if crawl is finished (all snapshots sealed or no snapshots exist)."""
from archivebox.core.models import Snapshot