diff --git a/Dockerfile b/Dockerfile index c72854fc..34935351 100644 --- a/Dockerfile +++ b/Dockerfile @@ -308,10 +308,11 @@ RUN --mount=type=cache,target=/var/cache/apt,sharing=locked,id=apt-$TARGETARCH$T --mount=type=cache,target=/root/.cache/puppeteer,sharing=locked,id=puppeteer-$TARGETARCH$TARGETVARIANT \ --mount=type=cache,target=/root/.cache/ms-playwright,sharing=locked,id=browsers-$TARGETARCH$TARGETVARIANT \ echo "[+] Installing plugin runtime dependencies into $LIB_DIR..." \ + && apt-get update -qq \ && PUID=0 PGID=0 abx-dl plugins --install \ && find "$LIB_DIR" "$DATA_DIR"/personas -type d -name __pycache__ -prune -exec rm -rf {} + \ && find "$LIB_DIR" "$DATA_DIR"/personas -type f \( -name '*.pyc' -o -name '*.pyo' \) -delete \ - && rm -rf /root/.cache /var/cache/apt/* /var/lib/apt/lists/* \ + && rm -rf /var/lib/apt/lists/* \ && (chown -R "$DEFAULT_PUID:$DEFAULT_PGID" "$DATA_DIR"/personas 2>/dev/null || true) \ && chown -R "$DEFAULT_PUID:$DEFAULT_PGID" "$LIB_DIR" @@ -348,6 +349,7 @@ RUN (echo -e "\n\n[√] Finished Docker build successfully. Saving build summary # Verify ArchiveBox is installed and write full version/dependency info. RUN chmod +x "$CODE_DIR"/bin/*.sh \ + && chmod g+w "$TMP_DIR" "$LIB_DIR" "$LIB_DIR"/bin "$PLAYWRIGHT_BROWSERS_PATH" \ && gosu "$ARCHIVEBOX_USER" archivebox version 2>&1 | tee -a /VERSION.txt \ && find /venv "$CODE_DIR" "$LIB_DIR" "$DATA_DIR" -type d -name __pycache__ -prune -exec rm -rf {} + \ && find /venv "$CODE_DIR" "$LIB_DIR" "$DATA_DIR" -type f \( -name '*.pyc' -o -name '*.pyo' \) -delete \ diff --git a/archivebox/api/v1_core.py b/archivebox/api/v1_core.py index db33b5ca..64cd4656 100644 --- a/archivebox/api/v1_core.py +++ b/archivebox/api/v1_core.py @@ -4,7 +4,7 @@ import math from collections import defaultdict from uuid import UUID from typing import Union, Any, Annotated -from datetime import datetime +from datetime import datetime, time from django.db.models import Model, Q, Sum from django.db.models.functions import Coalesce @@ -14,6 +14,8 @@ from django.contrib.auth import get_user_model from django.contrib.auth.models import User from django.shortcuts import redirect from django.utils import timezone +from django.utils.dateparse import parse_date, parse_datetime +from django.utils.feedgenerator import Rss201rev2Feed from ninja import Router, Schema, FilterLookup, FilterSchema, Query from ninja.pagination import paginate, PaginationBase @@ -22,6 +24,7 @@ from ninja.errors import HttpError from archivebox.core.models import Snapshot, ArchiveResult, Tag from archivebox.api.auth import auth_using_token from archivebox.config.common import get_config +from archivebox.core.host_utils import build_web_url from archivebox.core.tag_utils import ( build_tag_cards, delete_tag as delete_tag_record, @@ -285,6 +288,119 @@ def normalize_tag_list(tags: list[str] | None = None) -> list[str]: return [tag.strip() for tag in (tags or []) if tag and tag.strip()] +def _parse_rss_before(before: str | None) -> datetime: + if not before: + return timezone.now() + + value = before.strip() + parsed_dt = None + + if len(value) == 8 and value.isdigit(): + parsed_date = datetime.strptime(value, "%Y%m%d").date() + else: + parsed_dt = parse_datetime(value) + parsed_date = None if parsed_dt else parse_date(value) + + if parsed_dt is None: + if parsed_date is None: + raise HttpError(400, "before must be an ISO datetime, YYYY-MM-DD, or YYYYMMDD") + parsed_dt = datetime.combine(parsed_date, time.max) + + if timezone.is_naive(parsed_dt): + parsed_dt = timezone.make_aware(parsed_dt, timezone.get_current_timezone()) + return parsed_dt + + +def _filter_snapshots_for_rss( + *, + crawl_id: str = "", + created_by: str = "", + before: str | None = None, + limit: int = 50, +): + limit = max(1, min(int(limit or 50), 500)) + before_dt = _parse_rss_before(before) + queryset = ( + Snapshot.objects.select_related("crawl__created_by") + .prefetch_related("tags") + .only( + "id", + "url", + "title", + "timestamp", + "bookmarked_at", + "created_at", + "modified_at", + "fs_version", + "crawl_id", + "crawl__id", + "crawl__created_by_id", + "crawl__created_by__id", + "crawl__created_by__username", + ) + .filter(bookmarked_at__lte=before_dt) + ) + + crawl_id = crawl_id.strip() + if crawl_id: + queryset = queryset.filter(crawl__id__icontains=crawl_id) + + created_by = created_by.strip() + if created_by: + created_by_query = Q(crawl__created_by__username__iexact=created_by) + user_model = get_user_model() + try: + prepared_pk = user_model._meta.pk.get_prep_value(created_by) + except (TypeError, ValueError, ValidationError): + prepared_pk = None + if prepared_pk not in (None, ""): + created_by_query |= Q(crawl__created_by_id=prepared_pk) + queryset = queryset.filter(created_by_query) + + return queryset.order_by("-bookmarked_at", "-created_at", "-id")[:limit] + + +def _snapshots_rss_response( + request: HttpRequest, + *, + snapshots, + title: str = "ArchiveBox Snapshots", +) -> HttpResponse: + web_base_url = build_web_url("/", request=request).rstrip("/") + feed_query = request.GET.copy() + for sensitive_param in ("api_key", "token", "password"): + feed_query.pop(sensitive_param, None) + feed_path = request.path + feed_url = request.build_absolute_uri(f"{feed_path}?{feed_query.urlencode()}" if feed_query else feed_path) + + feed = Rss201rev2Feed( + title=title, + link=build_web_url("/public/", request=request), + description="Recently added ArchiveBox snapshots.", + language="en", + feed_url=feed_url, + ) + + for snapshot in snapshots: + archived_url = build_web_url(f"/{snapshot.archive_path_from_db}", request=request) + tags = [tag.name for tag in snapshot.tags.all()] + crawl_user = snapshot.crawl.created_by if snapshot.crawl_id else None + description = f"Original URL: {snapshot.url}\nArchived snapshot: {archived_url}" + feed.add_item( + title=snapshot.title or snapshot.url, + link=archived_url or web_base_url, + description=description, + unique_id=str(snapshot.id), + unique_id_is_permalink=False, + pubdate=snapshot.bookmarked_at or snapshot.created_at, + updateddate=snapshot.modified_at, + author_name=crawl_user.username if crawl_user else None, + categories=tags, + ) + + return HttpResponse(feed.writeString("utf-8"), content_type="application/rss+xml; charset=utf-8") + + class SnapshotFilterSchema(FilterSchema): id: Annotated[str | None, FilterLookup(["id__icontains", "timestamp__startswith"])] = None created_by_id: Annotated[str | None, FilterLookup("crawl__created_by_id")] = None @@ -316,6 +432,25 @@ def get_snapshots(request: HttpRequest, filters: Query[SnapshotFilterSchema], wi return filters.filter(queryset).distinct() +@router.get("/snapshots.rss", url_name="get_snapshots_rss") +@router.get("/snapshot.rss", url_name="get_snapshot_rss") +def get_snapshots_rss( + request: HttpRequest, + crawl_id: str = "", + created_by: str = "", + limit: int = 50, + before: str | None = None, +): + """Return matching snapshots as an RSS feed, newest first.""" + snapshots = _filter_snapshots_for_rss( + crawl_id=crawl_id, + created_by=created_by, + limit=limit, + before=before, + ) + return _snapshots_rss_response(request, snapshots=snapshots) + + @router.get("/snapshot/{snapshot_id}", response=SnapshotSchema, url_name="get_snapshot") def get_snapshot(request: HttpRequest, snapshot_id: str, with_archiveresults: bool = True): """Get a specific Snapshot by id.""" diff --git a/archivebox/api/v1_crawls.py b/archivebox/api/v1_crawls.py index bd77c480..8702de05 100644 --- a/archivebox/api/v1_crawls.py +++ b/archivebox/api/v1_crawls.py @@ -3,6 +3,7 @@ __package__ = "archivebox.api" from uuid import UUID from datetime import datetime from django.http import HttpRequest +from django.shortcuts import redirect from django.utils import timezone from django.contrib.auth import get_user_model @@ -126,7 +127,7 @@ def create_crawl(request: HttpRequest, data: CrawlCreateSchema): return crawl -@router.get("/crawl/{crawl_id}", response=CrawlSchema | str, url_name="get_crawl") +@router.get("/crawl/{crawl_id}", response=CrawlSchema, url_name="get_crawl") def get_crawl(request: HttpRequest, crawl_id: str, as_rss: bool = False, with_snapshots: bool = False, with_archiveresults: bool = False): """Get a specific Crawl by id.""" setattr(request, "with_snapshots", with_snapshots) @@ -134,16 +135,10 @@ def get_crawl(request: HttpRequest, crawl_id: str, as_rss: bool = False, with_sn crawl = Crawl.objects.get(id__icontains=crawl_id) if crawl and as_rss: - # return snapshots as XML rss feed - urls = [ - {"url": snapshot.url, "title": snapshot.title, "bookmarked_at": snapshot.bookmarked_at, "tags": snapshot.tags_str} - for snapshot in crawl.snapshot_set.all() - ] - xml = '' - for url in urls: - xml += f"{url['url']}{url['title']}{url['bookmarked_at']}{url['tags']}" - xml += "" - return xml + query = request.GET.copy() + query.pop("as_rss", None) + query["crawl_id"] = str(crawl.id) + return redirect(f"/api/v1/core/snapshots.rss?{query.urlencode()}") return crawl diff --git a/archivebox/config/common.py b/archivebox/config/common.py index 300c6bcb..ad07c1d5 100644 --- a/archivebox/config/common.py +++ b/archivebox/config/common.py @@ -87,8 +87,8 @@ class StorageConfig(BaseConfigSet): # should not be a remote/network/FUSE mount for speed reasons, otherwise extractors will be slow LIB_DIR: Path = Field(default=CONSTANTS.DEFAULT_LIB_DIR) - # LIB_BIN_DIR is where all installed binaries are symlinked for easy PATH management - # Derived from LIB_DIR / 'bin', should be prepended to PATH for all hook executions + # LIB_BIN_DIR is where installed binaries can be symlinked for shared runtime lookup. + # abxpkg/abx-dl build the executable lookup env at exec time. LIB_BIN_DIR: Path = Field(default=CONSTANTS.DEFAULT_LIB_BIN_DIR) # CUSTOM_TEMPLATES_DIR allows users to override default templates diff --git a/archivebox/core/admin_site.py b/archivebox/core/admin_site.py index ff5bfe97..f3a3b177 100644 --- a/archivebox/core/admin_site.py +++ b/archivebox/core/admin_site.py @@ -24,6 +24,37 @@ class ArchiveBoxAdmin(admin.AdminSite): site_title = "Admin" namespace = "admin" + @staticmethod + def _format_object_count(count: int) -> tuple[int, str, str]: + if count >= 1_000_000_000: + count_label = f"{count / 1_000_000_000:.1f}B" + elif count >= 1_000_000: + count_label = f"{count / 1_000_000:.1f}M" + elif count >= 1_000: + count_label = f"{count / 1_000:.1f}K" + else: + count_label = f"{count:,}" + count_label = count_label.replace(".0", "") + return count, count_label, f"Object count: {count:,}" + + def _set_model_object_count( + self, + models_by_table: dict[str, list[dict[str, Any]]], + table: str, + count: int, + title: str | None = None, + ) -> None: + models = models_by_table.get(table) + if not models: + return + count, count_label, count_title = self._format_object_count(count) + if title: + count_title = title + for model in models: + model["object_count"] = count + model["object_count_label"] = count_label + model["object_count_title"] = count_title + def get_app_list(self, request: "HttpRequest", app_label: str | None = None) -> list["AppDict"]: if app_label is None: return adv_get_app_list(self, request) @@ -52,29 +83,29 @@ class ArchiveBoxAdmin(admin.AdminSite): with connection.cursor() as cursor: cursor.execute("SELECT tbl, stat FROM sqlite_stat1") for table, stat in cursor.fetchall(): - models = models_by_table.get(table) - if not models: - continue try: count = int(str(stat).split()[0]) except (IndexError, TypeError, ValueError): continue - if count >= 1_000_000_000: - count_label = f"{count / 1_000_000_000:.1f}B" - elif count >= 1_000_000: - count_label = f"{count / 1_000_000:.1f}M" - elif count >= 1_000: - count_label = f"{count / 1_000:.1f}K" - else: - count_label = f"{count:,}" - count_label = count_label.replace(".0", "") - for model in models: - model["object_count"] = count - model["object_count_label"] = count_label - model["object_count_title"] = f"Approximate count from SQLite stats: {count:,}" + self._set_model_object_count( + models_by_table, + table, + count, + title=f"Approximate count from SQLite stats: {count:,}", + ) models_by_table.pop(table, None) except DatabaseError: pass + + for table in list(models_by_table): + try: + with connection.cursor() as cursor: + cursor.execute(f"SELECT COUNT(*) FROM {connection.ops.quote_name(table)}") + count = int(cursor.fetchone()[0]) + except DatabaseError: + continue + self._set_model_object_count(models_by_table, table, count) + models_by_table.pop(table, None) return response def get_admin_data_urls(self) -> list["URLResolver | URLPattern"]: diff --git a/archivebox/core/admin_tags.py b/archivebox/core/admin_tags.py index e2ff20da..843e75ee 100644 --- a/archivebox/core/admin_tags.py +++ b/archivebox/core/admin_tags.py @@ -6,6 +6,7 @@ from django import forms from django.contrib import admin, messages from django.contrib.admin.options import IS_POPUP_VAR from django.http import HttpRequest, HttpResponseRedirect +from django.template.response import TemplateResponse from django.urls import reverse from django.utils.html import format_html from django.utils.safestring import mark_safe @@ -117,8 +118,11 @@ class TagAdmin(BaseModelAdmin): created_by = normalize_created_by_filter((request.GET.get("created_by") or "").strip()) year = normalize_created_year_filter((request.GET.get("year") or "").strip()) has_snapshots = normalize_has_snapshots_filter((request.GET.get("has_snapshots") or "all").strip()) - extra_context = { + context = { + **self.admin_site.each_context(request), **(extra_context or {}), + "title": "Tags", + "opts": self.model._meta, "initial_query": query, "initial_sort": sort, "initial_created_by": created_by, @@ -131,6 +135,7 @@ class TagAdmin(BaseModelAdmin): "initial_tag_cards": build_tag_cards( query=query, request=request, + preview_limit=0, sort=sort, created_by=created_by, year=year, @@ -139,7 +144,7 @@ class TagAdmin(BaseModelAdmin): "tag_search_api_url": reverse("api-1:search_tags"), "tag_create_api_url": reverse("api-1:tags_create"), } - return super().changelist_view(request, extra_context=extra_context) + return TemplateResponse(request, self.change_list_template, context) def render_change_form(self, request, context, add=False, change=False, form_url="", obj=None): current_name = (request.POST.get("name") or "").strip() @@ -147,7 +152,9 @@ class TagAdmin(BaseModelAdmin): current_name = obj.name similar_tag_cards = ( - build_tag_cards(query=current_name, request=request, limit=12) if current_name else build_tag_cards(request=request, limit=12) + build_tag_cards(query=current_name, request=request, limit=12, preview_limit=0) + if current_name + else build_tag_cards(request=request, limit=12, preview_limit=0) ) if obj: similar_tag_cards = [card for card in similar_tag_cards if card["id"] != obj.pk] diff --git a/archivebox/core/admin_users.py b/archivebox/core/admin_users.py index 7a38271b..47bbe905 100644 --- a/archivebox/core/admin_users.py +++ b/archivebox/core/admin_users.py @@ -1,5 +1,7 @@ __package__ = "archivebox.core" +from urllib.parse import urlencode + from django.contrib import admin from django.contrib.auth.admin import UserAdmin from django.contrib.auth import get_user_model @@ -11,6 +13,7 @@ class CustomUserAdmin(UserAdmin): sort_fields = ["id", "email", "username", "is_superuser", "last_login", "date_joined"] list_display = ["username", "id", "email", "is_superuser", "last_login", "date_joined"] readonly_fields = ("snapshot_set", "archiveresult_set", "tag_set", "apitoken_set", "outboundwebhook_set") + change_form_template = "admin/auth/user/change_form.html" # Preserve Django's default user creation form and fieldsets # This ensures passwords are properly hashed and permissions are set correctly @@ -19,6 +22,37 @@ class CustomUserAdmin(UserAdmin): # Extend fieldsets for change form only (not user creation) fieldsets = [*(UserAdmin.fieldsets or ()), ("Data", {"fields": readonly_fields})] + def snapshot_rss_badge(self, obj, api_token: str = ""): + params = {"created_by": obj.username, "limit": 50} + if api_token: + params["api_key"] = api_token + url = f"/api/v1/core/snapshots.rss?{urlencode(params)}" + return format_html( + ( + '' + '' + "RSS" + ), + url, + obj.username, + ) + + def get_list_display(self, request): + from archivebox.api.auth import get_or_create_api_token + + api_token = get_or_create_api_token(request.user) + token = api_token.token if api_token else "" + + @admin.display(description="Feed") + def snapshot_rss_feed(obj): + return self.snapshot_rss_badge(obj, api_token=token) + + return ["username", snapshot_rss_feed, "id", "email", "is_superuser", "last_login", "date_joined"] + @admin.display(description="Snapshots") def snapshot_set(self, obj): total_count = obj.snapshot_set.count() diff --git a/archivebox/core/tag_utils.py b/archivebox/core/tag_utils.py index 2bc89f4b..8a3d5c60 100644 --- a/archivebox/core/tag_utils.py +++ b/archivebox/core/tag_utils.py @@ -11,6 +11,7 @@ from django.db.models.functions import Lower from django.http import HttpRequest from django.urls import reverse +from archivebox.config.common import get_config from archivebox.core.host_utils import build_snapshot_url, build_web_url from archivebox.core.models import Snapshot, SnapshotTag, Tag @@ -184,14 +185,14 @@ def _display_snapshot_title(snapshot: Snapshot) -> str: return title -def _build_snapshot_preview(snapshot: Snapshot, request: HttpRequest | None = None) -> dict[str, Any]: +def _build_snapshot_preview(snapshot: Snapshot, request: HttpRequest | None = None, config: Any | None = None) -> dict[str, Any]: return { "id": str(snapshot.pk), "title": _display_snapshot_title(snapshot), "url": snapshot.url, - "favicon_url": build_snapshot_url(str(snapshot.pk), "favicon.ico", request=request), + "favicon_url": build_snapshot_url(str(snapshot.pk), "favicon.ico", request=request, config=config), "admin_url": reverse("admin:core_snapshot_change", args=[snapshot.pk]), - "archive_url": build_web_url(f"/{snapshot.archive_path_from_db}/index.html", request=request), + "archive_url": build_web_url(f"/{snapshot.archive_path_from_db}/index.html", request=request, config=config), "downloaded_at": snapshot.downloaded_at.isoformat() if snapshot.downloaded_at else None, } @@ -202,7 +203,7 @@ def _build_snapshot_preview_map( preview_limit: int = TAG_SNAPSHOT_PREVIEW_LIMIT, ) -> dict[int, list[dict[str, Any]]]: tag_ids = [tag.pk for tag in tags] - if not tag_ids: + if not tag_ids or preview_limit <= 0: return {} snapshot_tags = ( @@ -217,16 +218,19 @@ def _build_snapshot_preview_map( ) preview_map: dict[int, list[dict[str, Any]]] = defaultdict(list) + config = get_config() for snapshot_tag in snapshot_tags: previews = preview_map[snapshot_tag.tag_id] if len(previews) >= preview_limit: continue - previews.append(_build_snapshot_preview(snapshot_tag.snapshot, request=request)) + previews.append(_build_snapshot_preview(snapshot_tag.snapshot, request=request, config=config)) return preview_map def build_tag_card(tag: Tag, snapshot_previews: list[dict[str, Any]] | None = None) -> dict[str, Any]: - count = getattr(tag, "num_snapshots", tag.snapshot_set.count()) + count = getattr(tag, "num_snapshots", None) + if count is None: + count = tag.snapshot_set.count() return { "id": tag.pk, "name": tag.name, diff --git a/archivebox/hooks.py b/archivebox/hooks.py index bf9c1eb7..bc6f22ab 100644 --- a/archivebox/hooks.py +++ b/archivebox/hooks.py @@ -437,14 +437,13 @@ def run_hook( except Exception: pass - # Get LIB_DIR and LIB_BIN_DIR from config + # Export runtime library roots; abx-dl/abxpkg own executable lookup env. lib_dir = resolved_config.LIB_DIR lib_bin_dir = resolved_config.LIB_BIN_DIR if lib_dir: env["LIB_DIR"] = str(lib_dir) - if not lib_bin_dir and lib_dir: - # Derive LIB_BIN_DIR from LIB_DIR if not set - lib_bin_dir = Path(lib_dir) / "bin" + if lib_bin_dir: + env["LIB_BIN_DIR"] = str(lib_bin_dir) # Set Node.js module resolution paths. # NODE_PATH may be a path list, but NODE_MODULES_DIR is a single canonical directory. @@ -491,41 +490,6 @@ def run_hook( else: env[key] = str(value) - # Build PATH with proper precedence: - # 1. path-like *_BINARY parents (explicit binary overrides / cached abspaths) - # 2. LIB_BIN_DIR (local symlinked binaries) - # 3. existing PATH - runtime_bin_dirs: list[str] = [] - if lib_bin_dir: - lib_bin_dir = str(lib_bin_dir) - env["LIB_BIN_DIR"] = lib_bin_dir - for key, raw_value in env.items(): - if not key.endswith("_BINARY"): - continue - value = str(raw_value or "").strip() - if not value: - continue - path_value = Path(value).expanduser() - if not (path_value.is_absolute() or "/" in value or "\\" in value): - continue - binary_dir = str(path_value.resolve(strict=False).parent) - if binary_dir and binary_dir not in runtime_bin_dirs: - runtime_bin_dirs.append(binary_dir) - if lib_bin_dir and lib_bin_dir not in runtime_bin_dirs: - runtime_bin_dirs.append(lib_bin_dir) - uv_value = str(env.get("UV") or "").strip() - if uv_value: - uv_bin_dir = str(Path(uv_value).expanduser().resolve(strict=False).parent) - if uv_bin_dir and uv_bin_dir not in runtime_bin_dirs: - runtime_bin_dirs.append(uv_bin_dir) - - current_path = env.get("PATH", "") - path_parts = [part for part in current_path.split(os.pathsep) if part] - for extra_dir in reversed(runtime_bin_dirs): - if extra_dir not in path_parts: - path_parts.insert(0, extra_dir) - env["PATH"] = os.pathsep.join(path_parts) - # Create output directory if needed output_dir.mkdir(parents=True, exist_ok=True) diff --git a/archivebox/machine/migrations/0013_alter_machine_config.py b/archivebox/machine/migrations/0013_alter_machine_config.py new file mode 100644 index 00000000..a3f98e57 --- /dev/null +++ b/archivebox/machine/migrations/0013_alter_machine_config.py @@ -0,0 +1,17 @@ +# Generated by Django 6.0.5 on 2026-05-24 09:59 + +from django.db import migrations, models + + +class Migration(migrations.Migration): + dependencies = [ + ("machine", "0012_add_machine_config_if_missing"), + ] + + operations = [ + migrations.AlterField( + model_name="machine", + name="config", + field=models.JSONField(blank=True, default=dict, help_text="Machine-specific config overrides.", null=True), + ), + ] diff --git a/archivebox/machine/models.py b/archivebox/machine/models.py index aa387f94..b520a45d 100755 --- a/archivebox/machine/models.py +++ b/archivebox/machine/models.py @@ -734,12 +734,12 @@ class Binary(ModelWithHealthStats, ModelWithStateMachine): def symlink_to_lib_bin(self, lib_bin_dir: str | Path) -> Path | None: """ - Symlink this binary into LIB_BIN_DIR for unified PATH management. + Symlink this binary into LIB_BIN_DIR for shared runtime lookup. After a binary is installed by any binprovider (pip, npm, brew, apt, etc), we symlink it into LIB_BIN_DIR so that: 1. All binaries can be found in a single directory - 2. PATH only needs LIB_BIN_DIR prepended (not multiple provider-specific paths) + 2. abxpkg/abx-dl can include the shared bin dir when constructing exec envs 3. Binary priorities are clear (symlink points to the canonical install location) Args: diff --git a/archivebox/templates/admin/auth/user/change_form.html b/archivebox/templates/admin/auth/user/change_form.html new file mode 100644 index 00000000..b825a830 --- /dev/null +++ b/archivebox/templates/admin/auth/user/change_form.html @@ -0,0 +1,47 @@ +{% extends "admin/change_form.html" %} +{% load core_tags %} + +{% block extrastyle %} +{{ block.super }} + +{% endblock %} + +{% block object-tools-items %} +{% if original %} + {% api_token as api_token %} +
  • + + RSS + Snapshot Feed + +
  • +{% endif %} +{{ block.super }} +{% endblock %} diff --git a/archivebox/templates/admin/core/tag/change_list.html b/archivebox/templates/admin/core/tag/change_list.html index 90366be8..ad43f8a9 100644 --- a/archivebox/templates/admin/core/tag/change_list.html +++ b/archivebox/templates/admin/core/tag/change_list.html @@ -1,4 +1,4 @@ -{% extends "admin/change_list.html" %} +{% extends "admin/base_site.html" %} {% block bodyclass %}{{ block.super }} app-core model-tag change-list tag-admin-page{% endblock %} @@ -471,54 +471,7 @@
    - {% if initial_tag_cards %} - {% for card in initial_tag_cards %} -
    -
    -
    - -
    - - - -
    -
    -
    - - - - - {{ card.num_snapshots }} -
    -
    -
    - {% if card.snapshots %} - {% for snapshot in card.snapshots %} - - - {{ snapshot.title }} - - {% endfor %} - {% else %} -
    No snapshots attached yet.
    - {% endif %} -
    -
    - {% endfor %} - {% else %} -
    No tags.
    - {% endif %} +
    Loading tags...
    diff --git a/archivebox/tests/test_admin_views.py b/archivebox/tests/test_admin_views.py index 6145e4ca..b1423065 100644 --- a/archivebox/tests/test_admin_views.py +++ b/archivebox/tests/test_admin_views.py @@ -1228,6 +1228,16 @@ class TestArchiveResultAdminListView: assert response.status_code == 200 assert b"Select user to change" in response.content + assert b"/api/v1/core/snapshots.rss?created_by=testadmin&limit=50&api_key=" in response.content + assert b"RSS" in response.content + + def test_user_admin_change_view_renders_rss_feed_link(self, client, admin_user): + client.login(username="testadmin", password="testpassword") + response = client.get(reverse("admin:auth_user_change", args=[admin_user.pk]), HTTP_HOST=ADMIN_HOST) + + assert response.status_code == 200 + assert b"Snapshot Feed" in response.content + assert b"/api/v1/core/snapshots.rss?created_by=testadmin&limit=50&api_key=" in response.content def test_archiveresult_model_has_no_retry_at_field(self): from archivebox.core.models import ArchiveResult diff --git a/archivebox/tests/test_api_rss.py b/archivebox/tests/test_api_rss.py new file mode 100644 index 00000000..e5d892c8 --- /dev/null +++ b/archivebox/tests/test_api_rss.py @@ -0,0 +1,166 @@ +from datetime import datetime +from typing import cast + +import pytest +from django.contrib.auth import get_user_model +from django.contrib.auth.models import UserManager +from django.utils import timezone + + +pytestmark = pytest.mark.django_db + + +User = get_user_model() +ADMIN_HOST = "admin.archivebox.localhost:8000" + + +@pytest.fixture +def admin_user(db): + return cast(UserManager, User.objects).create_superuser( + username="rssadmin", + email="rssadmin@test.com", + password="testpassword", + ) + + +@pytest.fixture +def other_user(db): + return cast(UserManager, User.objects).create_user( + username="rssother", + email="rssother@test.com", + password="testpassword", + ) + + +@pytest.fixture +def api_token(admin_user): + from archivebox.api.auth import get_or_create_api_token + + token = get_or_create_api_token(admin_user) + assert token is not None + return token.token + + +def make_snapshot(*, user, url: str, title: str, bookmarked_at: datetime): + from archivebox.core.models import Snapshot + from archivebox.crawls.models import Crawl + + crawl = Crawl.objects.create(urls=url, created_by=user) + snapshot = Snapshot.objects.create( + url=url, + title=title, + crawl=crawl, + bookmarked_at=bookmarked_at, + ) + return crawl, snapshot + + +def test_snapshots_rss_filters_by_user_and_orders_newest_first(client, api_token, admin_user, other_user): + from archivebox.core.models import Tag + + older_at = timezone.make_aware(datetime(2026, 5, 22, 8, 0, 0)) + newer_at = timezone.make_aware(datetime(2026, 5, 23, 8, 0, 0)) + _crawl, older_snapshot = make_snapshot( + user=admin_user, + url="https://example.com/rss-older", + title="Older & Escaped", + bookmarked_at=older_at, + ) + make_snapshot( + user=admin_user, + url="https://example.com/rss-newer", + title="Newer Snapshot", + bookmarked_at=newer_at, + ) + make_snapshot( + user=other_user, + url="https://example.com/rss-other-user", + title="Other User", + bookmarked_at=timezone.make_aware(datetime(2026, 5, 23, 9, 0, 0)), + ) + older_snapshot.tags.add(Tag.objects.create(name="rss-tag", created_by=admin_user)) + + response = client.get( + "/api/v1/core/snapshots.rss", + {"created_by": admin_user.username, "limit": 50, "api_key": api_token}, + HTTP_HOST=ADMIN_HOST, + ) + + assert response.status_code == 200 + assert response["Content-Type"].startswith("application/rss+xml") + body = response.content.decode() + assert 'rss-tag" in body + assert "rss-other-user" not in body + assert body.index("rss-newer") < body.index("rss-older") + + +def test_snapshots_rss_supports_before_yyyymmdd_and_limit(client, api_token, admin_user): + make_snapshot( + user=admin_user, + url="https://example.com/rss-before-too-new", + title="Too New", + bookmarked_at=timezone.make_aware(datetime(2026, 5, 24, 8, 0, 0)), + ) + make_snapshot( + user=admin_user, + url="https://example.com/rss-before-keep-one", + title="Keep One", + bookmarked_at=timezone.make_aware(datetime(2026, 5, 23, 12, 0, 0)), + ) + make_snapshot( + user=admin_user, + url="https://example.com/rss-before-keep-two", + title="Keep Two", + bookmarked_at=timezone.make_aware(datetime(2026, 5, 22, 12, 0, 0)), + ) + + response = client.get( + "/api/v1/core/snapshots.rss", + {"created_by": str(admin_user.pk), "before": "20260523", "limit": 1, "api_key": api_token}, + HTTP_HOST=ADMIN_HOST, + ) + + assert response.status_code == 200 + body = response.content.decode() + assert "rss-before-too-new" not in body + assert "rss-before-keep-one" in body + assert "rss-before-keep-two" not in body + + +def test_crawl_as_rss_redirects_to_canonical_snapshots_feed(client, api_token, admin_user, other_user): + crawl, _snapshot = make_snapshot( + user=admin_user, + url="https://example.com/rss-crawl-feed", + title="Crawl Feed Snapshot", + bookmarked_at=timezone.make_aware(datetime(2026, 5, 23, 8, 0, 0)), + ) + make_snapshot( + user=other_user, + url="https://example.com/rss-crawl-other", + title="Other Crawl Snapshot", + bookmarked_at=timezone.make_aware(datetime(2026, 5, 23, 9, 0, 0)), + ) + + response = client.get( + f"/api/v1/crawls/crawl/{crawl.id}", + {"as_rss": "true", "limit": 50, "api_key": api_token}, + HTTP_HOST=ADMIN_HOST, + follow=True, + ) + + assert response.status_code == 200 + assert response.redirect_chain + redirect_url = response.redirect_chain[0][0] + assert redirect_url.startswith("/api/v1/core/snapshots.rss?") + assert f"crawl_id={crawl.id}" in redirect_url + assert "as_rss" not in redirect_url + assert response["Content-Type"].startswith("application/rss+xml") + body = response.content.decode() + assert "rss-crawl-feed" in body + assert "rss-crawl-other" not in body diff --git a/docs b/docs index 7244076e..93828b52 160000 --- a/docs +++ b/docs @@ -1 +1 @@ -Subproject commit 7244076ecec0264dddfba14930f5f8bfe4fb4ef0 +Subproject commit 93828b52676861d56e98d6ba8d9ebf5385e76b3f diff --git a/etc/package.json b/etc/package.json index 9801a2a2..18a5cee8 100644 --- a/etc/package.json +++ b/etc/package.json @@ -1,6 +1,6 @@ { "name": "archivebox", - "version": "0.9.31rc30", + "version": "0.9.31rc31", "repository": "github:ArchiveBox/ArchiveBox", "license": "MIT", "dependencies": { diff --git a/pyproject.toml b/pyproject.toml index 25a143cc..bac6a8d9 100755 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "archivebox" -version = "0.9.31rc30" +version = "0.9.31rc31" requires-python = ">=3.13" description = "Self-hosted internet archiving solution." authors = [{name = "Nick Sweeting", email = "pyproject.toml@archivebox.io"}] @@ -79,9 +79,9 @@ dependencies = [ ### Extractor dependencies (optional binaries detected at runtime via shutil.which) ### Binary/Package Management "abxbus>=2.5.4", # EventBus API - "abxpkg>=1.10.20", # for: detecting, versioning, and installing binaries via apt/brew/pip/npm - "abx-plugins>=1.10.85", # shared ArchiveBox plugin package with Chrome/Puppeteer dependency wiring - "abx-dl>=1.10.85", # shared ArchiveBox downloader package with blocking install preflight + "abxpkg>=1.10.21", # for: detecting, versioning, and installing binaries via apt/brew/pip/npm + "abx-plugins>=1.10.86", # shared ArchiveBox plugin package with Chrome/Puppeteer dependency wiring + "abx-dl>=1.10.86", # shared ArchiveBox downloader package with blocking install preflight ### UUID7 backport for Python <3.14 "uuid7>=0.1.0; python_version < '3.14'", # provides the uuid_extensions module on Python 3.13 ]