ArchiveBox/archivebox/tests/test_title.py
2026-05-28 05:50:28 -07:00

78 lines
2.5 KiB
Python

import subprocess
import sys
import pytest
from archivebox.core.models import Snapshot
from archivebox.tests.test_orm_helpers import use_archivebox_db
from .conftest import _find_system_browser
from .fixtures import disable_extractors_dict, process
pytestmark = pytest.mark.django_db(transaction=True)
FIXTURES = (disable_extractors_dict, process)
def _install_chrome(tmp_path, env):
env["CHROME_ISOLATION"] = "snapshot"
system_browser = _find_system_browser()
if system_browser:
env["CHROME_BINARY"] = str(system_browser)
return
install_process = subprocess.run(
[sys.executable, "-m", "archivebox", "install", "chrome"],
cwd=tmp_path,
capture_output=True,
text=True,
env=env,
timeout=600,
)
assert install_process.returncode == 0, install_process.stderr or install_process.stdout
def test_title_is_extracted(tmp_path, process, disable_extractors_dict):
"""Test that title is extracted from the page."""
disable_extractors_dict.update({"SAVE_TITLE": "true"})
_install_chrome(tmp_path, disable_extractors_dict)
add_process = subprocess.run(
["archivebox", "add", "--plugins=chrome,wget,title", "https://example.com"],
capture_output=True,
text=True,
env=disable_extractors_dict,
)
assert add_process.returncode == 0, add_process.stderr or add_process.stdout
with use_archivebox_db(tmp_path):
title = Snapshot.objects.values_list("title", flat=True).get()
assert title is not None
assert "Example" in title
def test_title_is_htmlencoded_in_index_html(tmp_path, process, disable_extractors_dict):
"""
https://github.com/ArchiveBox/ArchiveBox/issues/330
Unencoded content should not be rendered as it facilitates xss injections
and breaks the layout.
"""
disable_extractors_dict.update({"SAVE_TITLE": "true"})
_install_chrome(tmp_path, disable_extractors_dict)
add_process = subprocess.run(
["archivebox", "add", "--plugins=chrome,wget,title", "https://example.com"],
capture_output=True,
text=True,
env=disable_extractors_dict,
)
assert add_process.returncode == 0, add_process.stderr or add_process.stdout
list_process = subprocess.run(
["archivebox", "search", "--html"],
capture_output=True,
text=True,
)
assert list_process.returncode == 0, list_process.stderr or list_process.stdout
# Should not contain unescaped HTML tags in output
output = list_process.stdout
assert "https://example.com" in output