ci: stabilize archivebox parallel shards

This commit is contained in:
Nick Sweeting 2026-06-02 13:22:03 -07:00
parent 1fb2d106c1
commit c7cd4de3da
No known key found for this signature in database
11 changed files with 53 additions and 20 deletions

View File

@ -1339,10 +1339,10 @@ def wait_for_archive_outputs(
if output_rel is None:
raise SystemExit(1)
responses_root = Path(snapshot.output_dir) / 'responses' / snapshot.domain
responses_root = Path(snapshot.output_dir) / 'responses'
if not responses_root.exists():
raise SystemExit(1)
if not any(candidate.is_file() for candidate in responses_root.rglob('*')):
if not any(candidate.is_file() and snapshot.domain in candidate.relative_to(responses_root).parts for candidate in responses_root.rglob('*')):
raise SystemExit(1)
print('READY')

View File

@ -391,7 +391,7 @@ def test_crawl_pause_resume_api_survives_server_restart_and_processes_after_resu
stop_server(tmp_path)
@pytest.mark.timeout(180)
@pytest.mark.timeout(420)
def test_update_index_only_runs_paused_search_rows_and_resume_later_runs_crawl(tmp_path, recursive_test_site):
init_archive(tmp_path)
@ -492,7 +492,7 @@ def test_update_index_only_runs_paused_search_rows_and_resume_later_runs_crawl(t
assert resume_response.status_code == 200, resume_response.text
assert resume_response.json()["status"] == "queued"
captured_text = wait_for_snapshot_capture(tmp_path, recursive_test_site["root_url"], timeout=180)
captured_text = wait_for_snapshot_capture(tmp_path, recursive_test_site["root_url"], timeout=240)
assert "Root" in captured_text
assert "About" in captured_text

View File

@ -201,9 +201,10 @@ def test_init_with_existing_data_preserves_snapshots(initialized_archive):
# Add a snapshot
run_archivebox_cmd(
["add", "--index-only", "--depth=0", "https://example.com"],
cwd=initialized_archive,
env=env,
)
run_queued_crawls(initialized_archive, env)
run_queued_crawls(initialized_archive, env, timeout=300)
# Check snapshot was created
with use_archivebox_db(initialized_archive):
@ -211,7 +212,7 @@ def test_init_with_existing_data_preserves_snapshots(initialized_archive):
assert count_before == 1
# Run init again
result = run_archivebox_cmd(["init"])
result = run_archivebox_cmd(["init"], cwd=initialized_archive)
assert result.returncode == 0
# Snapshot should still exist

View File

@ -8,6 +8,7 @@ import os
import asyncio
import json
import signal
import shutil
import socket
import subprocess
import sys
@ -167,8 +168,11 @@ def test_reload_workers_use_current_interpreter_and_supervisord_managed_runner()
def test_server_daemon_starts_real_plugin_owned_sonic_worker(archivebox_daemon_server):
if shutil.which("sonic") is None:
pytest.skip("sonic server binary is required for Sonic worker integration tests")
server = archivebox_daemon_server(
SEARCH_BACKEND_ENGINE="sqlite",
SEARCH_BACKEND_ENGINE="sonic",
)
state = server.wait_for_workers(("worker_daphne", "worker_sonic", "worker_runner"))
@ -289,6 +293,9 @@ def test_sonic_worker_is_disabled_when_sonic_disabled(tmp_path):
def test_sonic_daemon_event_handler_accepts_real_running_worker(archivebox_daemon_server):
if shutil.which("sonic") is None:
pytest.skip("sonic server binary is required for Sonic worker integration tests")
from abx_dl.events import ProcessStdoutEvent
from abx_dl.orchestrator import create_bus
from archivebox.search.sonic_daemon import register_sonic_daemon_event_handler

View File

@ -32,6 +32,7 @@ def test_title_is_extracted(tmp_path, initialized_archive):
_install_chrome(tmp_path, env)
add_process = run_archivebox_cmd(
["add", "--plugins=chrome,wget,title", "https://example.com"],
cwd=tmp_path,
env=env,
)
assert add_process.returncode == 0, add_process.stderr or add_process.stdout
@ -54,11 +55,13 @@ def test_title_is_listed_by_search_alias(tmp_path, initialized_archive):
_install_chrome(tmp_path, env)
add_process = run_archivebox_cmd(
["add", "--plugins=chrome,wget,title", "https://example.com"],
cwd=tmp_path,
env=env,
)
assert add_process.returncode == 0, add_process.stderr or add_process.stdout
list_process = run_archivebox_cmd(
["search"],
cwd=tmp_path,
env=env,
)
assert list_process.returncode == 0, list_process.stderr or list_process.stdout

View File

@ -31,12 +31,17 @@ def _write_tool_shim(bin_dir: Path, name: str, version: str) -> Path:
def _runtime_env(data_dir: Path, bin_dir: Path) -> dict[str, str]:
path_entries = [
str(bin_dir),
str(data_dir / "lib" / "env" / "bin"),
os.environ.get("PATH", ""),
]
return {
"LIB_DIR": str(data_dir / "lib"),
"LIB_BIN_DIR": str(data_dir / "lib" / "bin"),
"ABXPKG_LIB_DIR": str(data_dir / "lib"),
"LITEPARSE_ENABLED": "True",
"PATH": os.pathsep.join([str(bin_dir), str(data_dir / "lib" / "env" / "bin"), "/usr/bin", "/bin", "/usr/sbin", "/sbin"]),
"PATH": os.pathsep.join(entry for entry in path_entries if entry),
}

View File

@ -1,5 +1,6 @@
import os
import signal
import shutil
import socket
import subprocess
import time
@ -446,6 +447,9 @@ class TestPublicIndexSearch:
class TestSearchBackendsE2E:
def test_real_fulltext_search_backends_survive_reindex_transition(self, tmp_path):
if shutil.which("sonic") is None:
pytest.skip("sonic server binary is required for Sonic search integration tests")
data_dir = tmp_path / "archivebox_data"
data_dir.mkdir()
query = "documentation examples"

View File

@ -744,7 +744,7 @@ def test_archivewebpage_wacz_preview_serves_real_capture_frame(initialized_archi
url = "https://example.com"
port = get_free_port()
env = cli_env(
port,
port=port,
PLUGINS="archivewebpage",
URL_ALLOWLIST="",
PUBLIC_INDEX="True",

View File

@ -3,6 +3,7 @@
import os
import signal
import shutil
import subprocess
import sys
import time
@ -37,6 +38,11 @@ from archivebox.tests.test_orm_helpers import use_archivebox_db
pytestmark = pytest.mark.django_db(transaction=True)
def _require_sonic_binary() -> None:
if shutil.which("sonic") is None:
pytest.skip("sonic server binary is required for Sonic worker takeover tests")
def _archive_pages_for_sqlite_reindexing(data_dir: Path, env: dict[str, str], root_url: str) -> None:
add_env = dict(env)
add_env["SEARCH_BACKEND_ENGINE"] = "ripgrep"
@ -63,6 +69,7 @@ def _archive_pages_for_sqlite_reindexing(data_dir: Path, env: dict[str, str], ro
@pytest.mark.timeout(360)
def test_behavior_update_index_only_keeps_server_http_and_search_visible(tmp_path, initialized_archive, recursive_test_site):
_require_sonic_binary()
env = cli_env(
live=True,
@ -106,6 +113,7 @@ def test_behavior_update_index_only_keeps_server_http_and_search_visible(tmp_pat
@pytest.mark.timeout(420)
def test_behavior_update_yields_to_server_then_finishes_visible_indexing(tmp_path, initialized_archive, recursive_test_site):
_require_sonic_binary()
env = cli_env(
live=True,
@ -331,6 +339,7 @@ def test_live_update_index_only_does_not_take_over_server_runtime(tmp_path, init
@pytest.mark.timeout(360)
def test_live_server_keeps_http_runtime_while_update_runs_real_sqlite_indexer(tmp_path, initialized_archive, recursive_test_site):
_require_sonic_binary()
env = cli_env(
live=True,
@ -419,6 +428,7 @@ def test_live_server_keeps_http_runtime_while_update_runs_real_sqlite_indexer(tm
@pytest.mark.timeout(420)
def test_live_update_yields_to_server_then_reclaims_real_sqlite_indexing(tmp_path, initialized_archive, recursive_test_site):
_require_sonic_binary()
env = cli_env(
live=True,

View File

@ -28,7 +28,7 @@ def test_add_view_restarts_stopped_supervisord_runner(tmp_path, recursive_test_s
port = get_free_port()
env = cli_env(
port,
port=port,
PLUGINS="wget",
PUBLIC_ADD_VIEW="True",
PYTEST_CURRENT_TEST="",

View File

@ -110,25 +110,28 @@ def _build_script(body: str) -> str:
output_rel = "index.jsonl"
assert output_rel is not None
responses_root = Path(snapshot.output_dir) / "responses" / snapshot.domain
responses_root = Path(snapshot.output_dir) / "responses"
assert responses_root.exists()
response_file = None
response_rel = None
for candidate in responses_root.rglob("*"):
if not candidate.is_file():
continue
rel = candidate.relative_to(responses_root)
if snapshot.domain not in candidate.relative_to(responses_root).parts:
continue
rel = candidate.relative_to(snapshot.output_dir)
if str(rel) in reserved_snapshot_paths:
continue
if not (Path(snapshot.output_dir) / rel).exists():
response_file = candidate
response_rel = str(rel)
break
response_file = candidate
response_rel = str(rel)
break
if response_file is None:
for candidate in responses_root.rglob("*"):
if not candidate.is_file():
continue
rel = candidate.relative_to(responses_root)
if snapshot.domain not in candidate.relative_to(responses_root).parts:
continue
rel = candidate.relative_to(snapshot.output_dir)
if str(rel) in reserved_snapshot_paths:
continue
response_file = candidate
@ -136,7 +139,7 @@ def _build_script(body: str) -> str:
break
if response_file is None:
response_file = next(p for p in responses_root.rglob("*") if p.is_file())
response_rel = str(response_file.relative_to(responses_root))
response_rel = str(response_file.relative_to(snapshot.output_dir))
response_output_path = Path(snapshot.output_dir) / response_rel
return output_rel, response_file, response_rel, response_output_path
@ -148,14 +151,14 @@ def _build_script(body: str) -> str:
)
safe_json = Path(snapshot.output_dir) / "safe.json"
safe_json.write_text('{"ok": true}', encoding="utf-8")
responses_root = Path(snapshot.output_dir) / "responses" / snapshot.domain
responses_root = Path(snapshot.output_dir) / "responses" / "text" / snapshot.domain
responses_root.mkdir(parents=True, exist_ok=True)
sniffed_response = responses_root / "dangerous-response"
sniffed_response.write_text(
"<!doctype html><html><body><script>window.__archivebox_response__ = true;</script><p>Response Danger</p></body></html>",
encoding="utf-8",
)
return "dangerous.html", "safe.json", "dangerous-response"
return "dangerous.html", "safe.json", str(sniffed_response.relative_to(snapshot.output_dir))
""",
)
return prelude + "\n" + textwrap.dedent(body)