mirror of
https://github.com/ArchiveBox/ArchiveBox.git
synced 2026-09-14 11:06:13 +05:00
fix: improve crawl progress metadata
This commit is contained in:
parent
fc539673ca
commit
5256e5cd33
@ -18,7 +18,7 @@ from django.utils.safestring import mark_safe
|
||||
from django.views import View
|
||||
from django.views.generic.list import ListView
|
||||
from django.views.generic import FormView
|
||||
from django.db.models import Count, Q, Prefetch
|
||||
from django.db.models import Count, Q, Prefetch, Sum
|
||||
from django.contrib import messages
|
||||
from django.contrib.auth.mixins import UserPassesTestMixin
|
||||
from django.views.decorators.csrf import csrf_exempt
|
||||
@ -1573,6 +1573,14 @@ def live_progress_view(request):
|
||||
.annotate(count=Count("id"))
|
||||
):
|
||||
cancelled_snapshot_counts_by_crawl[str(row["crawl_id"])] = row["count"]
|
||||
crawl_output_sizes_by_crawl: dict[str, int] = {str(crawl_id): 0 for crawl_id in active_crawl_ids}
|
||||
if active_crawl_ids:
|
||||
for row in (
|
||||
archiveresult_scope.filter(snapshot__crawl_id__in=active_crawl_ids)
|
||||
.values("snapshot__crawl_id")
|
||||
.annotate(total_size=Sum("output_size"))
|
||||
):
|
||||
crawl_output_sizes_by_crawl[str(row["snapshot__crawl_id"])] = int(row["total_size"] or 0)
|
||||
|
||||
if machine_id is not None:
|
||||
running_processes = Process.objects.filter(
|
||||
@ -1943,6 +1951,8 @@ def live_progress_view(request):
|
||||
crawl_tags = [tag.strip() for tag in (crawl.tags_str or "").replace("\n", ",").split(",") if tag.strip()]
|
||||
persona_name = persona_names_by_id.get(str(crawl.persona_id)) if crawl.persona_id else None
|
||||
persona_name = persona_name or str((crawl.config or {}).get("DEFAULT_PERSONA") or "Default")
|
||||
crawl_output_size = crawl_output_sizes_by_crawl.get(str(crawl.id), 0)
|
||||
avg_snapshot_size = int(crawl_output_size / total_snapshots) if total_snapshots else 0
|
||||
|
||||
# Check if retry_at is in the future (would prevent worker from claiming)
|
||||
retry_at_future = crawl.retry_at > now if crawl.retry_at else False
|
||||
@ -1974,6 +1984,10 @@ def live_progress_view(request):
|
||||
"max_snapshot_size": crawl.snapshot_max_size,
|
||||
"max_crawl_size_display": printable_filesize(crawl.crawl_max_size) if crawl.crawl_max_size else "unlimited",
|
||||
"max_snapshot_size_display": printable_filesize(crawl.snapshot_max_size) if crawl.snapshot_max_size else "unlimited",
|
||||
"crawl_output_size": crawl_output_size,
|
||||
"avg_snapshot_size": avg_snapshot_size,
|
||||
"crawl_output_size_display": printable_filesize(crawl_output_size) if crawl_output_size else "0 B",
|
||||
"avg_snapshot_size_display": printable_filesize(avg_snapshot_size) if avg_snapshot_size else "0 B",
|
||||
"tags": crawl_tags,
|
||||
"urls_count": urls_count,
|
||||
"total_snapshots": total_snapshots,
|
||||
|
||||
@ -998,20 +998,18 @@
|
||||
`;
|
||||
}
|
||||
|
||||
// Show snapshot info or URL count if no snapshots yet
|
||||
const maxUrlsText = (crawl.max_urls || 0) > 0 ? `${crawl.max_urls} max URLs` : 'all URLs';
|
||||
const crawlSizeText = crawl.max_crawl_size_display || 'unlimited';
|
||||
const snapshotSizeText = crawl.max_snapshot_size_display || 'unlimited';
|
||||
const itemCountText = (crawl.total_snapshots || 0) > 0
|
||||
? `${crawl.total_snapshots} snapshot${(crawl.total_snapshots || 0) === 1 ? '' : 's'}`
|
||||
: ((crawl.urls_count || 0) > 0 ? `${crawl.urls_count} URL${(crawl.urls_count || 0) === 1 ? '' : 's'}` : 'no URLs');
|
||||
// Show crawl-scale limits and approximate output sizes from DB metadata.
|
||||
const currentUrlCount = Math.max(crawl.total_snapshots || 0, crawl.urls_count || 0);
|
||||
const maxUrlsText = (crawl.max_urls || 0) > 0 ? crawl.max_urls : 'unlimited';
|
||||
const urlLimitText = `${currentUrlCount} / ${maxUrlsText}`;
|
||||
const crawlSizeLimitText = `${crawl.crawl_output_size_display || '0 B'} / ${crawl.max_crawl_size_display || 'unlimited'}`;
|
||||
const snapshotSizeLimitText = `${crawl.avg_snapshot_size_display || '0 B'} / ${crawl.max_snapshot_size_display || 'unlimited'}`;
|
||||
const crawlBadges = [
|
||||
`<span class="crawl-badge persona"><strong>persona</strong>${escapeHtml(crawl.persona || 'Default')}</span>`,
|
||||
`<span class="crawl-badge limit"><strong>depth</strong>${crawl.max_depth || 0}</span>`,
|
||||
`<span class="crawl-badge limit"><strong>urls</strong>${escapeHtml(maxUrlsText)}</span>`,
|
||||
`<span class="crawl-badge size"><strong>crawl</strong>${escapeHtml(crawlSizeText)}</span>`,
|
||||
`<span class="crawl-badge size"><strong>snapshot</strong>${escapeHtml(snapshotSizeText)}</span>`,
|
||||
`<span class="crawl-badge count"><strong>items</strong>${escapeHtml(itemCountText)}</span>`,
|
||||
`<span class="crawl-badge limit"><strong>urls</strong>${escapeHtml(urlLimitText)}</span>`,
|
||||
`<span class="crawl-badge size"><strong>crawl size</strong>${escapeHtml(crawlSizeLimitText)}</span>`,
|
||||
`<span class="crawl-badge size"><strong>avg snap</strong>${escapeHtml(snapshotSizeLimitText)}</span>`,
|
||||
...(crawl.tags || []).map(tag => `<span class="crawl-badge tag">#${escapeHtml(tag)}</span>`),
|
||||
].join('');
|
||||
const statsHtml = [
|
||||
|
||||
@ -242,9 +242,42 @@ wait_for_runs() {
|
||||
sleep 10
|
||||
done
|
||||
|
||||
while read -r run_id; do
|
||||
gh run watch "${run_id}" --repo "${slug}" --exit-status
|
||||
done < <(jq -r '.[].databaseId' <<<"${runs_json}")
|
||||
while IFS=$'\t' read -r run_id workflow_name; do
|
||||
workflow_name_lower="${workflow_name,,}"
|
||||
if [[ "${workflow_name_lower}" == *"release state"* ]]; then
|
||||
gh run watch "${run_id}" --repo "${slug}" --exit-status
|
||||
continue
|
||||
fi
|
||||
if [[ "${workflow_name_lower}" != *"test"* ]]; then
|
||||
echo "Skipping non-gating workflow: ${workflow_name}"
|
||||
continue
|
||||
fi
|
||||
|
||||
attempts=0
|
||||
while :; do
|
||||
precheck_state="$(
|
||||
gh run view "${run_id}" --repo "${slug}" --json jobs --jq '
|
||||
[.jobs[] | select((.name | ascii_downcase) | test("precheck|pre-commit|prek"))][0]
|
||||
| if . == null then "missing:" else ((.status // "") + ":" + (.conclusion // "")) end
|
||||
'
|
||||
)"
|
||||
case "${precheck_state}" in
|
||||
completed:success|completed:skipped)
|
||||
break
|
||||
;;
|
||||
completed:failure|completed:cancelled|completed:timed_out)
|
||||
gh run view "${run_id}" --repo "${slug}"
|
||||
return 1
|
||||
;;
|
||||
esac
|
||||
attempts=$((attempts + 1))
|
||||
if [[ "${attempts}" -ge 120 ]]; then
|
||||
echo "Timed out waiting for ${workflow_name} precheck job" >&2
|
||||
return 1
|
||||
fi
|
||||
sleep 5
|
||||
done
|
||||
done < <(jq -r '.[] | [.databaseId, .workflowName] | @tsv' <<<"${runs_json}")
|
||||
}
|
||||
|
||||
wait_for_pypi() {
|
||||
@ -321,16 +354,32 @@ create_release() {
|
||||
publish_artifacts() {
|
||||
local version="$1"
|
||||
local pypi_token="${UV_PUBLISH_TOKEN:-${PYPI_TOKEN:-${PYPI_PAT_SECRET:-}}}"
|
||||
local artifact_prefix="${PYPI_PACKAGE//-/_}"
|
||||
local artifacts=()
|
||||
local dist_dir
|
||||
|
||||
shopt -s nullglob
|
||||
for dist_dir in "${WORKSPACE_DIR}/dist" "${REPO_DIR}/dist"; do
|
||||
artifacts+=("${dist_dir}/${PYPI_PACKAGE}-${version}"*)
|
||||
if [[ "${artifact_prefix}" != "${PYPI_PACKAGE}" ]]; then
|
||||
artifacts+=("${dist_dir}/${artifact_prefix}-${version}"*)
|
||||
fi
|
||||
done
|
||||
shopt -u nullglob
|
||||
|
||||
if curl -fsSL "https://pypi.org/pypi/${PYPI_PACKAGE}/json" | jq -e --arg version "${version}" '.releases[$version] | length > 0' >/dev/null 2>&1; then
|
||||
echo "${PYPI_PACKAGE} ${version} already published on PyPI"
|
||||
else
|
||||
if [[ -n "${pypi_token}" ]]; then
|
||||
UV_PUBLISH_TOKEN="${pypi_token}" uv publish --username=__token__ dist/*
|
||||
else
|
||||
echo "Missing PyPI credentials: set UV_PUBLISH_TOKEN or PYPI_TOKEN" >&2
|
||||
if [[ "${#artifacts[@]}" -eq 0 ]]; then
|
||||
echo "Missing build artifacts for ${PYPI_PACKAGE}==${version}" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [[ -n "${pypi_token}" ]]; then
|
||||
UV_PUBLISH_TOKEN="${pypi_token}" uv publish --username=__token__ "${artifacts[@]}"
|
||||
else
|
||||
uv publish --username=__token__ "${artifacts[@]}"
|
||||
fi
|
||||
fi
|
||||
|
||||
wait_for_pypi "${PYPI_PACKAGE}" "${version}"
|
||||
@ -377,7 +426,6 @@ main() {
|
||||
return 1
|
||||
fi
|
||||
run_checks
|
||||
wait_for_runs "${slug}" push "$(git rev-parse HEAD)" "push"
|
||||
else
|
||||
echo "Current version ${version} is behind latest GitHub release ${latest}" >&2
|
||||
return 1
|
||||
@ -386,10 +434,8 @@ main() {
|
||||
publish_artifacts "${version}"
|
||||
create_release "${slug}" "${version}"
|
||||
|
||||
latest="$(latest_release_version "${slug}")"
|
||||
relation="$(compare_versions "${latest}" "${version}")"
|
||||
if [[ "${relation}" != "eq" ]]; then
|
||||
echo "GitHub release version mismatch: expected ${version}, got ${latest}" >&2
|
||||
if ! gh release view "${TAG_PREFIX}${version}" --repo "${slug}" >/dev/null 2>&1; then
|
||||
echo "GitHub release ${TAG_PREFIX}${version} was not found after creation" >&2
|
||||
return 1
|
||||
fi
|
||||
|
||||
|
||||
Loading…
Reference in New Issue
Block a user