fix: improve crawl progress metadata

This commit is contained in:
Nick Sweeting 2026-05-27 10:58:54 -07:00
parent fc539673ca
commit 5256e5cd33
No known key found for this signature in database
3 changed files with 82 additions and 24 deletions

View File

@ -18,7 +18,7 @@ from django.utils.safestring import mark_safe
from django.views import View
from django.views.generic.list import ListView
from django.views.generic import FormView
from django.db.models import Count, Q, Prefetch
from django.db.models import Count, Q, Prefetch, Sum
from django.contrib import messages
from django.contrib.auth.mixins import UserPassesTestMixin
from django.views.decorators.csrf import csrf_exempt
@ -1573,6 +1573,14 @@ def live_progress_view(request):
.annotate(count=Count("id"))
):
cancelled_snapshot_counts_by_crawl[str(row["crawl_id"])] = row["count"]
crawl_output_sizes_by_crawl: dict[str, int] = {str(crawl_id): 0 for crawl_id in active_crawl_ids}
if active_crawl_ids:
for row in (
archiveresult_scope.filter(snapshot__crawl_id__in=active_crawl_ids)
.values("snapshot__crawl_id")
.annotate(total_size=Sum("output_size"))
):
crawl_output_sizes_by_crawl[str(row["snapshot__crawl_id"])] = int(row["total_size"] or 0)
if machine_id is not None:
running_processes = Process.objects.filter(
@ -1943,6 +1951,8 @@ def live_progress_view(request):
crawl_tags = [tag.strip() for tag in (crawl.tags_str or "").replace("\n", ",").split(",") if tag.strip()]
persona_name = persona_names_by_id.get(str(crawl.persona_id)) if crawl.persona_id else None
persona_name = persona_name or str((crawl.config or {}).get("DEFAULT_PERSONA") or "Default")
crawl_output_size = crawl_output_sizes_by_crawl.get(str(crawl.id), 0)
avg_snapshot_size = int(crawl_output_size / total_snapshots) if total_snapshots else 0
# Check if retry_at is in the future (would prevent worker from claiming)
retry_at_future = crawl.retry_at > now if crawl.retry_at else False
@ -1974,6 +1984,10 @@ def live_progress_view(request):
"max_snapshot_size": crawl.snapshot_max_size,
"max_crawl_size_display": printable_filesize(crawl.crawl_max_size) if crawl.crawl_max_size else "unlimited",
"max_snapshot_size_display": printable_filesize(crawl.snapshot_max_size) if crawl.snapshot_max_size else "unlimited",
"crawl_output_size": crawl_output_size,
"avg_snapshot_size": avg_snapshot_size,
"crawl_output_size_display": printable_filesize(crawl_output_size) if crawl_output_size else "0 B",
"avg_snapshot_size_display": printable_filesize(avg_snapshot_size) if avg_snapshot_size else "0 B",
"tags": crawl_tags,
"urls_count": urls_count,
"total_snapshots": total_snapshots,

View File

@ -998,20 +998,18 @@
`;
}
// Show snapshot info or URL count if no snapshots yet
const maxUrlsText = (crawl.max_urls || 0) > 0 ? `${crawl.max_urls} max URLs` : 'all URLs';
const crawlSizeText = crawl.max_crawl_size_display || 'unlimited';
const snapshotSizeText = crawl.max_snapshot_size_display || 'unlimited';
const itemCountText = (crawl.total_snapshots || 0) > 0
? `${crawl.total_snapshots} snapshot${(crawl.total_snapshots || 0) === 1 ? '' : 's'}`
: ((crawl.urls_count || 0) > 0 ? `${crawl.urls_count} URL${(crawl.urls_count || 0) === 1 ? '' : 's'}` : 'no URLs');
// Show crawl-scale limits and approximate output sizes from DB metadata.
const currentUrlCount = Math.max(crawl.total_snapshots || 0, crawl.urls_count || 0);
const maxUrlsText = (crawl.max_urls || 0) > 0 ? crawl.max_urls : 'unlimited';
const urlLimitText = `${currentUrlCount} / ${maxUrlsText}`;
const crawlSizeLimitText = `${crawl.crawl_output_size_display || '0 B'} / ${crawl.max_crawl_size_display || 'unlimited'}`;
const snapshotSizeLimitText = `${crawl.avg_snapshot_size_display || '0 B'} / ${crawl.max_snapshot_size_display || 'unlimited'}`;
const crawlBadges = [
`<span class="crawl-badge persona"><strong>persona</strong>${escapeHtml(crawl.persona || 'Default')}</span>`,
`<span class="crawl-badge limit"><strong>depth</strong>${crawl.max_depth || 0}</span>`,
`<span class="crawl-badge limit"><strong>urls</strong>${escapeHtml(maxUrlsText)}</span>`,
`<span class="crawl-badge size"><strong>crawl</strong>${escapeHtml(crawlSizeText)}</span>`,
`<span class="crawl-badge size"><strong>snapshot</strong>${escapeHtml(snapshotSizeText)}</span>`,
`<span class="crawl-badge count"><strong>items</strong>${escapeHtml(itemCountText)}</span>`,
`<span class="crawl-badge limit"><strong>urls</strong>${escapeHtml(urlLimitText)}</span>`,
`<span class="crawl-badge size"><strong>crawl size</strong>${escapeHtml(crawlSizeLimitText)}</span>`,
`<span class="crawl-badge size"><strong>avg snap</strong>${escapeHtml(snapshotSizeLimitText)}</span>`,
...(crawl.tags || []).map(tag => `<span class="crawl-badge tag">#${escapeHtml(tag)}</span>`),
].join('');
const statsHtml = [

View File

@ -242,9 +242,42 @@ wait_for_runs() {
sleep 10
done
while read -r run_id; do
gh run watch "${run_id}" --repo "${slug}" --exit-status
done < <(jq -r '.[].databaseId' <<<"${runs_json}")
while IFS=$'\t' read -r run_id workflow_name; do
workflow_name_lower="${workflow_name,,}"
if [[ "${workflow_name_lower}" == *"release state"* ]]; then
gh run watch "${run_id}" --repo "${slug}" --exit-status
continue
fi
if [[ "${workflow_name_lower}" != *"test"* ]]; then
echo "Skipping non-gating workflow: ${workflow_name}"
continue
fi
attempts=0
while :; do
precheck_state="$(
gh run view "${run_id}" --repo "${slug}" --json jobs --jq '
[.jobs[] | select((.name | ascii_downcase) | test("precheck|pre-commit|prek"))][0]
| if . == null then "missing:" else ((.status // "") + ":" + (.conclusion // "")) end
'
)"
case "${precheck_state}" in
completed:success|completed:skipped)
break
;;
completed:failure|completed:cancelled|completed:timed_out)
gh run view "${run_id}" --repo "${slug}"
return 1
;;
esac
attempts=$((attempts + 1))
if [[ "${attempts}" -ge 120 ]]; then
echo "Timed out waiting for ${workflow_name} precheck job" >&2
return 1
fi
sleep 5
done
done < <(jq -r '.[] | [.databaseId, .workflowName] | @tsv' <<<"${runs_json}")
}
wait_for_pypi() {
@ -321,16 +354,32 @@ create_release() {
publish_artifacts() {
local version="$1"
local pypi_token="${UV_PUBLISH_TOKEN:-${PYPI_TOKEN:-${PYPI_PAT_SECRET:-}}}"
local artifact_prefix="${PYPI_PACKAGE//-/_}"
local artifacts=()
local dist_dir
shopt -s nullglob
for dist_dir in "${WORKSPACE_DIR}/dist" "${REPO_DIR}/dist"; do
artifacts+=("${dist_dir}/${PYPI_PACKAGE}-${version}"*)
if [[ "${artifact_prefix}" != "${PYPI_PACKAGE}" ]]; then
artifacts+=("${dist_dir}/${artifact_prefix}-${version}"*)
fi
done
shopt -u nullglob
if curl -fsSL "https://pypi.org/pypi/${PYPI_PACKAGE}/json" | jq -e --arg version "${version}" '.releases[$version] | length > 0' >/dev/null 2>&1; then
echo "${PYPI_PACKAGE} ${version} already published on PyPI"
else
if [[ -n "${pypi_token}" ]]; then
UV_PUBLISH_TOKEN="${pypi_token}" uv publish --username=__token__ dist/*
else
echo "Missing PyPI credentials: set UV_PUBLISH_TOKEN or PYPI_TOKEN" >&2
if [[ "${#artifacts[@]}" -eq 0 ]]; then
echo "Missing build artifacts for ${PYPI_PACKAGE}==${version}" >&2
return 1
fi
if [[ -n "${pypi_token}" ]]; then
UV_PUBLISH_TOKEN="${pypi_token}" uv publish --username=__token__ "${artifacts[@]}"
else
uv publish --username=__token__ "${artifacts[@]}"
fi
fi
wait_for_pypi "${PYPI_PACKAGE}" "${version}"
@ -377,7 +426,6 @@ main() {
return 1
fi
run_checks
wait_for_runs "${slug}" push "$(git rev-parse HEAD)" "push"
else
echo "Current version ${version} is behind latest GitHub release ${latest}" >&2
return 1
@ -386,10 +434,8 @@ main() {
publish_artifacts "${version}"
create_release "${slug}" "${version}"
latest="$(latest_release_version "${slug}")"
relation="$(compare_versions "${latest}" "${version}")"
if [[ "${relation}" != "eq" ]]; then
echo "GitHub release version mismatch: expected ${version}, got ${latest}" >&2
if ! gh release view "${TAG_PREFIX}${version}" --repo "${slug}" >/dev/null 2>&1; then
echo "GitHub release ${TAG_PREFIX}${version} was not found after creation" >&2
return 1
fi