import asyncio
import html
import importlib
import json
import mimetypes
import os
import posixpath
import queue
import re
import stat
import sys
import threading
import time
import zipfile
from collections.abc import Callable
from datetime import UTC, datetime
from pathlib import Path
from urllib.parse import quote, urlencode, urljoin
from abx_plugins.plugins.archivewebpage import replay_preview as archivewebpage_replay
from django.contrib.staticfiles import finders
from django.core.handlers.asgi import ASGIRequest
from django.http import Http404, HttpResponse, HttpResponseNotModified, StreamingHttpResponse
from django.template import TemplateDoesNotExist, loader
from django.utils._os import safe_join
from django.utils.http import http_date
from django.utils.translation import gettext as _
from django.views import static
from archivebox.config.common import get_config
from archivebox.misc.logging_util import printable_filesize
_HASHES_CACHE: dict[Path, tuple[float, dict[str, str]]] = {}
IMG_SRC_ATTR_RE = re.compile(r'(]*?\s(?:src|data-src)=["\'])([^"\']+)(["\'])', re.IGNORECASE)
TRANSFORMED_HTML_PREVIEW_STYLE = """"""
def _load_hash_map(snapshot_dir: Path) -> dict[str, str] | None:
hashes_path = snapshot_dir / "hashes" / "hashes.json"
if not hashes_path.exists():
return None
try:
mtime = hashes_path.stat().st_mtime
except OSError:
return None
cached = _HASHES_CACHE.get(hashes_path)
if cached and cached[0] == mtime:
return cached[1]
try:
data = json.loads(hashes_path.read_text(encoding="utf-8"))
except (json.JSONDecodeError, OSError, TypeError, ValueError):
return None
file_map = {str(entry.get("path")): entry.get("hash") for entry in data.get("files", []) if entry.get("path")}
_HASHES_CACHE[hashes_path] = (mtime, file_map)
return file_map
def _hash_for_path(document_root: Path, rel_path: str) -> str | None:
file_map = _load_hash_map(document_root)
if not file_map:
return None
return file_map.get(rel_path)
def _resolve_archive_path(document_root: str | Path, rel_path: str) -> tuple[Path, str]:
rel_path = posixpath.normpath(rel_path).lstrip("/") if rel_path else ""
fullpath = Path(safe_join(document_root, rel_path))
if os.access(fullpath, os.R_OK):
return fullpath, rel_path
root = Path(document_root)
current = root
resolved_parts: list[str] = []
for part in Path(rel_path).parts:
exact = current / part
if os.access(exact, os.R_OK):
current = exact
resolved_parts.append(part)
continue
folded_part = part.casefold()
try:
match = next((child for child in current.iterdir() if child.name.casefold() == folded_part), None)
except OSError:
match = None
if match is None:
return fullpath, rel_path
current = match
resolved_parts.append(match.name)
return current, posixpath.join(*resolved_parts) if resolved_parts else ""
def _cache_policy(config=None, **config_kwargs) -> str:
config = config or get_config(resolve_plugins=False, **config_kwargs)
return "private" if config.PERMISSIONS == "private" else "public"
def _format_direntry_timestamp(stat_result: os.stat_result) -> str:
timestamp = stat_result.st_birthtime if sys.platform == "darwin" else stat_result.st_mtime
return datetime.fromtimestamp(timestamp, tz=UTC).strftime("%Y-%m-%d %H:%M")
def _safe_zip_stem(name: str) -> str:
safe_name = re.sub(r"[^A-Za-z0-9._-]+", "-", name).strip("._-")
return safe_name or "archivebox"
class _StreamingQueueWriter:
"""Expose a write-only file-like object so zipfile can stream into a queue."""
def __init__(self, output_queue: queue.Queue[bytes | BaseException | object]) -> None:
self.output_queue = output_queue
self.position = 0
def write(self, data: bytes) -> int:
if data:
self.output_queue.put(data)
self.position += len(data)
return len(data)
def tell(self) -> int:
return self.position
def flush(self) -> None:
return None
def close(self) -> None:
return None
def writable(self) -> bool:
return True
def seekable(self) -> bool:
return False
def _iter_visible_files(root: Path):
"""Yield non-hidden files in a stable order so ZIP output is deterministic."""
for current_root, dirnames, filenames in os.walk(root):
dirnames[:] = sorted(dirname for dirname in dirnames if not dirname.startswith("."))
for filename in sorted(name for name in filenames if not name.startswith(".")):
yield Path(current_root) / filename
def _build_directory_zip_response(
fullpath: Path,
path: str,
*,
is_archive_replay: bool,
use_async_stream: bool,
config=None,
) -> StreamingHttpResponse:
root_name = _safe_zip_stem(fullpath.name or Path(path).name or "archivebox")
sentinel = object()
output_queue: queue.Queue[bytes | BaseException | object] = queue.Queue(maxsize=8)
initial_chunk_target = 64 * 1024
initial_chunk_wait = 0.05
def build_zip() -> None:
# zipfile wants a write-only file object. Feed those bytes straight into
# a queue so the response can stream them out as soon as they are ready.
writer = _StreamingQueueWriter(output_queue)
try:
with zipfile.ZipFile(writer, mode="w", compression=zipfile.ZIP_DEFLATED, compresslevel=6) as zip_file:
for entry in _iter_visible_files(fullpath):
rel_parts = entry.relative_to(fullpath).parts
arcname = Path(root_name, *rel_parts).as_posix()
zip_file.write(entry, arcname)
except (OSError, RuntimeError, TypeError, ValueError, zipfile.BadZipFile) as err:
output_queue.put(err)
finally:
output_queue.put(sentinel)
threading.Thread(target=build_zip, name=f"zip-stream-{root_name}", daemon=True).start()
def iter_zip_chunks():
# Emit a meaningful first chunk quickly so browsers show the download
# immediately instead of waiting on dozens of tiny ZIP header writes.
first_chunk = bytearray()
initial_deadline = time.monotonic() + initial_chunk_wait
while True:
timeout = max(initial_deadline - time.monotonic(), 0) if len(first_chunk) < initial_chunk_target else None
try:
chunk = output_queue.get(timeout=timeout) if timeout is not None else output_queue.get()
except queue.Empty:
if first_chunk:
yield bytes(first_chunk)
first_chunk.clear()
continue
chunk = output_queue.get()
if chunk is sentinel:
if first_chunk:
yield bytes(first_chunk)
break
if isinstance(chunk, BaseException):
raise chunk
if len(first_chunk) < initial_chunk_target:
first_chunk.extend(chunk)
if len(first_chunk) >= initial_chunk_target or time.monotonic() >= initial_deadline:
yield bytes(first_chunk)
first_chunk.clear()
continue
yield chunk
async def stream_zip_async():
# Django ASGI buffers sync StreamingHttpResponse iterators by consuming
# them into a list. Drive the same sync iterator from a worker thread so
# Daphne can send each chunk as it arrives instead of buffering the ZIP.
iterator = iter(iter_zip_chunks())
while True:
chunk = await asyncio.to_thread(next, iterator, None)
if chunk is None:
break
yield chunk
response = StreamingHttpResponse(
stream_zip_async() if use_async_stream else iter_zip_chunks(),
content_type="application/zip",
)
response.headers["Content-Disposition"] = f'attachment; filename="{root_name}.zip"'
response.headers["Cache-Control"] = f"{_cache_policy(config=config)}, max-age=60, stale-while-revalidate=300"
response.headers["Last-Modified"] = http_date(fullpath.stat().st_mtime)
response.headers["X-Accel-Buffering"] = "no"
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type="application/zip",
is_archive_replay=is_archive_replay,
config=config,
)
async def _stream_ranged_file_async(ranged_file: "RangedFileReader"):
iterator = iter(ranged_file)
try:
while True:
chunk = await asyncio.to_thread(next, iterator, None)
if chunk is None:
break
yield chunk
finally:
ranged_file.close()
def _render_directory_index(request, path: str, fullpath: Path) -> HttpResponse:
try:
template = loader.select_template(
[
"static/directory_index.html",
"static/directory_index",
],
)
except TemplateDoesNotExist:
return static.directory_index(path, fullpath)
entries = []
file_list = []
visible_entries = sorted(
(entry for entry in fullpath.iterdir() if not entry.name.startswith(".")),
key=lambda entry: (not entry.is_dir(), entry.name.lower()),
)
for entry in visible_entries:
url = str(entry.relative_to(fullpath))
if entry.is_dir():
url += "/"
file_list.append(url)
stat_result = entry.stat()
entries.append(
{
"name": url,
"url": url,
"is_dir": entry.is_dir(),
"size": "ā" if entry.is_dir() else printable_filesize(stat_result.st_size),
"timestamp": _format_direntry_timestamp(stat_result),
},
)
zip_query = request.GET.copy()
zip_query["download"] = "zip"
zip_url = request.path
if zip_query:
zip_url = f"{zip_url}?{zip_query.urlencode()}"
directory_query = "?files=1" if request.GET.get("files") else ""
path_suffix = f"{path.strip('/')}/" if path.strip("/") else ""
snapshot_page_url = request.path
if path_suffix and snapshot_page_url.endswith(path_suffix):
snapshot_page_url = snapshot_page_url[: -len(path_suffix)]
context = {
"directory": f"{path}/",
"file_list": file_list,
"entries": entries,
"zip_url": zip_url,
"directory_query": directory_query,
"snapshot_page_url": snapshot_page_url,
}
return HttpResponse(template.render(context))
# Ensure common web types are mapped consistently across platforms.
mimetypes.add_type("text/html", ".html")
mimetypes.add_type("text/html", ".htm")
mimetypes.add_type("text/css", ".css")
mimetypes.add_type("application/javascript", ".js")
mimetypes.add_type("application/json", ".json")
mimetypes.add_type("application/x-ndjson", ".jsonl")
mimetypes.add_type("text/markdown", ".md")
mimetypes.add_type("text/yaml", ".yml")
mimetypes.add_type("text/yaml", ".yaml")
mimetypes.add_type("text/csv", ".csv")
mimetypes.add_type("text/tab-separated-values", ".tsv")
mimetypes.add_type("application/xml", ".xml")
mimetypes.add_type("image/svg+xml", ".svg")
mimetypes.add_type("multipart/related", ".mhtml")
mimetypes.add_type("multipart/related", ".mht")
try:
_markdown = importlib.import_module("markdown").markdown
except ImportError:
_markdown: Callable[..., str] | None = None
MARKDOWN_INLINE_LINK_RE = re.compile(r"\[([^\]]+)\]\(([^)\s]+(?:\([^)]*\)[^)\s]*)*)\)")
MARKDOWN_INLINE_IMAGE_RE = re.compile(r"!\[([^\]]*)\]\(([^)]+)\)")
MARKDOWN_BOLD_RE = re.compile(r"\*\*([^*]+)\*\*")
MARKDOWN_ITALIC_RE = re.compile(r"(?]*>")
HTML_BODY_RE = re.compile(r"
]*>", "", candidate, flags=re.IGNORECASE) candidate = re.sub(r"
\s*$", "", candidate, flags=re.IGNORECASE) return candidate.strip() def _looks_like_markdown(text: str) -> bool: lower = text.lower() if "" in lower: return False md_markers = 0 md_markers += len(re.findall(r"^\s{0,3}#{1,6}\s+\S", text, flags=re.MULTILINE)) md_markers += len(re.findall(r"^\s*[-*+]\s+\S", text, flags=re.MULTILINE)) md_markers += len(re.findall(r"^\s*\d+\.\s+\S", text, flags=re.MULTILINE)) md_markers += text.count("[TOC]") md_markers += len(MARKDOWN_INLINE_LINK_RE.findall(text)) md_markers += text.count("\n---") + text.count("\n***") return md_markers >= 6 def _render_text_preview_document(text: str, title: str) -> str: escaped_title = html.escape(title) escaped_text = html.escape(text) return f"""{escaped_text}
"""
def _render_image_preview_document(image_url: str, title: str) -> str:
escaped_title = html.escape(title)
escaped_url = html.escape(image_url, quote=True)
return f"""
")
in_code = True
continue
if in_code:
html_lines.append(html.escape(line))
continue
if not stripped:
close_lists()
if in_blockquote:
html_lines.append("")
in_blockquote = False
html_lines.append("
")
continue
heading_match = re.match(r"^\s*((?:<[^>]+>\s*)*)(#{1,6})\s+(.*)$", line)
if heading_match:
close_lists()
if in_blockquote:
html_lines.append("")
in_blockquote = False
leading_tags = heading_match.group(1).strip()
level = len(heading_match.group(2))
content = heading_match.group(3).strip()
if leading_tags:
html_lines.append(leading_tags)
html_lines.append(f'{render_inline(content)} ')
continue
if stripped in ("---", "***"):
close_lists()
html_lines.append("
")
continue
if stripped.startswith("> "):
if not in_blockquote:
close_lists()
html_lines.append("")
in_blockquote = True
content = stripped[2:]
html_lines.append(render_inline(content))
continue
else:
if in_blockquote:
html_lines.append("
")
in_blockquote = False
ul_match = re.match(r"^\s*[-*+]\s+(.*)$", line)
if ul_match:
if in_ol:
html_lines.append("")
in_ol = False
if not in_ul:
html_lines.append("")
in_ul = True
html_lines.append(f"- {render_inline(ul_match.group(1))}
")
continue
ol_match = re.match(r"^\s*\d+\.\s+(.*)$", line)
if ol_match:
if in_ul:
html_lines.append("
")
in_ul = False
if not in_ol:
html_lines.append("")
in_ol = True
html_lines.append(f"- {render_inline(ol_match.group(1))}
")
continue
close_lists()
# Inline conversions (leave raw HTML intact)
if stripped == "[TOC]":
toc_items = []
for level, title, slug in headings:
toc_items.append(
f'- {title}
',
)
html_lines.append(
'",
)
continue
html_lines.append(f"{render_inline(line)}
")
close_lists()
if in_blockquote:
html_lines.append("")
if in_code:
html_lines.append("
")
return "\n".join(html_lines)
def _render_markdown_document(markdown_text: str) -> str:
body = _render_markdown_fallback(markdown_text)
wrapped = (
''
''
""
""
f"{body}"
""
)
return wrapped
def _content_type_base(content_type: str) -> str:
return (content_type or "").split(";", 1)[0].strip().lower()
def _is_risky_replay_document(fullpath: Path, content_type: str) -> bool:
if fullpath.suffix.lower() in RISKY_REPLAY_EXTENSIONS:
return True
if _content_type_base(content_type) in RISKY_REPLAY_MIMETYPES:
return True
# Unknown archived response paths often have no extension. Sniff a small prefix
# so one-domain no-JS mode still catches HTML/SVG documents.
try:
head = fullpath.read_bytes()[:4096].decode("utf-8", errors="ignore").lower()
except (OSError, UnicodeDecodeError):
return False
return any(marker in head for marker in RISKY_REPLAY_MARKERS)
def _apply_archive_replay_headers(
response: HttpResponse,
*,
fullpath: Path,
content_type: str,
is_archive_replay: bool,
config=None,
**config_kwargs,
) -> HttpResponse:
if not is_archive_replay:
return response
response.headers.setdefault("X-Content-Type-Options", "nosniff")
config = config or get_config(resolve_plugins=False, **config_kwargs)
response.headers.setdefault("X-ArchiveBox-Security-Mode", config.SERVER_SECURITY_MODE)
is_risky_replay = _is_risky_replay_document(fullpath, content_type)
if config.SHOULD_NEUTER_RISKY_REPLAY and is_risky_replay and "Content-Security-Policy" not in response.headers:
response.headers["Content-Security-Policy"] = (
"sandbox; "
"default-src 'self' data: blob:; "
"script-src 'none'; "
"object-src 'none'; "
"base-uri 'none'; "
"form-action 'none'; "
"connect-src 'none'; "
"worker-src 'none'; "
"frame-ancestors 'self'; "
"style-src 'self' 'unsafe-inline' data: blob:; "
"img-src 'self' data: blob:; "
"media-src 'self' data: blob:; "
"font-src 'self' data: blob:;"
)
response.headers.setdefault("Referrer-Policy", "no-referrer")
return response
def _is_asgi_request(request) -> bool:
return isinstance(request, ASGIRequest) or "scope" in request.__dict__
def serve_static_with_byterange_support(request, path, document_root=None, show_indexes=False, is_archive_replay: bool = False):
"""
Overrides Django's built-in django.views.static.serve function to support byte range requests.
This allows you to do things like seek into the middle of a huge mp4 or WACZ without downloading the whole file.
https://github.com/satchamo/django/commit/2ce75c5c4bee2a858c0214d136bfcd351fcde11d
"""
assert document_root
config = request.__dict__.get("archivebox_config")
if config is None:
config = get_config(resolve_plugins=False)
fullpath, path = _resolve_archive_path(document_root, path)
if os.access(fullpath, os.R_OK) and fullpath.is_dir():
if request.GET.get("download") == "zip" and show_indexes:
return _build_directory_zip_response(
fullpath,
path,
is_archive_replay=is_archive_replay,
use_async_stream=_is_asgi_request(request),
config=config,
)
if show_indexes:
response = _render_directory_index(request, path, fullpath)
response.headers["Cache-Control"] = f"{_cache_policy(config=config)}, max-age=60, stale-while-revalidate=300"
response.headers["Last-Modified"] = http_date(fullpath.stat().st_mtime)
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type="text/html",
is_archive_replay=is_archive_replay,
config=config,
)
raise Http404(_("Directory indexes are not allowed here."))
if not os.access(fullpath, os.R_OK):
raise Http404(_("ā%(path)sā does not exist") % {"path": fullpath})
statobj = fullpath.stat()
document_root = Path(document_root) if document_root else None
rel_path = path
etag = None
if document_root:
file_hash = _hash_for_path(document_root, rel_path)
if file_hash:
etag = f'"{file_hash}"'
if etag:
inm = request.META.get("HTTP_IF_NONE_MATCH")
if inm:
inm_list = [item.strip() for item in inm.split(",")]
if etag in inm_list or etag.strip('"') in [i.strip('"') for i in inm_list]:
not_modified = HttpResponseNotModified()
not_modified.headers["ETag"] = etag
not_modified.headers["Cache-Control"] = f"{_cache_policy(config=config)}, max-age=31536000, immutable"
not_modified.headers["Last-Modified"] = http_date(statobj.st_mtime)
return _apply_archive_replay_headers(
not_modified,
fullpath=fullpath,
content_type="",
is_archive_replay=is_archive_replay,
config=config,
)
content_type, encoding = mimetypes.guess_type(str(fullpath))
content_type = content_type or "application/octet-stream"
# Add charset for text-like types (best guess), but don't override the type.
is_text_like = content_type.startswith("text/") or content_type in {
"application/json",
"application/javascript",
"application/xml",
"application/x-ndjson",
"image/svg+xml",
}
if is_text_like and "charset=" not in content_type:
content_type = f"{content_type}; charset=utf-8"
preview_as_text_html = (
bool(request.GET.get("preview"))
and is_text_like
and not content_type.startswith("text/html")
and not content_type.startswith("image/svg+xml")
)
preview_as_image_html = (
bool(request.GET.get("preview")) and content_type.startswith("image/") and not content_type.startswith("image/svg+xml")
)
preview_as_archivewebpage_html = bool(request.GET.get("preview")) and archivewebpage_replay.is_replay_target(fullpath.name)
# Respect the If-Modified-Since header for non-markdown responses.
if not content_type.startswith(("text/plain", "text/html")) and not static.was_modified_since(
request.META.get("HTTP_IF_MODIFIED_SINCE"),
statobj.st_mtime,
):
return _apply_archive_replay_headers(
HttpResponseNotModified(),
fullpath=fullpath,
content_type=content_type,
is_archive_replay=is_archive_replay,
config=config,
)
# Wrap text-like outputs in HTML when explicitly requested for iframe previewing.
if preview_as_text_html:
try:
max_preview_size = 10 * 1024 * 1024
if statobj.st_size <= max_preview_size:
decoded = fullpath.read_text(encoding="utf-8", errors="replace")
wrapped = _render_text_preview_document(decoded, fullpath.name)
response = HttpResponse(wrapped, content_type="text/html; charset=utf-8")
_set_transformed_response_headers(response, fullpath, statobj, encoding, config)
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type="text/html; charset=utf-8",
is_archive_replay=is_archive_replay,
config=config,
)
except (OSError, UnicodeDecodeError, ValueError):
preview_as_text_html = False
if preview_as_image_html:
try:
preview_query = request.GET.copy()
preview_query.pop("preview", None)
raw_image_url = request.path
if preview_query:
raw_image_url = f"{raw_image_url}?{urlencode(list(preview_query.lists()), doseq=True)}"
wrapped = _render_image_preview_document(raw_image_url, fullpath.name)
response = HttpResponse(wrapped, content_type="text/html; charset=utf-8")
_set_transformed_response_headers(response, fullpath, statobj, encoding, config)
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type="text/html; charset=utf-8",
is_archive_replay=is_archive_replay,
config=config,
)
except (OSError, ValueError):
preview_as_image_html = False
if preview_as_archivewebpage_html:
try:
raw_query = request.GET.copy()
raw_query.pop("preview", None)
raw_output_path = request.path
if raw_query:
raw_output_path = f"{raw_output_path}?{raw_query.urlencode()}"
body, preview_content_type, headers = archivewebpage_replay.render_preview_response(
fullpath.name,
raw_output_path,
wacz_path=fullpath,
fallback_url=request.archivebox_snapshot_url or "",
last_modified=http_date(statobj.st_mtime),
etag=etag or "",
cache_control=(
f"{_cache_policy(config=config)}, max-age=31536000, immutable"
if etag
else f"{_cache_policy(config=config)}, max-age=60, stale-while-revalidate=300"
),
content_encoding=encoding or "",
)
response = HttpResponse(body, content_type=preview_content_type)
for key, value in headers.items():
response.headers[key] = value
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type=preview_content_type,
is_archive_replay=is_archive_replay,
config=config,
)
except (OSError, RuntimeError, ValueError):
preview_as_archivewebpage_html = False
# Heuristic fix: some archived HTML outputs are stored with HTML-escaped markup
# or markdown sources. If so, render sensibly.
if content_type.startswith(("text/plain", "text/html")):
try:
max_unescape_size = 10 * 1024 * 1024 # 10MB cap to avoid heavy memory use
if statobj.st_size <= max_unescape_size:
raw = fullpath.read_bytes()
decoded = raw.decode("utf-8", errors="replace")
escaped_count = decoded.count("<") + decoded.count(">")
tag_count = decoded.count("<")
if escaped_count and escaped_count > tag_count * 2:
decoded = html.unescape(decoded)
rewritten_html, rewritten_count = ("", 0)
if content_type.startswith("text/html") and document_root:
rewritten_html, rewritten_count = _rewrite_html_image_sources_for_request(request, decoded, document_root, rel_path)
markdown_candidate = _extract_markdown_candidate(decoded)
if _looks_like_markdown(markdown_candidate):
wrapped = _render_markdown_document(markdown_candidate)
wrapped, _rewrite_count = _rewrite_html_image_sources_for_request(request, wrapped, document_root, rel_path)
wrapped = _apply_transformed_html_preview_style(wrapped)
response = HttpResponse(wrapped, content_type="text/html; charset=utf-8")
_set_transformed_response_headers(response, fullpath, statobj, encoding, config)
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type="text/html; charset=utf-8",
is_archive_replay=is_archive_replay,
config=config,
)
if rewritten_count:
rewritten_html = _apply_transformed_html_preview_style(rewritten_html)
response = HttpResponse(rewritten_html, content_type=content_type)
_set_transformed_response_headers(response, fullpath, statobj, encoding, config)
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type=content_type,
is_archive_replay=is_archive_replay,
config=config,
)
if escaped_count and escaped_count > tag_count * 2:
decoded = _apply_transformed_html_preview_style(decoded)
response = HttpResponse(decoded, content_type=content_type)
_set_transformed_response_headers(response, fullpath, statobj, encoding, config)
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type=content_type,
is_archive_replay=is_archive_replay,
config=config,
)
except (OSError, UnicodeDecodeError, ValueError):
pass
# setup response object
ranged_file = RangedFileReader(fullpath.open("rb"))
response = StreamingHttpResponse(
_stream_ranged_file_async(ranged_file) if _is_asgi_request(request) else ranged_file,
content_type=content_type,
)
response.headers["Last-Modified"] = http_date(statobj.st_mtime)
if etag:
response.headers["ETag"] = etag
response.headers["Cache-Control"] = f"{_cache_policy(config=config)}, max-age=31536000, immutable"
else:
response.headers["Cache-Control"] = f"{_cache_policy(config=config)}, max-age=60, stale-while-revalidate=300"
if is_text_like:
response.headers["Content-Disposition"] = f'inline; filename="{fullpath.name}"'
if content_type.startswith("image/"):
response.headers["Cache-Control"] = "public, max-age=604800, immutable"
# handle byte-range requests by serving chunk of file
if stat.S_ISREG(statobj.st_mode):
size = statobj.st_size
response["Content-Length"] = size
response["Accept-Ranges"] = "bytes"
response["X-Django-Ranges-Supported"] = "1"
# Respect the Range header.
if "HTTP_RANGE" in request.META:
try:
ranges = parse_range_header(request.META["HTTP_RANGE"], size)
except ValueError:
ranges = None
# only handle syntactically valid headers, that are simple (no
# multipart byteranges)
if ranges is not None and len(ranges) == 1:
start, stop = ranges[0]
if stop > size:
# requested range not satisfiable
return HttpResponse(status=416)
ranged_file.start = start
ranged_file.stop = stop
response["Content-Range"] = f"bytes {start}-{stop - 1}/{size}"
response["Content-Length"] = stop - start
response.status_code = 206
if encoding:
response.headers["Content-Encoding"] = encoding
return _apply_archive_replay_headers(
response,
fullpath=fullpath,
content_type=content_type,
is_archive_replay=is_archive_replay,
config=config,
)
def serve_static(request, path, **kwargs):
"""
Serve static files below a given point in the directory structure or
from locations inferred from the staticfiles finders.
To use, put a URL pattern such as::
from django.contrib.staticfiles import views
path('