""" documents.py - the document shelf behind /documents. Files live OUTSIDE the repo at /srv/scout-website-assets/docs, bind-mounted read-only at /docs, the same arrangement as the photos. PDFs never bloat a clone, and publishing one is a file drop plus a manifest line - no rebuild, no redeploy. manifest.json sits beside the files and maps a STABLE SLUG to a filename. That indirection is the point: next year's permission slip can replace this year's without breaking a link already printed on a flyer or sitting in somebody's inbox. Documents are served THROUGH the app, never from a static mount. Anything under /app/static is public forever. Serving through a route means that when member login exists, gating a document is one change in visible() and not a single URL moves. visibility: "public" anyone, and listed on /documents. "unlisted" served at its slug to anyone holding the link, but kept off the index and sent with X-Robots-Tag: noindex. For internal papers that need a durable link before member login exists. This is obscurity, not access control: treat an unlisted link as forwardable, because it is. "members" reserved for the login that does not exist yet. Until it does, these are hidden AND unservable, returning 404 rather than 403, because a 403 advertises a document we cannot actually gate yet. """ import datetime import json import os import re from pathlib import Path DOCS_DIR = Path(os.environ.get("DOCS_DIR", "/docs")) MANIFEST = DOCS_DIR / "manifest.json" # Slugs are the public contract. Keep them boring and permanent. SLUG_RE = re.compile(r"^[a-z0-9][a-z0-9-]{0,63}$") EMPTY = {"categories": [], "documents": []} # extension -> (badge label, media type). Anything unlisted downloads as a blob. KINDS = { ".pdf": ("PDF", "application/pdf"), ".doc": ("DOC", "application/msword"), ".docx": ("DOC", "application/vnd.openxmlformats-officedocument.wordprocessingml.document"), ".xls": ("XLS", "application/vnd.ms-excel"), ".xlsx": ("XLS", "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet"), ".ppt": ("PPT", "application/vnd.ms-powerpoint"), ".pptx": ("PPT", "application/vnd.openxmlformats-officedocument.presentationml.presentation"), ".csv": ("CSV", "text/csv"), ".txt": ("TXT", "text/plain; charset=utf-8"), ".md": ("TXT", "text/plain; charset=utf-8"), ".ics": ("ICS", "text/calendar; charset=utf-8"), ".jpg": ("IMG", "image/jpeg"), ".jpeg": ("IMG", "image/jpeg"), ".png": ("IMG", "image/png"), ".zip": ("ZIP", "application/zip"), } # Browsers should open these in a tab; the rest download. INLINE = {".pdf", ".txt", ".md", ".jpg", ".jpeg", ".png"} _cache = {"key": None, "value": EMPTY} def _stat_key(): try: st = MANIFEST.stat() except OSError: return None return (st.st_mtime_ns, st.st_size) def _clean(raw): """Drop anything malformed rather than serving a half-valid shelf.""" cats, seen_cat = [], set() for c in raw.get("categories") or []: cid = str(c.get("id", "")).strip() if not cid or cid in seen_cat: continue seen_cat.add(cid) cats.append({ "id": cid, "title": str(c.get("title") or cid).strip(), "blurb": str(c.get("blurb") or "").strip(), }) docs, seen_slug = [], set() for d in raw.get("documents") or []: slug = str(d.get("slug", "")).strip().lower() name = str(d.get("file", "")).strip() if not SLUG_RE.match(slug) or slug in seen_slug or not name: continue # Path traversal guard: the resolved file must sit inside DOCS_DIR. path = (DOCS_DIR / name).resolve() try: inside = path.is_relative_to(DOCS_DIR.resolve()) except AttributeError: # pragma: no cover - Python < 3.9 inside = str(path).startswith(str(DOCS_DIR.resolve()) + os.sep) if not inside: continue seen_slug.add(slug) vis = str(d.get("visibility") or "public").strip().lower() unit = str(d.get("unit") or "both").strip().lower() docs.append({ "slug": slug, "file": name, "path": path, "title": str(d.get("title") or slug).strip(), "description": str(d.get("description") or "").strip(), "category": str(d.get("category") or "").strip(), "unit": unit if unit in ("pack", "troop", "both") else "both", "visibility": vis if vis in ("public", "unlisted", "members") else "members", "updated": str(d.get("updated") or "").strip(), }) return {"categories": cats, "documents": docs} def manifest(): """Reload only when manifest.json actually changed on disk.""" key = _stat_key() if key != _cache["key"]: if key is None: _cache["value"] = EMPTY else: try: with MANIFEST.open(encoding="utf-8") as fh: _cache["value"] = _clean(json.load(fh)) except Exception as e: print("DOCUMENTS: manifest unreadable: %s" % e, flush=True) _cache["value"] = EMPTY _cache["key"] = key return _cache["value"] def visible(doc): """The single gate on SERVING. Member login plugs in here and nowhere else.""" return doc.get("visibility") in ("public", "unlisted") and doc["path"].is_file() def listed(doc): """The separate, weaker question of whether it appears on the index.""" return doc.get("visibility") == "public" and visible(doc) def noindex(doc): """Unlisted documents should not turn up in a search result.""" return doc.get("visibility") == "unlisted" def listing(): """Listed documents grouped into their categories, in manifest order.""" m = manifest() docs = [d for d in m["documents"] if listed(d)] known = {c["id"] for c in m["categories"]} groups = [] for cat in m["categories"]: rows = [d for d in docs if d["category"] == cat["id"]] if rows: groups.append((cat, rows)) loose = [d for d in docs if d["category"] not in known] if loose: groups.append(({"id": "", "title": "Everything else", "blurb": ""}, loose)) return groups def find(slug): slug = (slug or "").strip().lower() if not SLUG_RE.match(slug): return None for d in manifest()["documents"]: if d["slug"] == slug: return d if visible(d) else None return None def kind(doc): return KINDS.get(doc["path"].suffix.lower(), ("FILE", "application/octet-stream")) def disposition(doc): return "inline" if doc["path"].suffix.lower() in INLINE else "attachment" def size_text(doc): try: n = doc["path"].stat().st_size except OSError: return "" if n < 1024: return "%d B" % n if n < 1024 * 1024: return "%d KB" % round(n / 1024) return "%.1f MB" % (n / 1024 / 1024) def updated_text(doc): """Manifest date wins; otherwise the file's own mtime, which cannot lie.""" stamp = doc.get("updated") if stamp: try: return datetime.date.fromisoformat(stamp).strftime("%b %-d, %Y") except ValueError: return stamp try: ts = doc["path"].stat().st_mtime except OSError: return "" return datetime.date.fromtimestamp(ts).strftime("%b %-d, %Y")