From f6d5d18f0d3a4dfc73352dc860ef3a1ea224b760 Mon Sep 17 00:00:00 2001 From: Bernard Siebens Date: Thu, 6 Aug 2026 21:58:09 +0200 Subject: [PATCH 1/4] Fix media volume permissions for non-root container user MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit /app/media didn't exist in the image, so the media_data volume had nothing to copy ownership from on first mount — Docker created the mount point owned by root, and the container runs as rosterchief. Uploads then failed with PermissionError. Create the directory before the chown so it carries the right ownership into the volume. --- Dockerfile | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/Dockerfile b/Dockerfile index 86771e5..0c6d40b 100644 --- a/Dockerfile +++ b/Dockerfile @@ -77,7 +77,13 @@ RUN DJANGO_SECRET_KEY=build-only-not-a-secret \ DJANGO_STATICFILES_BACKEND=whitenoise.storage.CompressedManifestStaticFilesStorage \ python manage.py collectstatic --noinput -RUN useradd --system --uid 1000 rosterchief && chown -R rosterchief /app +# mkdir before chown, and before the volume ever mounts: media_data has nothing to copy from +# at /app/media otherwise, so Docker creates the mount point itself, owned by root — and the +# app runs as rosterchief, not root. Existing image content (even an empty, correctly-owned +# dir) is what a named volume copies its initial ownership from on first use. +RUN useradd --system --uid 1000 rosterchief \ + && mkdir -p /app/media \ + && chown -R rosterchief /app USER rosterchief EXPOSE 8000 From 30be424985f03d63f9c6e7d6ccfb505793b0eaa3 Mon Sep 17 00:00:00 2001 From: Bernard Siebens Date: Thu, 6 Aug 2026 22:15:16 +0200 Subject: [PATCH 2/4] Serve /media/* in production without the static() DEBUG gate MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit django.conf.urls.static.static() hard-codes its own `if not settings.DEBUG: return []` internally, so the earlier AWS_STORAGE_BUCKET_NAME guard around the call never mattered — no route was ever added outside DEBUG, and every logo still 404d. Build the pattern directly against django.views.static.serve, which has no such gate. --- rosterchief/urls.py | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/rosterchief/urls.py b/rosterchief/urls.py index 3d88e5a..d4c2ec2 100644 --- a/rosterchief/urls.py +++ b/rosterchief/urls.py @@ -7,11 +7,13 @@ second factors. ``RequireMFAMiddleware`` then blocks any staff user who has not enrolled. """ +import re + from django.conf import settings -from django.conf.urls.static import static from django.contrib import admin -from django.urls import include, path +from django.urls import include, path, re_path from django.views.generic import RedirectView +from django.views.static import serve from api.urls import api from club.views import root @@ -45,4 +47,11 @@ if not settings.AWS_STORAGE_BUCKET_NAME: # /media/* itself — so without this route every uploaded club logo 404s in production too. # Once AWS_STORAGE_BUCKET_NAME is set, club.logo.url points straight at the bucket and this # route is simply never hit. - urlpatterns += static(settings.MEDIA_URL, document_root=settings.MEDIA_ROOT) + # + # django.conf.urls.static.static() looks like the right helper, but it hard-codes its own + # `if not settings.DEBUG: return []` — it is documented as dev-only and silently no-ops in + # production no matter what guards the call site. Build the pattern directly against the + # view it wraps instead, which has no such gate. + urlpatterns += [ + re_path(rf"^{re.escape(settings.MEDIA_URL.lstrip('/'))}(?P.*)$", serve, {"document_root": settings.MEDIA_ROOT}), + ] From 1be9959481bf8de1ba1d98eaea9a941d5b71505d Mon Sep 17 00:00:00 2001 From: Bernard Siebens Date: Thu, 6 Aug 2026 22:18:27 +0200 Subject: [PATCH 3/4] Serve club logos directly from Caddy instead of Django MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Every image request was round-tripping through a gunicorn worker for what is just a static file on disk. Caddy now serves /media/* straight off the shared media_data volume (mounted read-only) and only falls through to Django for anything else — Django's own /media/* route stays as a fallback for compose.behind-proxy.yaml and runserver, where there is no bundled Caddy container. --- DEPLOYMENT.md | 14 ++++++++------ compose.yaml | 3 +++ deploy/caddy/Caddyfile | 9 +++++++++ 3 files changed, 20 insertions(+), 6 deletions(-) diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index 4bdedd8..edba93b 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -35,12 +35,14 @@ Caddy terminates TLS, so without it Django believes every request is plain HTTP: **4. Uploads must move to object storage before the second app server.** Club logos go to `MEDIA_ROOT` on local disk by default. `compose.yaml` mounts a `media_data` -volume so that survives a rebuild, and `rosterchief/urls.py` serves `/media/*` itself whenever -`AWS_STORAGE_BUCKET_NAME` is unset — Caddy only reverse-proxies, it never serves media on its -own, so without that route every logo 404s even on one box. On two boxes local disk stops -working regardless: a logo uploaded to node A is still a 404 on node B, since nothing shares -the volume between them. Setting `AWS_STORAGE_BUCKET_NAME` switches the default storage to S3 -— do it *before* you scale, not during. +volume, shared read-write with `web` and read-only with `caddy`, so uploads both survive a +rebuild and get served by Caddy directly (`handle_path /media/*` in the Caddyfile) rather than +round-tripping through a gunicorn worker. `rosterchief/urls.py` still serves `/media/*` itself +as a fallback whenever `AWS_STORAGE_BUCKET_NAME` is unset — needed for `compose.behind-proxy.yaml` +(no bundled Caddy there) and for `runserver`. On two boxes local disk stops working regardless +of any of this: a logo uploaded to node A is still a 404 on node B, since nothing shares the +volume between them. Setting `AWS_STORAGE_BUCKET_NAME` switches the default storage to S3 — do +it *before* you scale, not during. **5. PDF invoices need native libraries.** WeasyPrint binds to pango/cairo. The image installs them; a bare-metal deploy would need diff --git a/compose.yaml b/compose.yaml index 25d2d3c..7969e53 100644 --- a/compose.yaml +++ b/compose.yaml @@ -25,6 +25,9 @@ services: - ./deploy/caddy/Caddyfile:/etc/caddy/Caddyfile:ro - caddy_data:/data - caddy_config:/config + # Read-only: Caddy serves club logos straight off disk instead of round-tripping every + # image request through a gunicorn worker. Same volume `web` writes uploads into. + - media_data:/srv/media:ro depends_on: - web diff --git a/deploy/caddy/Caddyfile b/deploy/caddy/Caddyfile index dd1c219..a4d4a53 100644 --- a/deploy/caddy/Caddyfile +++ b/deploy/caddy/Caddyfile @@ -10,6 +10,15 @@ encode zstd gzip + # Club logos, served straight off the shared volume — no gunicorn worker involved. Only + # matters while storage is local disk; once AWS_STORAGE_BUCKET_NAME is set, club.logo.url + # points at the bucket directly and this block simply never matches. A missing file 404s + # here exactly as django.views.static.serve would, so there is no need to fall through. + handle_path /media/* { + root * /srv/media + file_server + } + # X-Forwarded-Proto is what SECURE_PROXY_SSL_HEADER reads. Without it Django believes every # request is plain HTTP: request.is_secure() goes false, WebAuthn disagrees with the browser # about the origin, and the SSL redirect becomes a loop. From fe19a6f08afdcbd59e91a4549842aaf308e4914a Mon Sep 17 00:00:00 2001 From: Bernard Siebens Date: Thu, 6 Aug 2026 22:29:20 +0200 Subject: [PATCH 4/4] Tune gunicorn/Postgres/Redis for a memory-limited server - gunicorn: 3 workers -> 2 (this workload isn't CPU-bound per DEPLOYMENT.md's own sizing), add --preload so workers share immutable memory via copy-on-write instead of each independently importing Django, add --max-requests so a worker that renders a WeasyPrint invoice doesn't carry that memory forever. - Postgres: trim shared_buffers/max_connections from the image defaults (128MB/100), sized for a ~0.2GB dataset instead. - Redis: cap with --maxmemory as a ceiling, not a saving. --- DEPLOYMENT.md | 10 +++++++++- Dockerfile | 11 ++++++++++- compose.behind-proxy.yaml | 4 +++- compose.yaml | 11 +++++++++-- 4 files changed, 31 insertions(+), 5 deletions(-) diff --git a/DEPLOYMENT.md b/DEPLOYMENT.md index edba93b..e4f7eff 100644 --- a/DEPLOYMENT.md +++ b/DEPLOYMENT.md @@ -423,7 +423,8 @@ So do not size for the data. Size for the **processes**. ### What actually consumes the box -Measured, running this app under gunicorn with `DEBUG=False`: +Measured, running this app under gunicorn with `DEBUG=False`, before the tuning below — +`--workers 3`, no `--preload`, Postgres and Redis on their image defaults: | | memory | |---|---| @@ -434,6 +435,13 @@ Measured, running this app under gunicorn with `DEBUG=False`: | OS + Docker daemon | ~400 MB | | **steady state** | **~1.0–1.2 GB** | +Since then, `Dockerfile`/`compose.yaml` were tuned for smaller boxes: `--workers 2 --preload` +(one fewer duplicated Django process, and `--preload` shares immutable memory across workers +via copy-on-write instead of each worker importing Django independently), plus trimmed Postgres +`shared_buffers`/`max_connections` and a Redis `--maxmemory` cap. Expect the gunicorn and +Postgres rows to come in lower than above — not yet re-measured, so treat the table as the +shape of where memory goes rather than exact numbers on the current config. + 2 GB would run it. 4 GB is the recommendation for three reasons, all of which are the kind of thing that bites at the worst moment: diff --git a/Dockerfile b/Dockerfile index 0c6d40b..d300434 100644 --- a/Dockerfile +++ b/Dockerfile @@ -91,10 +91,19 @@ EXPOSE 8000 # Migrations are NOT run here. With more than one app container they would race, and a failed # migration inside a starting web process is a bad place to find out — deploy runs them once, # explicitly (see DEPLOYMENT.md). +# 2 workers, not 3: DEPLOYMENT.md's own sizing says this workload isn't CPU-bound, and each +# worker duplicates a full Django process — the single biggest lever on a memory-limited box. +# --preload imports the app once in the master and forks workers via copy-on-write instead of +# each re-importing Django independently (safe here: no app's ready() touches DB/Redis eagerly, +# checked club/features/news/events). --max-requests recycles a worker periodically so the one +# that happens to render a WeasyPrint invoice doesn't carry that +50-100MB forever. CMD ["gunicorn", "rosterchief.wsgi:application", \ "--bind", "0.0.0.0:8000", \ - "--workers", "3", \ + "--workers", "2", \ "--threads", "4", \ + "--preload", \ + "--max-requests", "500", \ + "--max-requests-jitter", "50", \ "--timeout", "60", \ "--access-logfile", "-", \ "--error-logfile", "-"] diff --git a/compose.behind-proxy.yaml b/compose.behind-proxy.yaml index 26fb2ad..2b2f348 100644 --- a/compose.behind-proxy.yaml +++ b/compose.behind-proxy.yaml @@ -43,6 +43,8 @@ services: POSTGRES_DB: ${POSTGRES_DB:-rosterchief} POSTGRES_USER: ${POSTGRES_USER:-rosterchief} POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?set a database password} + # See compose.yaml for why these are trimmed from the defaults. + command: ["postgres", "-c", "shared_buffers=64MB", "-c", "max_connections=20"] volumes: - pgdata:/var/lib/postgresql/data healthcheck: @@ -54,7 +56,7 @@ services: redis: image: redis:7-alpine restart: unless-stopped - command: ["redis-server", "--save", "", "--appendonly", "no"] + command: ["redis-server", "--save", "", "--appendonly", "no", "--maxmemory", "32mb", "--maxmemory-policy", "allkeys-lru"] volumes: pgdata: diff --git a/compose.yaml b/compose.yaml index 7969e53..a38d681 100644 --- a/compose.yaml +++ b/compose.yaml @@ -59,6 +59,11 @@ services: POSTGRES_DB: ${POSTGRES_DB:-rosterchief} POSTGRES_USER: ${POSTGRES_USER:-rosterchief} POSTGRES_PASSWORD: ${POSTGRES_PASSWORD:?set a database password} + # shared_buffers/max_connections default to 128MB / 100 — sized for a much bigger database + # than this app's (DEPLOYMENT.md: ~0.2GB after 5 years). 20 connections is comfortably above + # 2 gunicorn workers x 4 threads plus the odd `manage.py` one-off; trimmed both for the box, + # not for the data. + command: ["postgres", "-c", "shared_buffers=64MB", "-c", "max_connections=20"] volumes: - pgdata:/var/lib/postgresql/data healthcheck: @@ -70,9 +75,11 @@ services: redis: image: redis:7-alpine restart: unless-stopped - command: ["redis-server", "--save", "", "--appendonly", "no"] # Cache only, so nothing here needs to survive a restart. It is not optional though: it - # is what keeps every gunicorn worker agreeing about which feature flags are on. + # is what keeps every gunicorn worker agreeing about which feature flags are on. maxmemory + # is a ceiling, not a saving — this is already the smallest process in the stack — but on a + # memory-limited box it should evict cache entries under pressure, not grow unbounded. + command: ["redis-server", "--save", "", "--appendonly", "no", "--maxmemory", "32mb", "--maxmemory-policy", "allkeys-lru"] volumes: pgdata: