Add a health check and the deployment runbook

/healthz checks the database and does a cache ROUND TRIP, not a ping. Both matter:
a node that cannot reach Postgres serves nothing, and a cache that accepts writes
and returns nothing would have waffle read every feature flag as unset -- so
"healthy" has to mean more than "the process is listening", or the load balancer
will keep feeding traffic to a node that only looks alive.

No auth and no tenant on it: the proxy, and later a load balancer, must reach it on
any host.

DEPLOYMENT.md is the runbook, and leads with the five things that make this app not
a generic Django deploy: the wildcard cert forces DNS-01 (Let's Encrypt will not
issue a wildcard over HTTP-01); Redis is required on one server, not two, because
of the per-process flag cache; SECURE_PROXY_SSL_HEADER plus Caddy's
X-Forwarded-Proto or WebAuthn and the SSL redirect both break; uploads must reach
object storage BEFORE the second app server, not during; and invoices need native
pango.

Also documents why the archive job ships with --commit off, why migrations are run
explicitly rather than from the entrypoint, and how to test the restore before the
day you need it.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
2026-07-14 09:45:21 +02:00
parent 35d1ec45a7
commit c0a44093d9
4 changed files with 247 additions and 1 deletions

44
rosterchief/health.py Normal file
View File

@@ -0,0 +1,44 @@
"""Liveness for the proxy today, for a load balancer later.
Checks the two dependencies whose absence makes the app lie rather than fail: without the
database it cannot serve anything, and without a shared cache the feature flags drift apart
between workers. A health check that only proves the process is listening would call that
healthy.
"""
from django.core.cache import cache
from django.db import connection
from django.http import JsonResponse
from django.views.decorators.cache import never_cache
PROBE_KEY = "healthz"
@never_cache
def healthz(request):
checks = {"database": _database(), "cache": _cache()}
healthy = all(checks.values())
return JsonResponse({"status": "ok" if healthy else "degraded", "checks": checks}, status=200 if healthy else 503)
def _database() -> bool:
try:
with connection.cursor() as cursor:
cursor.execute("SELECT 1")
cursor.fetchone()
except Exception:
return False
return True
def _cache() -> bool:
"""A round trip, not a ping: a cache that accepts writes and returns nothing is worse
than one that is plainly down, because waffle would read every flag as unset."""
try:
cache.set(PROBE_KEY, "ok", 10)
return cache.get(PROBE_KEY) == "ok"
except Exception:
return False

View File

@@ -1,7 +1,9 @@
import importlib
from unittest import mock
from django.db.utils import OperationalError
from django.test import SimpleTestCase, override_settings
from django.urls import Resolver404, clear_url_caches, resolve
from django.urls import Resolver404, clear_url_caches, resolve, reverse
from . import urls
@@ -28,3 +30,51 @@ class BrowserReloadUrlTests(SimpleTestCase):
with self.assertRaises(Resolver404):
resolve("/__reload__/events/")
class HealthCheckTests(SimpleTestCase):
databases = {"default"}
def test_it_reports_ok_when_the_database_and_cache_answer(self):
response = self.client.get(reverse("healthz"))
self.assertEqual(response.status_code, 200)
self.assertEqual(response.json(), {"status": "ok", "checks": {"database": True, "cache": True}})
def test_a_dead_database_is_a_503(self):
# 200 while the database is unreachable is worse than no health check at all: the load
# balancer would keep sending traffic to a node that cannot serve a single page.
with mock.patch("rosterchief.health.connection") as db:
db.cursor.side_effect = OperationalError("connection refused")
response = self.client.get(reverse("healthz"))
self.assertEqual(response.status_code, 503)
self.assertFalse(response.json()["checks"]["database"])
def test_a_cache_that_swallows_writes_is_unhealthy(self):
# Not a ping: a cache that accepts writes and returns nothing would have waffle read
# every feature flag as unset.
with mock.patch("rosterchief.health.cache") as broken:
broken.get.return_value = None
response = self.client.get(reverse("healthz"))
self.assertEqual(response.status_code, 503)
self.assertFalse(response.json()["checks"]["cache"])
def test_an_unreachable_cache_is_a_503(self):
# Distinct from the cache that answers wrongly above: here Redis is simply down, and
# the check must fail rather than raise its way to a 500.
with mock.patch("rosterchief.health.cache") as down:
down.set.side_effect = ConnectionError("redis is down")
response = self.client.get(reverse("healthz"))
self.assertEqual(response.status_code, 503)
self.assertFalse(response.json()["checks"]["cache"])
def test_it_is_never_cached(self):
response = self.client.get(reverse("healthz"))
self.assertIn("no-cache", response["Cache-Control"])

View File

@@ -15,7 +15,11 @@ from django.views.generic import RedirectView
from club.views import root
from .health import healthz
urlpatterns = [
# No auth and no tenant: the proxy and the load balancer must reach it on any host.
path("healthz", healthz, name="healthz"),
path("admin/login/", RedirectView.as_view(pattern_name="account_login", query_string=True), name="admin_login_redirect"),
path("admin/", admin.site.urls),
path("accounts/", include("allauth.urls")),