Add a health check and the deployment runbook
/healthz checks the database and does a cache ROUND TRIP, not a ping. Both matter: a node that cannot reach Postgres serves nothing, and a cache that accepts writes and returns nothing would have waffle read every feature flag as unset -- so "healthy" has to mean more than "the process is listening", or the load balancer will keep feeding traffic to a node that only looks alive. No auth and no tenant on it: the proxy, and later a load balancer, must reach it on any host. DEPLOYMENT.md is the runbook, and leads with the five things that make this app not a generic Django deploy: the wildcard cert forces DNS-01 (Let's Encrypt will not issue a wildcard over HTTP-01); Redis is required on one server, not two, because of the per-process flag cache; SECURE_PROXY_SSL_HEADER plus Caddy's X-Forwarded-Proto or WebAuthn and the SSL redirect both break; uploads must reach object storage BEFORE the second app server, not during; and invoices need native pango. Also documents why the archive job ships with --commit off, why migrations are run explicitly rather than from the entrypoint, and how to test the restore before the day you need it. Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
@@ -1,7 +1,9 @@
|
||||
import importlib
|
||||
from unittest import mock
|
||||
|
||||
from django.db.utils import OperationalError
|
||||
from django.test import SimpleTestCase, override_settings
|
||||
from django.urls import Resolver404, clear_url_caches, resolve
|
||||
from django.urls import Resolver404, clear_url_caches, resolve, reverse
|
||||
|
||||
from . import urls
|
||||
|
||||
@@ -28,3 +30,51 @@ class BrowserReloadUrlTests(SimpleTestCase):
|
||||
|
||||
with self.assertRaises(Resolver404):
|
||||
resolve("/__reload__/events/")
|
||||
|
||||
|
||||
class HealthCheckTests(SimpleTestCase):
|
||||
databases = {"default"}
|
||||
|
||||
def test_it_reports_ok_when_the_database_and_cache_answer(self):
|
||||
response = self.client.get(reverse("healthz"))
|
||||
|
||||
self.assertEqual(response.status_code, 200)
|
||||
self.assertEqual(response.json(), {"status": "ok", "checks": {"database": True, "cache": True}})
|
||||
|
||||
def test_a_dead_database_is_a_503(self):
|
||||
# 200 while the database is unreachable is worse than no health check at all: the load
|
||||
# balancer would keep sending traffic to a node that cannot serve a single page.
|
||||
with mock.patch("rosterchief.health.connection") as db:
|
||||
db.cursor.side_effect = OperationalError("connection refused")
|
||||
|
||||
response = self.client.get(reverse("healthz"))
|
||||
|
||||
self.assertEqual(response.status_code, 503)
|
||||
self.assertFalse(response.json()["checks"]["database"])
|
||||
|
||||
def test_a_cache_that_swallows_writes_is_unhealthy(self):
|
||||
# Not a ping: a cache that accepts writes and returns nothing would have waffle read
|
||||
# every feature flag as unset.
|
||||
with mock.patch("rosterchief.health.cache") as broken:
|
||||
broken.get.return_value = None
|
||||
|
||||
response = self.client.get(reverse("healthz"))
|
||||
|
||||
self.assertEqual(response.status_code, 503)
|
||||
self.assertFalse(response.json()["checks"]["cache"])
|
||||
|
||||
def test_an_unreachable_cache_is_a_503(self):
|
||||
# Distinct from the cache that answers wrongly above: here Redis is simply down, and
|
||||
# the check must fail rather than raise its way to a 500.
|
||||
with mock.patch("rosterchief.health.cache") as down:
|
||||
down.set.side_effect = ConnectionError("redis is down")
|
||||
|
||||
response = self.client.get(reverse("healthz"))
|
||||
|
||||
self.assertEqual(response.status_code, 503)
|
||||
self.assertFalse(response.json()["checks"]["cache"])
|
||||
|
||||
def test_it_is_never_cached(self):
|
||||
response = self.client.get(reverse("healthz"))
|
||||
|
||||
self.assertIn("no-cache", response["Cache-Control"])
|
||||
|
||||
Reference in New Issue
Block a user