Keep a broken Redis from 500ing the whole API

Redis backs the DRF throttles, and the stock RedisCache raises inside the
throttle check when Redis is unreachable or refusing writes (a failed RDB
snapshot disables writes by default) — turning a cache problem into a
blanket 500, which is exactly what took prod down. ResilientRedisCache
treats backend failures as cache misses, logs the first one per worker, and
lets rate limits degrade until Redis is back.
This commit is contained in:
2026-09-20 21:08:38 -05:00
parent 70d7a4f606
commit bd2417aff8
3 changed files with 145 additions and 1 deletions
+36
View File
@@ -0,0 +1,36 @@
"""A broken Redis must degrade the cache, not 500 the API.
A Redis that cannot persist (the default ``stop-writes-on-bgsave-error``)
or is simply unreachable used to raise inside the DRF throttle check on
every request; ``ResilientRedisCache`` treats that as a cache miss.
"""
from unittest import mock
from django.core.cache import cache
from django.core.cache.backends.redis import RedisCacheClient
from django.test import SimpleTestCase
def failing(method: str):
return mock.patch.object(
RedisCacheClient,
method,
side_effect=RuntimeError("redis is down"),
)
class ResilientCacheTests(SimpleTestCase):
def test_get_returns_the_default_when_redis_fails(self):
with failing("get"):
self.assertIsNone(cache.get("j621-cache-test"))
self.assertEqual(cache.get("j621-cache-test", "fallback"), "fallback")
def test_writes_report_failure_without_raising(self):
with failing("set"):
self.assertIsNone(cache.set("j621-cache-test", "value"))
def test_bulk_and_delete_operations_degrade(self):
with failing("get_many"), failing("delete"):
self.assertEqual(cache.get_many(["a", "b"]), {})
self.assertFalse(cache.delete("a"))