Coverage for server / rate_limit.py: 87%

31 statements  

« prev     ^ index     » next       coverage.py v7.13.4, created at 2026-10-04 09:33 +0000

1"""Shared slowapi rate limiter. 

2 

3Defined here (not in server/app.py) so route modules can apply 

4``@limiter.limit(...)`` without a circular import — app.py imports the routers, 

5so the routers cannot import the limiter back from app.py. 

6 

7Storage (SEC-8) 

8--------------- 

9slowapi defaults to IN-PROCESS storage. Behind a load balancer that is not a 

10rate limit at all: each Gunicorn worker and each container keeps its own 

11counters, so an attacker's effective ceiling is the configured limit multiplied 

12by the number of running processes, and it silently rises every time the service 

13is scaled out. 

14 

15When ``REDIS_URL`` is set the counters live in Redis instead, so one limit is 

16enforced across every worker and instance. When it is unset (local dev, tests) 

17the in-memory default is used: limits stay per-process, which is fine for a 

18single process and is what keeps test isolation intact. 

19 

20``in_memory_fallback_enabled=True`` is deliberate. If Redis becomes unreachable, 

21slowapi falls back to in-memory counters rather than raising — so a cache outage 

22degrades rate limiting to today's per-process behaviour instead of returning 500 

23from every limited route. Note the trade-off: it also means a Redis outage 

24weakens the limit rather than hardening it. That is the right way round here — 

25rate limiting is a mitigation, and failing closed would turn a Redis blip into a 

26platform-wide outage for students mid-assignment. 

27 

28Rate limiting is BYPASSED during tests so it never causes flaky failures: 

29 - any pytest run (``pytest`` present in ``sys.modules`` at import time) — covers 

30 the in-repo unit suite regardless of when ``TESTING`` is set, 

31 - ``TESTING=true`` (the repo's existing test signal), 

32 - ``RATE_LIMIT_ENABLED=false`` (explicit override for E2E / QA test runs). 

33Enabled by default everywhere else (prod). 

34""" 

35 

36import logging 

37import os 

38import sys 

39 

40from slowapi import Limiter, _rate_limit_exceeded_handler 

41from slowapi.util import get_remote_address 

42 

43 

44def _build_storage_uri() -> str | None: 

45 """Return the Redis URI for shared rate-limit storage, or None for in-memory. 

46 

47 Read at import time, like ``enabled``. Returns None for an unset or blank 

48 value so a stray ``REDIS_URL=`` does not become an invalid storage URI. 

49 """ 

50 url = os.getenv("REDIS_URL", "").strip() 

51 if url: 

52 return url 

53 

54 # In-memory storage enforces the cap PER WORKER PROCESS, so the effective limit 

55 # is silently multiplied by the worker count. Fine locally; on a deployed host it 

56 # is a weakened brute-force control that nothing would otherwise report. Only warn 

57 # when limiting is actually on -- under pytest/TESTING it is disabled and the line 

58 # would be pure noise. 

59 if _rate_limit_enabled(): 

60 logging.getLogger(__name__).warning( 

61 "REDIS_URL is not set; rate-limit storage falls back to IN-MEMORY, so the " 

62 "per-IP cap is enforced per worker process rather than globally." 

63 ) 

64 return None 

65 

66 

67def _rate_limit_enabled() -> bool: 

68 # Bypass for any pytest-driven run (unit tests import the app in-process). 

69 if "pytest" in sys.modules: 

70 return False 

71 if os.getenv("TESTING", "").strip().lower() in ("1", "true", "yes"): 

72 return False 

73 return os.getenv("RATE_LIMIT_ENABLED", "true").strip().lower() in ( 

74 "1", 

75 "true", 

76 "yes", 

77 ) 

78 

79 

80# Per-IP limiter. enabled=False makes every @limiter.limit(...) a no-op. 

81# storage_uri=None → in-memory (dev/test); Redis URL → shared across instances. 

82limiter = Limiter( 

83 key_func=get_remote_address, 

84 enabled=_rate_limit_enabled(), 

85 storage_uri=_build_storage_uri(), 

86 in_memory_fallback_enabled=True, 

87) 

88 

89 

90# --------------------------------------------------------------------------- 

91# Limits, sized for the networks our users are actually on 

92# --------------------------------------------------------------------------- 

93# 

94# ONE IP IS NOT ONE USER HERE, AND MAY NOT EVEN BE ONE SCHOOL. 

95# 

96# Texas K-12 internet egress is architected at the DISTRICT level, not the 

97# campus. Cisco's own Service Ready Architecture for Schools puts the Internet 

98# gateway at the district office ("The district office network includes the 

99# Internet gateway"), and real Texas districts run it that way — Midlothian ISD 

100# backhauls its whole multi-campus estate to a single subscribed connection. 

101# So one source IP can be: 

102# 

103# a large Texas high school Allen HS 6,798 students, Conroe HS 5,304 

104# an entire district Frisco ISD ~65,800, Cy-Fair ~118,000, 

105# Houston ISD ~168,000 

106# 

107# A login limit of 10/minute per IP was sized as though an IP were one person. 

108# It is not a theoretical problem: over six weeks on QA — with NO real users, 

109# only the automation suite — 8,706 of 51,026 login attempts were refused with 

110# 429. That is 17%, and the suite needed a bespoke pacing helper 

111# (teacher-student-automation/service/utils/login_rate_limit.py) to function at 

112# all. Its docstring notes the account suite alone "issues roughly thirty logins 

113# in about fourteen seconds" — which is exactly the shape of a class of thirty 

114# signing in at the start of a period. 

115# 

116# WHY RAISING THIS IS SAFE. The real brute-force defence is per-ACCOUNT and it 

117# lives in Auth0, which is where OWASP's Authentication Cheat Sheet and NIST 

118# SP 800-63B both put it ("the counter of failed logins should be associated 

119# with the account itself, rather than the source IP address"). Verified live on 

120# the tenant: 

121# 

122# brute-force protection enabled, max_attempts=10, 

123# mode=count_per_identifier_and_ip, shields=[block] 

124# suspicious-IP throttling enabled, pre-login max_attempts=100 

125# 

126# Both dimensions are already covered upstream. What is left for this limiter is 

127# a coarse volumetric backstop — catching a flood, not policing a classroom. 

128# 

129# Every value is overridable by environment variable so the ceiling can be 

130# tuned per environment without a deploy, and so it can be raised the moment 

131# real district traffic shows a shape we did not predict. 

132 

133_DEFAULT_LOGIN_LIMIT = "300/minute" 

134_DEFAULT_REFRESH_LIMIT = "600/minute" 

135_DEFAULT_PASSWORD_UPDATE_LIMIT = "60/minute" 

136 

137 

138def _limit(env_var: str, default: str) -> str: 

139 """A limit string from the environment, or the default if unset/blank.""" 

140 return os.getenv(env_var, "").strip() or default 

141 

142 

143LOGIN_RATE_LIMIT = _limit("LOGIN_RATE_LIMIT", _DEFAULT_LOGIN_LIMIT) 

144REFRESH_RATE_LIMIT = _limit("REFRESH_RATE_LIMIT", _DEFAULT_REFRESH_LIMIT) 

145PASSWORD_UPDATE_RATE_LIMIT = _limit( 

146 "PASSWORD_UPDATE_RATE_LIMIT", _DEFAULT_PASSWORD_UPDATE_LIMIT 

147) 

148 

149 

150def rate_limit_exceeded_handler(request, exc): 

151 """slowapi's handler, plus a log line naming who hit what. 

152 

153 The default handler answers 429 silently, so the only trace a refusal left 

154 was a status code in the uvicorn access log. Before the first real district 

155 is onboarded we want the opposite: every refusal visible, with the source 

156 and the route, so the ceilings above can be checked against real burst 

157 shapes rather than the estimates that set them. 

158 """ 

159 logger = logging.getLogger(__name__) 

160 logger.warning( 

161 "rate_limit_exceeded ip=%s method=%s path=%s limit=%s", 

162 get_remote_address(request), 

163 request.method, 

164 request.url.path, 

165 getattr(exc, "detail", "unknown"), 

166 ) 

167 return _rate_limit_exceeded_handler(request, exc)