Coverage for server / rate_limit.py: 87%
31 statements
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
« prev ^ index » next coverage.py v7.13.4, created at 2026-10-04 09:33 +0000
1"""Shared slowapi rate limiter.
3Defined here (not in server/app.py) so route modules can apply
4``@limiter.limit(...)`` without a circular import — app.py imports the routers,
5so the routers cannot import the limiter back from app.py.
7Storage (SEC-8)
8---------------
9slowapi defaults to IN-PROCESS storage. Behind a load balancer that is not a
10rate limit at all: each Gunicorn worker and each container keeps its own
11counters, so an attacker's effective ceiling is the configured limit multiplied
12by the number of running processes, and it silently rises every time the service
13is scaled out.
15When ``REDIS_URL`` is set the counters live in Redis instead, so one limit is
16enforced across every worker and instance. When it is unset (local dev, tests)
17the in-memory default is used: limits stay per-process, which is fine for a
18single process and is what keeps test isolation intact.
20``in_memory_fallback_enabled=True`` is deliberate. If Redis becomes unreachable,
21slowapi falls back to in-memory counters rather than raising — so a cache outage
22degrades rate limiting to today's per-process behaviour instead of returning 500
23from every limited route. Note the trade-off: it also means a Redis outage
24weakens the limit rather than hardening it. That is the right way round here —
25rate limiting is a mitigation, and failing closed would turn a Redis blip into a
26platform-wide outage for students mid-assignment.
28Rate limiting is BYPASSED during tests so it never causes flaky failures:
29 - any pytest run (``pytest`` present in ``sys.modules`` at import time) — covers
30 the in-repo unit suite regardless of when ``TESTING`` is set,
31 - ``TESTING=true`` (the repo's existing test signal),
32 - ``RATE_LIMIT_ENABLED=false`` (explicit override for E2E / QA test runs).
33Enabled by default everywhere else (prod).
34"""
36import logging
37import os
38import sys
40from slowapi import Limiter, _rate_limit_exceeded_handler
41from slowapi.util import get_remote_address
44def _build_storage_uri() -> str | None:
45 """Return the Redis URI for shared rate-limit storage, or None for in-memory.
47 Read at import time, like ``enabled``. Returns None for an unset or blank
48 value so a stray ``REDIS_URL=`` does not become an invalid storage URI.
49 """
50 url = os.getenv("REDIS_URL", "").strip()
51 if url:
52 return url
54 # In-memory storage enforces the cap PER WORKER PROCESS, so the effective limit
55 # is silently multiplied by the worker count. Fine locally; on a deployed host it
56 # is a weakened brute-force control that nothing would otherwise report. Only warn
57 # when limiting is actually on -- under pytest/TESTING it is disabled and the line
58 # would be pure noise.
59 if _rate_limit_enabled():
60 logging.getLogger(__name__).warning(
61 "REDIS_URL is not set; rate-limit storage falls back to IN-MEMORY, so the "
62 "per-IP cap is enforced per worker process rather than globally."
63 )
64 return None
67def _rate_limit_enabled() -> bool:
68 # Bypass for any pytest-driven run (unit tests import the app in-process).
69 if "pytest" in sys.modules:
70 return False
71 if os.getenv("TESTING", "").strip().lower() in ("1", "true", "yes"):
72 return False
73 return os.getenv("RATE_LIMIT_ENABLED", "true").strip().lower() in (
74 "1",
75 "true",
76 "yes",
77 )
80# Per-IP limiter. enabled=False makes every @limiter.limit(...) a no-op.
81# storage_uri=None → in-memory (dev/test); Redis URL → shared across instances.
82limiter = Limiter(
83 key_func=get_remote_address,
84 enabled=_rate_limit_enabled(),
85 storage_uri=_build_storage_uri(),
86 in_memory_fallback_enabled=True,
87)
90# ---------------------------------------------------------------------------
91# Limits, sized for the networks our users are actually on
92# ---------------------------------------------------------------------------
93#
94# ONE IP IS NOT ONE USER HERE, AND MAY NOT EVEN BE ONE SCHOOL.
95#
96# Texas K-12 internet egress is architected at the DISTRICT level, not the
97# campus. Cisco's own Service Ready Architecture for Schools puts the Internet
98# gateway at the district office ("The district office network includes the
99# Internet gateway"), and real Texas districts run it that way — Midlothian ISD
100# backhauls its whole multi-campus estate to a single subscribed connection.
101# So one source IP can be:
102#
103# a large Texas high school Allen HS 6,798 students, Conroe HS 5,304
104# an entire district Frisco ISD ~65,800, Cy-Fair ~118,000,
105# Houston ISD ~168,000
106#
107# A login limit of 10/minute per IP was sized as though an IP were one person.
108# It is not a theoretical problem: over six weeks on QA — with NO real users,
109# only the automation suite — 8,706 of 51,026 login attempts were refused with
110# 429. That is 17%, and the suite needed a bespoke pacing helper
111# (teacher-student-automation/service/utils/login_rate_limit.py) to function at
112# all. Its docstring notes the account suite alone "issues roughly thirty logins
113# in about fourteen seconds" — which is exactly the shape of a class of thirty
114# signing in at the start of a period.
115#
116# WHY RAISING THIS IS SAFE. The real brute-force defence is per-ACCOUNT and it
117# lives in Auth0, which is where OWASP's Authentication Cheat Sheet and NIST
118# SP 800-63B both put it ("the counter of failed logins should be associated
119# with the account itself, rather than the source IP address"). Verified live on
120# the tenant:
121#
122# brute-force protection enabled, max_attempts=10,
123# mode=count_per_identifier_and_ip, shields=[block]
124# suspicious-IP throttling enabled, pre-login max_attempts=100
125#
126# Both dimensions are already covered upstream. What is left for this limiter is
127# a coarse volumetric backstop — catching a flood, not policing a classroom.
128#
129# Every value is overridable by environment variable so the ceiling can be
130# tuned per environment without a deploy, and so it can be raised the moment
131# real district traffic shows a shape we did not predict.
133_DEFAULT_LOGIN_LIMIT = "300/minute"
134_DEFAULT_REFRESH_LIMIT = "600/minute"
135_DEFAULT_PASSWORD_UPDATE_LIMIT = "60/minute"
138def _limit(env_var: str, default: str) -> str:
139 """A limit string from the environment, or the default if unset/blank."""
140 return os.getenv(env_var, "").strip() or default
143LOGIN_RATE_LIMIT = _limit("LOGIN_RATE_LIMIT", _DEFAULT_LOGIN_LIMIT)
144REFRESH_RATE_LIMIT = _limit("REFRESH_RATE_LIMIT", _DEFAULT_REFRESH_LIMIT)
145PASSWORD_UPDATE_RATE_LIMIT = _limit(
146 "PASSWORD_UPDATE_RATE_LIMIT", _DEFAULT_PASSWORD_UPDATE_LIMIT
147)
150def rate_limit_exceeded_handler(request, exc):
151 """slowapi's handler, plus a log line naming who hit what.
153 The default handler answers 429 silently, so the only trace a refusal left
154 was a status code in the uvicorn access log. Before the first real district
155 is onboarded we want the opposite: every refusal visible, with the source
156 and the route, so the ceilings above can be checked against real burst
157 shapes rather than the estimates that set them.
158 """
159 logger = logging.getLogger(__name__)
160 logger.warning(
161 "rate_limit_exceeded ip=%s method=%s path=%s limit=%s",
162 get_remote_address(request),
163 request.method,
164 request.url.path,
165 getattr(exc, "detail", "unknown"),
166 )
167 return _rate_limit_exceeded_handler(request, exc)