feat: add per-IP rate limiting to WS LLM triggers
Per-account limits alone don't stop one account script-hitting Ollama from many source IPs. Adds matching per-IP limiters alongside the existing per-account ones for fragment/question/ summon triggers, per spec's dual per-account-and-per-IP requirement. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_013PphXq1s43DNRj1uWKGXof
This commit is contained in:
@@ -7,6 +7,7 @@ import app.ws
|
||||
from app.entities import fallback_profile
|
||||
from app.models.contact_session import ContactSession
|
||||
from app.models.event import Event
|
||||
from app.rate_limit import RateLimiter
|
||||
|
||||
|
||||
class FakeSpiritService:
|
||||
@@ -175,3 +176,59 @@ async def test_same_signature_recontacts_same_entity(sync_client):
|
||||
assert second_frame["entity"]["name"] == first
|
||||
assert second_frame["is_new"] is False
|
||||
assert second_frame["entity"]["contact_count"] == 2
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_summon_rate_limited_per_account(sync_client, monkeypatch):
|
||||
# Swap in a tight, test-scoped limiter so this doesn't depend on (or
|
||||
# pollute) the shared module-level budget other tests draw from.
|
||||
monkeypatch.setattr(
|
||||
app.ws, "summon_limiter", RateLimiter(max_requests=2, window_seconds=60)
|
||||
)
|
||||
_login(sync_client, "account-limited")
|
||||
|
||||
with _ws_connect(sync_client, sync_client.cookies.get("qm_session")) as ws:
|
||||
_read_until(ws, "session")
|
||||
for _ in range(2):
|
||||
ws.send_json({"type": "summon"})
|
||||
_read_until(ws, "entity")
|
||||
_read_until(ws, "utterance", kind="greeting")
|
||||
|
||||
ws.send_json({"type": "summon"})
|
||||
rejection = _read_until(ws, "error")
|
||||
assert rejection["code"] == "rate_limited"
|
||||
assert rejection["message"] == (
|
||||
"The veil is crowded. The spirits need a moment before another summoning."
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_summon_rate_limited_per_ip_even_with_fresh_account(
|
||||
sync_client, monkeypatch
|
||||
):
|
||||
# Starve only the IP bucket; the per-account limiter stays at its
|
||||
# production default so each account below has plenty of its own budget
|
||||
# left. This proves the IP limiter alone can reject a request — the
|
||||
# gap the per-account-only limiters left open.
|
||||
monkeypatch.setattr(
|
||||
app.ws, "summon_ip_limiter", RateLimiter(max_requests=1, window_seconds=60)
|
||||
)
|
||||
|
||||
_login(sync_client, "ip-limited-a")
|
||||
with _ws_connect(sync_client, sync_client.cookies.get("qm_session")) as ws:
|
||||
_read_until(ws, "session")
|
||||
ws.send_json({"type": "summon"})
|
||||
_read_until(ws, "entity") # spends the single per-IP slot
|
||||
|
||||
# A different account — its own per-account budget is untouched — but
|
||||
# every connection in this test shares the same (simulated) source IP,
|
||||
# which is already spent.
|
||||
_login(sync_client, "ip-limited-b")
|
||||
with _ws_connect(sync_client, sync_client.cookies.get("qm_session")) as ws:
|
||||
_read_until(ws, "session")
|
||||
ws.send_json({"type": "summon"})
|
||||
rejection = _read_until(ws, "error")
|
||||
assert rejection["code"] == "rate_limited"
|
||||
assert rejection["message"] == (
|
||||
"The veil is crowded. The spirits need a moment before another summoning."
|
||||
)
|
||||
|
||||
Reference in New Issue
Block a user