Files
qtalker---/backend/tests/test_ws_reward_replay.py
Indiana c22b2f9c08 fix: reward guards now survive a reconnect
The three replay guards shipped earlier lived on the in-memory SeanceState,
which made them per-CONNECTION. I flagged that as an open residual at the
time: drop the socket and reconnect — or just open a second one — and the
client got a fresh empty guard and could be paid again for the same spirit.
summon_limiter bounded the rate of that, never the total.

`award_claims` is the durable form: one row per (seeker, presence,
milestone), with a UNIQUE constraint doing the actual enforcement. The claim
is a bare INSERT and losing the race raises IntegrityError, which is caught
and read as "already paid" — a check-then-insert would let two sockets both
read "unclaimed" and both pay. `crossing` is claimed by BOTH roads, so a
spirit crosses once whichever road arrives first.

Measured with the durable claim disabled: 5 reconnects paid 75 extra essence
on the ritual, 60 on a verdict, 140 on the passage, and two simultaneous
sockets paid 30 for one 15-essence ritual.

The four tests that were failing were the tests, not the guard. They compared
raw balances across reconnects, but re-opening a channel IS a summon, and
SUMMON_ESSENCE_TRICKLE is paid per summon by design (inventory.py:42, bounded
by summon_limiter rather than by any once-per-presence rule). The expected
trickle is now stated explicitly so the assertion speaks about the milestone
it is actually testing. Favor has no trickle, so it must not move at all —
asserted separately.

Anti-overshoot covered in both directions: a genuinely fresh presence still
pays in full across a reconnect, a corrected verdict still pays on a second
connection, and `test` stays freely repeatable since it never touches the
ledger.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-01 09:37:32 +00:00

507 lines
18 KiB
Python

"""WS integration tests for the OTHER two reward faucets in app.ws — the
ritual milestone and the judgment verdict.
Same class of bug, and the same reason it survived a 400-test suite:
`tests/test_judgment.py` covers app/judgment.py, which is pure — it decides
outcomes and returns them. It cannot see the handler that banks them.
`tests/test_ws_ritual_judgment.py` drives the handlers over a real socket but
only ever performs each rite ONCE, so it could not see what happens on the
second pass.
The faucets:
* `ritual_start` rewinds `ritual_completed` to False at any time, and the UI
offers exactly that button (RitualPanel's "attempt again", which honest
play genuinely needs after a failed roll). Every re-walk of
ritual_start + 4x ritual_step re-credited RITUAL_SUCCESS_ESSENCE and
re-rolled an item.
* `judgment` had no once-per-presence guard at all except for `cross_over`.
Replaying {"verdict": "trust"} against a benevolent spirit paid essence
AND favor AND an item roll on every single frame — favor pinning to +1.0,
which then biases every future mint.
Both are bounded in RATE by their limiters and unbounded in TOTAL, off a
single summon. These tests are written against the exploit path, not the
implementation: they send the frames a browser sends and then read the ledger
straight out of the database.
"""
import uuid
import pytest
from sqlalchemy import select
import app.judgment as judgment_module
import app.ws
from app.entities import fallback_profile
from app.inventory import (
CORRECT_JUDGMENT_ESSENCE,
RITUAL_SUCCESS_ESSENCE,
SUMMON_ESSENCE_TRICKLE,
)
from app.models.inventory_item import InventoryItem
from app.models.user import User
from app.rate_limit import RateLimiter
class FakeSpiritService:
async def mint_profile(
self, signature, channel, anomalies, language="en", entropy=None, sky=None
):
return fallback_profile(signature)
async def fragment(self, source, anomaly, language="en"):
return "listen"
async def wire_whisper(self, telemetry, language="en"):
return "the wire hums"
def chat_stream(self, entity, question, history, language="en"):
async def gen():
yield "here."
return gen()
def ambient_ready(self):
return False
async def _fake_synth(text, voice, profile, instability=0.0):
return b"RIFFfake wav bytes"
@pytest.fixture(autouse=True)
def _fake_spirits(monkeypatch):
monkeypatch.setattr(app.ws, "spirit_service", FakeSpiritService())
monkeypatch.setattr(app.ws, "synthesize_spirit_voice", _fake_synth)
# Module-level limiters are shared singletons accumulating real hit counts
# across the whole session (precedent: test_ws_ritual_judgment.py). The
# ritual/judgment ones matter most here: the exploits replayed below are
# deliberately high-volume, and the real 6/60s and 10/60s caps would mask
# the faucet behind a rate limit instead of letting us observe the total.
for name in (
"summon_limiter",
"summon_ip_limiter",
"question_limiter",
"question_ip_limiter",
"fragment_limiter",
"fragment_ip_limiter",
"ritual_limiter",
"ritual_ip_limiter",
"judgment_limiter",
"judgment_ip_limiter",
"passage_limiter",
"passage_ip_limiter",
):
monkeypatch.setattr(app.ws, name, RateLimiter(max_requests=10_000, window_seconds=60))
REPLAYS = 10
def _read_until(ws, msg_type, max_frames=80, **match):
for _ in range(max_frames):
frame = ws.receive_json()
if frame.get("type") != msg_type:
continue
if all(frame.get(key) == value for key, value in match.items()):
return frame
raise AssertionError(f"never saw frame of type {msg_type!r} matching {match!r}")
def _login(sync_client, username):
sync_client.post("/auth/register", json={"username": username, "password": "spookyspooky"})
sync_client.post("/auth/login", json={"username": username, "password": "spookyspooky"})
return sync_client.cookies.get("qm_session")
def _user_id(sync_client, token):
return uuid.UUID(
sync_client.get("/auth/me", headers={"cookie": f"qm_session={token}"}).json()["id"]
)
def _ws_connect(sync_client, token):
return sync_client.websocket_connect(
"/ws/session", headers={"cookie": f"qm_session={token}"}
)
def _summon(ws):
ws.send_json({"type": "summon"})
frame = _read_until(ws, "entity")
_read_until(ws, "utterance", kind="greeting")
return frame
def _settle(ws):
"""Force a full round trip so any post-frame DB write has landed.
Rewards are credited before/around the frame being queued, but the sender
task drains that queue concurrently with the handler's remaining awaits.
The connection's message loop is sequential, so a pong is a hard guarantee
that the previous handler returned — not a poll-and-hope.
"""
ws.send_json({"type": "ping"})
_read_until(ws, "pong")
def _run_ritual(ws, steps=4):
"""One full attempt: the rewind button, then every step."""
ws.send_json({"type": "ritual_start"})
for i in range(1, steps + 1):
ws.send_json({"type": "ritual_step", "step": i})
result = _read_until(ws, "ritual_complete")
_settle(ws)
return result
def _judge(ws, verdict):
ws.send_json({"type": "judgment", "verdict": verdict})
result = _read_until(ws, "judgment_result")
_settle(ws)
return result
async def _ledger(db_session, user_id):
db_session.expire_all()
user = await db_session.get(User, user_id)
items = (
await db_session.execute(
select(InventoryItem).where(InventoryItem.user_id == user_id)
)
).scalars().all()
return user.essence, user.favor, len(items)
def _always_wins(monkeypatch):
monkeypatch.setattr(
app.ws.judgment, "roll_ritual_success", lambda traits, rng=None: True
)
def _verdict_outcomes(monkeypatch, table):
"""Pin `judge_verdict` to a fixed verdict -> JudgmentOutcome table.
The real roll depends on the minted entity's hidden traits; these tests
are about the ledger, not about which spirit answered, so the outcome is
made deterministic exactly the way test_ws_ritual_judgment.py does it.
"""
monkeypatch.setattr(
app.ws.judgment, "judge_verdict", lambda verdict, traits, **kw: table[verdict]
)
def _force_new_presence(monkeypatch):
"""Pin the summon draw so the NEXT summon mints a genuinely new spirit.
`_summon` draws against RETURN_CHANCE to decide whether the presence
already on this channel answers again, and re-calling the same channel
usually returns the SAME entity row — measured on this suite, 8 of 11
consecutive re-summons came back with `is_new: false` and an identical id.
That matters now that the purse is keyed on the entity (award_claims),
because "summon again" and "the same spirit answers again" are then two
different things. These tests are about what a genuinely FRESH presence is
worth, so the draw is pinned rather than left to a coin flip; a draw of
1.0 is >= any possible `return_chance`, so the familiar presence never
answers. Every other draw stays real.
"""
real = app.ws.veil_float
def selective(entropy, context, *args, **kwargs):
if context == "answers":
return 1.0
return real(entropy, context, *args, **kwargs)
monkeypatch.setattr(app.ws, "veil_float", selective)
# --- the ritual faucet -----------------------------------------------------
@pytest.mark.asyncio
async def test_replaying_the_ritual_never_mints_essence_twice(
sync_client, db_session, monkeypatch
):
"""The exploit, executed literally.
Complete the rite, hit `ritual_start` to rewind, complete it again — ten
times over. The seeker must finish with exactly one milestone's worth of
essence, because there was only ever one spirit.
Before the fix this credited RITUAL_SUCCESS_ESSENCE on every pass: eleven
completions paid 165 instead of 15.
"""
_always_wins(monkeypatch)
token = _login(sync_client, "ritual-faucet")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
_summon(ws)
baseline, _, _ = await _ledger(db_session, user_id)
first = _run_ritual(ws)
after_one, _, items_after_one = await _ledger(db_session, user_id)
for _ in range(REPLAYS):
result = _run_ritual(ws)
assert result["success"] is True, "the replay did not even complete"
after_replays, _, items_after_replays = await _ledger(db_session, user_id)
assert first["success"] is True
earned = after_one - baseline
assert earned == RITUAL_SUCCESS_ESSENCE, "the rite paid nothing — the test proves nothing"
assert after_replays == after_one, (
f"replaying the ritual minted {after_replays - after_one} extra essence "
f"across {REPLAYS} rewinds; the ritual is an unbounded faucet again"
)
assert items_after_replays == items_after_one, (
f"replaying the ritual rolled {items_after_replays - items_after_one} extra "
"item drops off a single summon"
)
@pytest.mark.asyncio
async def test_replaying_the_ritual_still_reveals_the_traits(
sync_client, db_session, monkeypatch
):
"""The guard must bite the ledger only.
Re-attempting is a real, UI-offered action (it is how honest play recovers
from a failed roll), so a repeat must still roll and still reveal the true
traits on success — it just must not pay again. A guard that silently
swallowed the attempt would be a worse bug than the faucet.
"""
_always_wins(monkeypatch)
token = _login(sync_client, "ritual-rewalker")
_user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
_summon(ws)
first = _run_ritual(ws)
second = _run_ritual(ws)
assert second["success"] is True
assert second["revealed"] == first["revealed"] is not None
@pytest.mark.asyncio
async def test_a_genuinely_new_presence_reopens_the_ritual_purse(
sync_client, db_session, monkeypatch
):
"""The guard must not overshoot: each spirit is worth its own rite.
A durable claim that survived a summon would silently make every spirit
after the first worthless.
"""
_always_wins(monkeypatch)
# The second summon must actually bring a DIFFERENT spirit — see
# `_force_new_presence`. Without this the test asserts a coin flip.
_force_new_presence(monkeypatch)
token = _login(sync_client, "ritual-second-spirit")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
first_entity = _summon(ws)
baseline, _, _ = await _ledger(db_session, user_id)
_run_ritual(ws)
after_first, _, _ = await _ledger(db_session, user_id)
second_entity = _summon(ws)
assert second_entity["entity"]["id"] != first_entity["entity"]["id"], (
"the second summon returned the same spirit — this test is about "
"a genuinely new presence"
)
_run_ritual(ws)
after_second, _, _ = await _ledger(db_session, user_id)
first_rite = after_first - baseline
# The second summon also pays its own trickle, so compare the rite only.
second_rite = after_second - after_first - SUMMON_ESSENCE_TRICKLE
assert first_rite == RITUAL_SUCCESS_ESSENCE
assert second_rite == first_rite, (
"a fresh presence did not re-open the purse — every spirit after the "
"first has a worthless ritual"
)
# --- the judgment faucet ---------------------------------------------------
_CORRECT_TRUST = judgment_module.JudgmentOutcome(
True, judgment_module.FAVOR_CORRECT_TRUST, CORRECT_JUDGMENT_ESSENCE, False, "reward"
)
_WRONG_TRUST = judgment_module.JudgmentOutcome(
False, judgment_module.FAVOR_WRONG_TRUST, 0, False, "escalation"
)
_CORRECT_BANISH = judgment_module.JudgmentOutcome(
True, judgment_module.FAVOR_CORRECT_BANISH, CORRECT_JUDGMENT_ESSENCE, False, "reward"
)
@pytest.mark.asyncio
async def test_replaying_a_correct_judgment_never_pays_twice(
sync_client, db_session, monkeypatch
):
"""One spirit, one verdict, one payment.
Before the fix, eleven `trust` frames against one benevolent spirit paid
11 x CORRECT_JUDGMENT_ESSENCE (132 instead of 12) and 11 x +0.05 favor
(0.55, clamped toward the +1.0 ceiling instead of 0.05), plus eleven item
rolls — all off a single summon.
"""
_verdict_outcomes(monkeypatch, {"trust": _CORRECT_TRUST})
token = _login(sync_client, "judgment-faucet")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
_summon(ws)
baseline_essence, baseline_favor, _ = await _ledger(db_session, user_id)
_judge(ws, "trust")
essence_one, favor_one, items_one = await _ledger(db_session, user_id)
for _ in range(REPLAYS):
_judge(ws, "trust")
essence_end, favor_end, items_end = await _ledger(db_session, user_id)
assert essence_one - baseline_essence == CORRECT_JUDGMENT_ESSENCE
assert favor_one - baseline_favor == pytest.approx(judgment_module.FAVOR_CORRECT_TRUST)
assert essence_end == essence_one, (
f"replaying one verdict minted {essence_end - essence_one} extra essence "
f"across {REPLAYS} frames; judgment is an unbounded faucet again"
)
assert favor_end == pytest.approx(favor_one), (
f"replaying one verdict moved favor by a further {favor_end - favor_one}; "
"a scripted client can pin favor at the +1.0 ceiling"
)
assert items_end == items_one, (
f"replaying one verdict rolled {items_end - items_one} extra item drops"
)
@pytest.mark.asyncio
async def test_a_replayed_judgment_reports_zero_deltas(
sync_client, db_session, monkeypatch
):
"""The client sums `essence_delta`/`favor_delta` into its displayed
totals (frontend/src/state/seance.tsx's `judgment_result` case). A frame
that advertises the verdict's face value while the ledger credits nothing
shows the seeker currency their account never received — the most
corrosive kind of bug in a game about trust.
"""
_verdict_outcomes(monkeypatch, {"trust": _CORRECT_TRUST})
token = _login(sync_client, "judgment-honest-frame")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
_summon(ws)
baseline_essence, baseline_favor, _ = await _ledger(db_session, user_id)
frames = [_judge(ws, "trust") for _ in range(3)]
essence_end, favor_end, _ = await _ledger(db_session, user_id)
assert sum(f["essence_delta"] for f in frames) == essence_end - baseline_essence
assert sum(f["favor_delta"] for f in frames) == pytest.approx(favor_end - baseline_favor)
for frame in frames[1:]:
assert frame["essence_delta"] == 0 and frame["favor_delta"] == 0, (
"a replayed verdict advertised essence/favor it did not pay"
)
@pytest.mark.asyncio
async def test_an_honest_correction_still_resolves(sync_client, db_session, monkeypatch):
"""The guard must not overshoot: a seeker who calls it wrong and then
calls it right is doing something real, not replaying. A blanket
"judged once" latch would make that second, different verdict a no-op.
"""
_verdict_outcomes(monkeypatch, {"trust": _WRONG_TRUST, "banish": _CORRECT_BANISH})
token = _login(sync_client, "judgment-corrector")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
_summon(ws)
baseline_essence, baseline_favor, _ = await _ledger(db_session, user_id)
wrong = _judge(ws, "trust")
right = _judge(ws, "banish")
essence_end, favor_end, _ = await _ledger(db_session, user_id)
assert wrong["consequence"] == "escalation"
assert right["consequence"] == "reward"
assert essence_end - baseline_essence == CORRECT_JUDGMENT_ESSENCE, (
"the corrected verdict paid nothing — the replay guard is too broad"
)
assert favor_end - baseline_favor == pytest.approx(
judgment_module.FAVOR_WRONG_TRUST + judgment_module.FAVOR_CORRECT_BANISH
)
@pytest.mark.asyncio
async def test_a_genuinely_new_presence_reopens_the_judgment_purse(
sync_client, db_session, monkeypatch
):
_verdict_outcomes(monkeypatch, {"trust": _CORRECT_TRUST})
# The second summon must actually bring a DIFFERENT spirit — see
# `_force_new_presence`. Without this the test asserts a coin flip.
_force_new_presence(monkeypatch)
token = _login(sync_client, "judgment-second-spirit")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
first_entity = _summon(ws)
_judge(ws, "trust")
after_first, _, _ = await _ledger(db_session, user_id)
second_entity = _summon(ws)
assert second_entity["entity"]["id"] != first_entity["entity"]["id"], (
"the second summon returned the same spirit — this test is about "
"a genuinely new presence"
)
_judge(ws, "trust")
after_second, _, _ = await _ledger(db_session, user_id)
second_verdict = after_second - after_first - SUMMON_ESSENCE_TRICKLE
assert second_verdict == CORRECT_JUDGMENT_ESSENCE, (
"a fresh presence did not re-open the purse — every spirit after the "
"first is unjudgeable for reward"
)
@pytest.mark.asyncio
async def test_the_test_verdict_stays_freely_repeatable(sync_client, db_session, monkeypatch):
"""`test` is a diagnostic pulse that never touches the ledger, so it is
deliberately exempt from the once-per-presence guard — the panel lets a
seeker pulse a spirit as often as they like.
"""
token = _login(sync_client, "judgment-tester")
user_id = _user_id(sync_client, token)
with _ws_connect(sync_client, token) as ws:
_read_until(ws, "session")
_summon(ws)
baseline, baseline_favor, _ = await _ledger(db_session, user_id)
results = [_judge(ws, "test") for _ in range(3)]
essence_end, favor_end, _ = await _ledger(db_session, user_id)
assert all(r["consequence"] == "neutral" for r in results)
assert essence_end == baseline
assert favor_end == baseline_favor