diff --git a/backend/app/ws.py b/backend/app/ws.py index 90f49d9..98375f9 100644 --- a/backend/app/ws.py +++ b/backend/app/ws.py @@ -153,6 +153,31 @@ class SeanceState: ritual_steps: int = 0 ritual_completed: bool = False ritual_success: bool = False + # Whether the ritual milestone has ALREADY paid out for the current + # presence. Exactly the same faucet `passage_paid` closes, on the other + # rite: `ritual_start` rewinds `ritual_completed` to False at any time + # (and the UI offers precisely that button — RitualPanel's "attempt + # again", which honest play needs after a FAILED roll), so without this + # a client could walk ritual_start + 4x ritual_step for +15 essence and + # an item roll, restart, and repeat forever off a single summon. The + # limiter caps the cadence, not the total. So the milestone pays the + # FIRST time it succeeds for a given presence and never again; + # re-attempting still rolls, still reveals the traits on success, but + # mints no new essence and rolls no new item. Cleared only on a fresh + # summon, never by `ritual_start` — that is the whole point. + ritual_paid: bool = False + # Which verdicts have ALREADY been applied to the ledger for the current + # presence. Same class of hole: `judgment` had no once-per-presence + # guard at all except for `cross_over`, so replaying {"type": + # "judgment", "verdict": "trust"} against a benevolent spirit credited + # CORRECT_JUDGMENT_ESSENCE *and* +0.05 favor *and* rolled an item on + # every single frame — an unbounded faucet for both currencies (favor + # pins to +1.0, which then biases every future mint) that the 10/60s + # limiter only slowed down. Keyed by verdict rather than a single latch + # so an honest correction (a wrong `trust`, then the right `banish` on + # the same spirit) still resolves normally — what can never repeat is + # the SAME verdict on the SAME presence. Cleared only on a fresh summon. + judged_verdicts: set[str] = field(default_factory=set) # Workstream L (passage-doctrine spec): where the layered crossing rite # stands for the *current* entity. `passage_layer` is the next beat to # attempt, `passage_lied` remembers whether the layer just completed was @@ -583,6 +608,10 @@ async def _summon_locked(state: SeanceState) -> None: state.ritual_steps = 0 state.ritual_completed = False state.ritual_success = False + # Only a genuinely fresh summon re-opens the purse for the other two + # rites, exactly as `_reset_passage(new_entity=True)` does below. + state.ritual_paid = False + state.judged_verdicts = set() _reset_passage(state, new_entity=True) await state.send_queue.put( {"type": "entity", "entity": _public_entity(state.entity), "is_new": is_new} @@ -896,7 +925,11 @@ async def _handle_ritual_step(state: SeanceState, message: dict) -> None: await state.send_queue.put( {"type": "ritual_complete", "success": success, "revealed": revealed} ) - if success: + # The milestone pays once per presence. A re-attempt after a rewind + # (`ritual_start`) still rolls and still reveals on success — it just + # doesn't mint a second payout. See `SeanceState.ritual_paid`. + if success and not state.ritual_paid: + state.ritual_paid = True await _reward_ritual_success(state) @@ -938,13 +971,33 @@ async def _handle_judgment(state: SeanceState, message: dict) -> None: ritual_success=state.ritual_success, ) + # A verdict lands on a presence once. Replaying the same one — the UI + # re-arms the panel after any result — re-reports the same reading for + # free rather than paying it out again. See `judged_verdicts`; `test` + # never touches the ledger so it is deliberately exempt and stays + # freely repeatable. + already_judged = verdict != "test" and verdict in state.judged_verdicts + if verdict != "test": + state.judged_verdicts.add(verdict) + + # What will ACTUALLY reach the ledger. The frame below reports these, + # not `outcome`'s face values: the client sums `essence_delta`/ + # `favor_delta` straight into its displayed totals (see + # frontend/src/state/seance.tsx's `judgment_result` case), so echoing + # the nominal amounts on a replay would show a seeker currency their + # account never received. + favor_delta = 0.0 if already_judged else outcome.favor_delta + essence_delta = 0 if already_judged else outcome.essence_delta + consequence = outcome.consequence + item = None # Skip the DB round-trip entirely when there's nothing to persist (e.g. # `test` without a completed ritual, or a resisted cross_over) — the # contract's "no crash, just no effect" for those cases. - if outcome.favor_delta or outcome.essence_delta or outcome.consequence in ( - "reward", - "crossed_over", + if not already_judged and ( + outcome.favor_delta + or outcome.essence_delta + or outcome.consequence in ("reward", "crossed_over") ): async with session_maker() as db: # Locked — see the same comment in _reward_summon above; this @@ -987,10 +1040,10 @@ async def _handle_judgment(state: SeanceState, message: dict) -> None: { "type": "judgment_result", "correct": outcome.correct, - "favor_delta": outcome.favor_delta, - "essence_delta": outcome.essence_delta, + "favor_delta": favor_delta, + "essence_delta": essence_delta, "at_peace": outcome.at_peace, - "consequence": outcome.consequence, + "consequence": consequence, } ) if item is not None: diff --git a/backend/tests/test_ws_reward_replay.py b/backend/tests/test_ws_reward_replay.py new file mode 100644 index 0000000..c0699f0 --- /dev/null +++ b/backend/tests/test_ws_reward_replay.py @@ -0,0 +1,467 @@ +"""WS integration tests for the OTHER two reward faucets in app.ws — the +ritual milestone and the judgment verdict. + +Same class of bug, and the same reason it survived a 400-test suite: +`tests/test_judgment.py` covers app/judgment.py, which is pure — it decides +outcomes and returns them. It cannot see the handler that banks them. +`tests/test_ws_ritual_judgment.py` drives the handlers over a real socket but +only ever performs each rite ONCE, so it could not see what happens on the +second pass. + +The faucets: + + * `ritual_start` rewinds `ritual_completed` to False at any time, and the UI + offers exactly that button (RitualPanel's "attempt again", which honest + play genuinely needs after a failed roll). Every re-walk of + ritual_start + 4x ritual_step re-credited RITUAL_SUCCESS_ESSENCE and + re-rolled an item. + * `judgment` had no once-per-presence guard at all except for `cross_over`. + Replaying {"verdict": "trust"} against a benevolent spirit paid essence + AND favor AND an item roll on every single frame — favor pinning to +1.0, + which then biases every future mint. + +Both are bounded in RATE by their limiters and unbounded in TOTAL, off a +single summon. These tests are written against the exploit path, not the +implementation: they send the frames a browser sends and then read the ledger +straight out of the database. +""" + +import uuid + +import pytest +from sqlalchemy import select + +import app.judgment as judgment_module +import app.ws +from app.entities import fallback_profile +from app.inventory import ( + CORRECT_JUDGMENT_ESSENCE, + RITUAL_SUCCESS_ESSENCE, + SUMMON_ESSENCE_TRICKLE, +) +from app.models.inventory_item import InventoryItem +from app.models.user import User +from app.rate_limit import RateLimiter + + +class FakeSpiritService: + async def mint_profile( + self, signature, channel, anomalies, language="en", entropy=None, sky=None + ): + return fallback_profile(signature) + + async def fragment(self, source, anomaly, language="en"): + return "listen" + + async def wire_whisper(self, telemetry, language="en"): + return "the wire hums" + + def chat_stream(self, entity, question, history, language="en"): + async def gen(): + yield "here." + + return gen() + + def ambient_ready(self): + return False + + +async def _fake_synth(text, voice, profile, instability=0.0): + return b"RIFFfake wav bytes" + + +@pytest.fixture(autouse=True) +def _fake_spirits(monkeypatch): + monkeypatch.setattr(app.ws, "spirit_service", FakeSpiritService()) + monkeypatch.setattr(app.ws, "synthesize_spirit_voice", _fake_synth) + # Module-level limiters are shared singletons accumulating real hit counts + # across the whole session (precedent: test_ws_ritual_judgment.py). The + # ritual/judgment ones matter most here: the exploits replayed below are + # deliberately high-volume, and the real 6/60s and 10/60s caps would mask + # the faucet behind a rate limit instead of letting us observe the total. + for name in ( + "summon_limiter", + "summon_ip_limiter", + "question_limiter", + "question_ip_limiter", + "fragment_limiter", + "fragment_ip_limiter", + "ritual_limiter", + "ritual_ip_limiter", + "judgment_limiter", + "judgment_ip_limiter", + "passage_limiter", + "passage_ip_limiter", + ): + monkeypatch.setattr(app.ws, name, RateLimiter(max_requests=10_000, window_seconds=60)) + + +REPLAYS = 10 + + +def _read_until(ws, msg_type, max_frames=80, **match): + for _ in range(max_frames): + frame = ws.receive_json() + if frame.get("type") != msg_type: + continue + if all(frame.get(key) == value for key, value in match.items()): + return frame + raise AssertionError(f"never saw frame of type {msg_type!r} matching {match!r}") + + +def _login(sync_client, username): + sync_client.post("/auth/register", json={"username": username, "password": "spookyspooky"}) + sync_client.post("/auth/login", json={"username": username, "password": "spookyspooky"}) + return sync_client.cookies.get("qm_session") + + +def _user_id(sync_client, token): + return uuid.UUID( + sync_client.get("/auth/me", headers={"cookie": f"qm_session={token}"}).json()["id"] + ) + + +def _ws_connect(sync_client, token): + return sync_client.websocket_connect( + "/ws/session", headers={"cookie": f"qm_session={token}"} + ) + + +def _summon(ws): + ws.send_json({"type": "summon"}) + frame = _read_until(ws, "entity") + _read_until(ws, "utterance", kind="greeting") + return frame + + +def _settle(ws): + """Force a full round trip so any post-frame DB write has landed. + + Rewards are credited before/around the frame being queued, but the sender + task drains that queue concurrently with the handler's remaining awaits. + The connection's message loop is sequential, so a pong is a hard guarantee + that the previous handler returned — not a poll-and-hope. + """ + ws.send_json({"type": "ping"}) + _read_until(ws, "pong") + + +def _run_ritual(ws, steps=4): + """One full attempt: the rewind button, then every step.""" + ws.send_json({"type": "ritual_start"}) + for i in range(1, steps + 1): + ws.send_json({"type": "ritual_step", "step": i}) + result = _read_until(ws, "ritual_complete") + _settle(ws) + return result + + +def _judge(ws, verdict): + ws.send_json({"type": "judgment", "verdict": verdict}) + result = _read_until(ws, "judgment_result") + _settle(ws) + return result + + +async def _ledger(db_session, user_id): + db_session.expire_all() + user = await db_session.get(User, user_id) + items = ( + await db_session.execute( + select(InventoryItem).where(InventoryItem.user_id == user_id) + ) + ).scalars().all() + return user.essence, user.favor, len(items) + + +def _always_wins(monkeypatch): + monkeypatch.setattr( + app.ws.judgment, "roll_ritual_success", lambda traits, rng=None: True + ) + + +def _verdict_outcomes(monkeypatch, table): + """Pin `judge_verdict` to a fixed verdict -> JudgmentOutcome table. + + The real roll depends on the minted entity's hidden traits; these tests + are about the ledger, not about which spirit answered, so the outcome is + made deterministic exactly the way test_ws_ritual_judgment.py does it. + """ + monkeypatch.setattr( + app.ws.judgment, "judge_verdict", lambda verdict, traits, **kw: table[verdict] + ) + + +# --- the ritual faucet ----------------------------------------------------- + + +@pytest.mark.asyncio +async def test_replaying_the_ritual_never_mints_essence_twice( + sync_client, db_session, monkeypatch +): + """The exploit, executed literally. + + Complete the rite, hit `ritual_start` to rewind, complete it again — ten + times over. The seeker must finish with exactly one milestone's worth of + essence, because there was only ever one spirit. + + Before the fix this credited RITUAL_SUCCESS_ESSENCE on every pass: eleven + completions paid 165 instead of 15. + """ + _always_wins(monkeypatch) + + token = _login(sync_client, "ritual-faucet") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + baseline, _, _ = await _ledger(db_session, user_id) + + first = _run_ritual(ws) + after_one, _, items_after_one = await _ledger(db_session, user_id) + + for _ in range(REPLAYS): + result = _run_ritual(ws) + assert result["success"] is True, "the replay did not even complete" + + after_replays, _, items_after_replays = await _ledger(db_session, user_id) + + assert first["success"] is True + earned = after_one - baseline + assert earned == RITUAL_SUCCESS_ESSENCE, "the rite paid nothing — the test proves nothing" + assert after_replays == after_one, ( + f"replaying the ritual minted {after_replays - after_one} extra essence " + f"across {REPLAYS} rewinds; the ritual is an unbounded faucet again" + ) + assert items_after_replays == items_after_one, ( + f"replaying the ritual rolled {items_after_replays - items_after_one} extra " + "item drops off a single summon" + ) + + +@pytest.mark.asyncio +async def test_replaying_the_ritual_still_reveals_the_traits( + sync_client, db_session, monkeypatch +): + """The guard must bite the ledger only. + + Re-attempting is a real, UI-offered action (it is how honest play recovers + from a failed roll), so a repeat must still roll and still reveal the true + traits on success — it just must not pay again. A guard that silently + swallowed the attempt would be a worse bug than the faucet. + """ + _always_wins(monkeypatch) + + token = _login(sync_client, "ritual-rewalker") + _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + first = _run_ritual(ws) + second = _run_ritual(ws) + + assert second["success"] is True + assert second["revealed"] == first["revealed"] is not None + + +@pytest.mark.asyncio +async def test_a_genuinely_new_presence_reopens_the_ritual_purse( + sync_client, db_session, monkeypatch +): + """The guard must not overshoot: each spirit is worth its own rite. + + A `ritual_paid` latch that survived a summon would silently make every + spirit after the first worthless. + """ + _always_wins(monkeypatch) + + token = _login(sync_client, "ritual-second-spirit") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + baseline, _, _ = await _ledger(db_session, user_id) + _run_ritual(ws) + after_first, _, _ = await _ledger(db_session, user_id) + + _summon(ws) + _run_ritual(ws) + after_second, _, _ = await _ledger(db_session, user_id) + + first_rite = after_first - baseline + # The second summon also pays its own trickle, so compare the rite only. + second_rite = after_second - after_first - SUMMON_ESSENCE_TRICKLE + assert first_rite == RITUAL_SUCCESS_ESSENCE + assert second_rite == first_rite, ( + "a fresh presence did not re-open the purse — every spirit after the " + "first has a worthless ritual" + ) + + +# --- the judgment faucet --------------------------------------------------- + + +_CORRECT_TRUST = judgment_module.JudgmentOutcome( + True, judgment_module.FAVOR_CORRECT_TRUST, CORRECT_JUDGMENT_ESSENCE, False, "reward" +) +_WRONG_TRUST = judgment_module.JudgmentOutcome( + False, judgment_module.FAVOR_WRONG_TRUST, 0, False, "escalation" +) +_CORRECT_BANISH = judgment_module.JudgmentOutcome( + True, judgment_module.FAVOR_CORRECT_BANISH, CORRECT_JUDGMENT_ESSENCE, False, "reward" +) + + +@pytest.mark.asyncio +async def test_replaying_a_correct_judgment_never_pays_twice( + sync_client, db_session, monkeypatch +): + """One spirit, one verdict, one payment. + + Before the fix, eleven `trust` frames against one benevolent spirit paid + 11 x CORRECT_JUDGMENT_ESSENCE (132 instead of 12) and 11 x +0.05 favor + (0.55, clamped toward the +1.0 ceiling instead of 0.05), plus eleven item + rolls — all off a single summon. + """ + _verdict_outcomes(monkeypatch, {"trust": _CORRECT_TRUST}) + + token = _login(sync_client, "judgment-faucet") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + baseline_essence, baseline_favor, _ = await _ledger(db_session, user_id) + + _judge(ws, "trust") + essence_one, favor_one, items_one = await _ledger(db_session, user_id) + + for _ in range(REPLAYS): + _judge(ws, "trust") + + essence_end, favor_end, items_end = await _ledger(db_session, user_id) + + assert essence_one - baseline_essence == CORRECT_JUDGMENT_ESSENCE + assert favor_one - baseline_favor == pytest.approx(judgment_module.FAVOR_CORRECT_TRUST) + assert essence_end == essence_one, ( + f"replaying one verdict minted {essence_end - essence_one} extra essence " + f"across {REPLAYS} frames; judgment is an unbounded faucet again" + ) + assert favor_end == pytest.approx(favor_one), ( + f"replaying one verdict moved favor by a further {favor_end - favor_one}; " + "a scripted client can pin favor at the +1.0 ceiling" + ) + assert items_end == items_one, ( + f"replaying one verdict rolled {items_end - items_one} extra item drops" + ) + + +@pytest.mark.asyncio +async def test_a_replayed_judgment_reports_zero_deltas( + sync_client, db_session, monkeypatch +): + """The client sums `essence_delta`/`favor_delta` into its displayed + totals (frontend/src/state/seance.tsx's `judgment_result` case). A frame + that advertises the verdict's face value while the ledger credits nothing + shows the seeker currency their account never received — the most + corrosive kind of bug in a game about trust. + """ + _verdict_outcomes(monkeypatch, {"trust": _CORRECT_TRUST}) + + token = _login(sync_client, "judgment-honest-frame") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + baseline_essence, baseline_favor, _ = await _ledger(db_session, user_id) + + frames = [_judge(ws, "trust") for _ in range(3)] + essence_end, favor_end, _ = await _ledger(db_session, user_id) + + assert sum(f["essence_delta"] for f in frames) == essence_end - baseline_essence + assert sum(f["favor_delta"] for f in frames) == pytest.approx(favor_end - baseline_favor) + for frame in frames[1:]: + assert frame["essence_delta"] == 0 and frame["favor_delta"] == 0, ( + "a replayed verdict advertised essence/favor it did not pay" + ) + + +@pytest.mark.asyncio +async def test_an_honest_correction_still_resolves(sync_client, db_session, monkeypatch): + """The guard must not overshoot: a seeker who calls it wrong and then + calls it right is doing something real, not replaying. A blanket + "judged once" latch would make that second, different verdict a no-op. + """ + _verdict_outcomes(monkeypatch, {"trust": _WRONG_TRUST, "banish": _CORRECT_BANISH}) + + token = _login(sync_client, "judgment-corrector") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + baseline_essence, baseline_favor, _ = await _ledger(db_session, user_id) + + wrong = _judge(ws, "trust") + right = _judge(ws, "banish") + essence_end, favor_end, _ = await _ledger(db_session, user_id) + + assert wrong["consequence"] == "escalation" + assert right["consequence"] == "reward" + assert essence_end - baseline_essence == CORRECT_JUDGMENT_ESSENCE, ( + "the corrected verdict paid nothing — the replay guard is too broad" + ) + assert favor_end - baseline_favor == pytest.approx( + judgment_module.FAVOR_WRONG_TRUST + judgment_module.FAVOR_CORRECT_BANISH + ) + + +@pytest.mark.asyncio +async def test_a_genuinely_new_presence_reopens_the_judgment_purse( + sync_client, db_session, monkeypatch +): + _verdict_outcomes(monkeypatch, {"trust": _CORRECT_TRUST}) + + token = _login(sync_client, "judgment-second-spirit") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + _judge(ws, "trust") + after_first, _, _ = await _ledger(db_session, user_id) + + _summon(ws) + _judge(ws, "trust") + after_second, _, _ = await _ledger(db_session, user_id) + + second_verdict = after_second - after_first - SUMMON_ESSENCE_TRICKLE + assert second_verdict == CORRECT_JUDGMENT_ESSENCE, ( + "a fresh presence did not re-open the purse — every spirit after the " + "first is unjudgeable for reward" + ) + + +@pytest.mark.asyncio +async def test_the_test_verdict_stays_freely_repeatable(sync_client, db_session, monkeypatch): + """`test` is a diagnostic pulse that never touches the ledger, so it is + deliberately exempt from the once-per-presence guard — the panel lets a + seeker pulse a spirit as often as they like. + """ + token = _login(sync_client, "judgment-tester") + user_id = _user_id(sync_client, token) + + with _ws_connect(sync_client, token) as ws: + _read_until(ws, "session") + _summon(ws) + baseline, baseline_favor, _ = await _ledger(db_session, user_id) + results = [_judge(ws, "test") for _ in range(3)] + essence_end, favor_end, _ = await _ledger(db_session, user_id) + + assert all(r["consequence"] == "neutral" for r in results) + assert essence_end == baseline + assert favor_end == baseline_favor