feat: possession presentation layer — glitchy text + degraded audio
First sub-project of the "make contact feel real" arc (spec: docs/superpowers/specs/2026-07-23-possession-presentation-design.md). Direct Contact replies now feel like a spirit fighting through static to hold the channel rather than a plain chat bubble: - backend/app/possession.py: compute_stability(rarity, magnitude, rng) — a 0.05-0.98 score per reply (rarer entity + stronger triggering anomaly = cleaner signal), rng injectable for a later quantum-RNG source. - ws.py sends stability on reply_start; audio synthesis for that reply gets noise/bitcrush scaled by instability (1 - stability) via a new instability param on synthesize_spirit_voice — effects.py itself is untouched, only the params fed into it. - frontend/src/lib/possession.ts: renderPossessedText — pure, deterministic (tick-seeded, no Math.random) text corruption with self-correcting glitch bursts, wired into Transcript.tsx's streaming reply display. Stored transcript/reply text is unaffected — this is presentation only. 78/78 backend, 137/137 frontend tests passing.
This commit is contained in:
39
backend/app/possession.py
Normal file
39
backend/app/possession.py
Normal file
@@ -0,0 +1,39 @@
|
||||
"""Possession stability: a single number expressing how cleanly a spirit
|
||||
holds the channel for one reply. Shared contract between the backend audio
|
||||
degradation and the frontend text-glitch renderer (see
|
||||
docs/superpowers/specs/2026-07-23-possession-presentation-design.md)."""
|
||||
|
||||
import random
|
||||
from typing import Callable
|
||||
|
||||
_BASE_BY_RARITY = {
|
||||
"common": 0.35,
|
||||
"uncommon": 0.5,
|
||||
"rare": 0.65,
|
||||
"mythic": 0.8,
|
||||
}
|
||||
|
||||
_MIN_STABILITY = 0.05
|
||||
_MAX_STABILITY = 0.98
|
||||
_MAX_MAGNITUDE_BONUS = 0.15
|
||||
_JITTER_SPAN = 0.1 # rng() in [0, 1) is scaled to +/- this amount
|
||||
|
||||
|
||||
def compute_stability(
|
||||
rarity: str,
|
||||
magnitude: float,
|
||||
rng: Callable[[], float] = random.random,
|
||||
) -> float:
|
||||
"""How cleanly the spirit holds the channel for this reply, 0.05-0.98.
|
||||
|
||||
Rarer spirits hold the channel more steadily (higher base). A stronger
|
||||
triggering anomaly gives a cleaner line, up to a capped bonus. A final
|
||||
rng-driven jitter of +/- 0.1 keeps it from feeling deterministic to the
|
||||
seeker. `rng` is injectable so a later quantum-RNG source can swap in
|
||||
real entropy without touching call sites.
|
||||
"""
|
||||
base = _BASE_BY_RARITY.get(rarity, _BASE_BY_RARITY["common"])
|
||||
magnitude_bonus = min(_MAX_MAGNITUDE_BONUS, magnitude / 100)
|
||||
jitter = (rng() * 2 - 1) * _JITTER_SPAN
|
||||
stability = base + magnitude_bonus + jitter
|
||||
return max(_MIN_STABILITY, min(_MAX_STABILITY, stability))
|
||||
@@ -50,18 +50,48 @@ class PiperTTS:
|
||||
|
||||
|
||||
async def synthesize_spirit_voice(
|
||||
text: str, voice: Voice, voice_profile: dict | None = None
|
||||
text: str,
|
||||
voice: Voice,
|
||||
voice_profile: dict | None = None,
|
||||
instability: float = 0.0,
|
||||
) -> bytes:
|
||||
"""One call: Piper synth + the spirit's signature effects chain."""
|
||||
"""One call: Piper synth + the spirit's signature effects chain.
|
||||
|
||||
`instability` (0.0-1.0, i.e. `1 - stability`) scales up the noise and
|
||||
bitcrush fed into the effects chain when the spirit is fighting to hold
|
||||
the channel — a shaky possession sounds shakier. It never touches
|
||||
`apply_effects` itself, only the values handed to it here.
|
||||
"""
|
||||
profile = voice_profile or {}
|
||||
instability = max(0.0, min(1.0, instability))
|
||||
model_path = Path(settings.piper_voices_dir) / voice.model_file
|
||||
|
||||
base_noise = float(profile.get("noise", 0.03))
|
||||
# Noise ramps linearly across the whole range so degradation is audible
|
||||
# from the first sign of instability, but doubling the base level at
|
||||
# instability=1.0 (rather than, say, 5x) keeps the words intelligible
|
||||
# even at a full-static possession — this is texture, not white noise.
|
||||
noise_level = base_noise + instability * base_noise
|
||||
|
||||
base_bitcrush = int(profile.get("bitcrush", 0))
|
||||
# Bitcrush only kicks in once instability crosses ~0.5 -- a slightly
|
||||
# unstable channel just sounds noisier, not crunchy/robotic. Above that
|
||||
# threshold it ramps up to +6 bits of crush on top of whatever the
|
||||
# voice's own profile already applies, capped at 15 (apply_effects only
|
||||
# crushes when 0 < bits < 16).
|
||||
if instability > 0.5:
|
||||
bitcrush_bonus = round((instability - 0.5) / 0.5 * 6)
|
||||
else:
|
||||
bitcrush_bonus = 0
|
||||
bitcrush_bits = min(15, base_bitcrush + bitcrush_bonus)
|
||||
|
||||
wav = await PiperTTS(str(model_path)).synthesize(text)
|
||||
return await asyncio.to_thread(
|
||||
apply_effects,
|
||||
wav,
|
||||
noise_level=float(profile.get("noise", 0.03)),
|
||||
noise_level=noise_level,
|
||||
pitch_semitones=float(profile.get("pitch", 0.0)),
|
||||
rate=float(profile.get("rate", 1.0)),
|
||||
bitcrush_bits=int(profile.get("bitcrush", 0)),
|
||||
bitcrush_bits=bitcrush_bits,
|
||||
echo=float(profile.get("echo", 0.2)),
|
||||
)
|
||||
|
||||
@@ -36,6 +36,7 @@ from app.models.contact_session import ContactSession
|
||||
from app.models.entity import Entity
|
||||
from app.models.entity_sighting import EntitySighting
|
||||
from app.models.event import Event
|
||||
from app.possession import compute_stability
|
||||
from app.rate_limit import RateLimiter, resolve_client_ip
|
||||
from app.telemetry import detect_wire_spike, sample_network
|
||||
from app.tts.piper import synthesize_spirit_voice
|
||||
@@ -147,7 +148,9 @@ async def _record_event(
|
||||
return event.id
|
||||
|
||||
|
||||
async def _speak(state: SeanceState, kind: str, text: str) -> None:
|
||||
async def _speak(
|
||||
state: SeanceState, kind: str, text: str, instability: float = 0.0
|
||||
) -> None:
|
||||
"""Persist an utterance, push its text immediately, synthesize audio in
|
||||
the background, and push the audio URL when the effects chain finishes."""
|
||||
event_id = await _record_event(
|
||||
@@ -167,7 +170,9 @@ async def _speak(state: SeanceState, kind: str, text: str) -> None:
|
||||
try:
|
||||
voice_profile = state.entity.get("voice", {}) if state.entity else {}
|
||||
voice = pick_voice(voice_profile.get("voice_id"), state.language)
|
||||
wav = await synthesize_spirit_voice(text, voice, voice_profile)
|
||||
wav = await synthesize_spirit_voice(
|
||||
text, voice, voice_profile, instability=instability
|
||||
)
|
||||
filename = f"{event_id}.wav"
|
||||
_audio_dir().joinpath(filename).write_bytes(wav)
|
||||
async with session_maker() as db:
|
||||
@@ -323,7 +328,12 @@ async def _handle_question(state: SeanceState, text: str) -> None:
|
||||
text = text.strip()[:500]
|
||||
await _record_event(state.session_id, "question", text=text)
|
||||
await state.send_queue.put({"type": "status", "state": "gathering"})
|
||||
await state.send_queue.put({"type": "reply_start"})
|
||||
|
||||
last_magnitude = state.anomalies[-1].get("magnitude") if state.anomalies else 50
|
||||
if last_magnitude is None:
|
||||
last_magnitude = 50
|
||||
stability = compute_stability(state.entity["rarity"], float(last_magnitude))
|
||||
await state.send_queue.put({"type": "reply_start", "stability": stability})
|
||||
|
||||
reply_parts: list[str] = []
|
||||
try:
|
||||
@@ -350,7 +360,7 @@ async def _handle_question(state: SeanceState, text: str) -> None:
|
||||
state.history = state.history[-8:]
|
||||
await state.send_queue.put({"type": "reply_end", "id": str(reply_id), "text": reply})
|
||||
if reply:
|
||||
await _speak(state, "reply", reply)
|
||||
await _speak(state, "reply", reply, instability=1 - stability)
|
||||
|
||||
|
||||
async def _ambient_loop(state: SeanceState) -> None:
|
||||
|
||||
69
backend/tests/test_possession.py
Normal file
69
backend/tests/test_possession.py
Normal file
@@ -0,0 +1,69 @@
|
||||
from app.possession import compute_stability
|
||||
|
||||
|
||||
def test_rarity_ordering_holds_magnitude_and_rng_constant():
|
||||
# Midpoint rng (0.5) zeroes out jitter, isolating the rarity base.
|
||||
rng = lambda: 0.5
|
||||
common = compute_stability("common", magnitude=50, rng=rng)
|
||||
uncommon = compute_stability("uncommon", magnitude=50, rng=rng)
|
||||
rare = compute_stability("rare", magnitude=50, rng=rng)
|
||||
mythic = compute_stability("mythic", magnitude=50, rng=rng)
|
||||
|
||||
assert common < uncommon < rare < mythic
|
||||
|
||||
|
||||
def test_unknown_rarity_falls_back_to_common_base():
|
||||
rng = lambda: 0.5
|
||||
assert compute_stability("legendary???", magnitude=50, rng=rng) == compute_stability(
|
||||
"common", magnitude=50, rng=rng
|
||||
)
|
||||
|
||||
|
||||
def test_magnitude_bonus_increases_stability_up_to_cap():
|
||||
rng = lambda: 0.5
|
||||
low = compute_stability("common", magnitude=0, rng=rng)
|
||||
mid = compute_stability("common", magnitude=10, rng=rng)
|
||||
high = compute_stability("common", magnitude=1000, rng=rng)
|
||||
|
||||
assert low < mid < high
|
||||
# Bonus caps at 0.15, so magnitude=15 and magnitude=1000 must land the
|
||||
# same once base and jitter are held constant.
|
||||
at_cap = compute_stability("common", magnitude=15, rng=rng)
|
||||
past_cap = compute_stability("common", magnitude=1000, rng=rng)
|
||||
assert at_cap == past_cap
|
||||
|
||||
|
||||
def test_jitter_scaled_to_plus_minus_point_one():
|
||||
# rng()=0 -> minimum jitter (-0.1); rng()=1 -> maximum jitter (+0.1).
|
||||
base = 0.35 # common base
|
||||
low_jitter = compute_stability("common", magnitude=0, rng=lambda: 0.0)
|
||||
high_jitter = compute_stability("common", magnitude=0, rng=lambda: 1.0)
|
||||
|
||||
assert low_jitter == max(0.05, base - 0.1)
|
||||
assert high_jitter == min(0.98, base + 0.1)
|
||||
|
||||
|
||||
def test_clamped_at_lower_bound():
|
||||
# common base 0.35 minus full jitter (0.1) minus nothing else is 0.25,
|
||||
# comfortably above the floor -- force the floor with a negative-leaning
|
||||
# setup instead: rarity base + jitter alone can't go below 0.05, but the
|
||||
# clamp itself must still be exercised directly via a pathological rng.
|
||||
stability = compute_stability("common", magnitude=0, rng=lambda: -100.0)
|
||||
assert stability == 0.05
|
||||
|
||||
|
||||
def test_clamped_at_upper_bound():
|
||||
stability = compute_stability("mythic", magnitude=1000, rng=lambda: 100.0)
|
||||
assert stability == 0.98
|
||||
|
||||
|
||||
def test_deterministic_with_injected_rng():
|
||||
rng = lambda: 0.73
|
||||
first = compute_stability("rare", magnitude=42, rng=rng)
|
||||
second = compute_stability("rare", magnitude=42, rng=rng)
|
||||
assert first == second
|
||||
|
||||
|
||||
def test_default_rng_produces_value_in_range():
|
||||
stability = compute_stability("uncommon", magnitude=25)
|
||||
assert 0.05 <= stability <= 0.98
|
||||
118
backend/tests/test_tts_piper.py
Normal file
118
backend/tests/test_tts_piper.py
Normal file
@@ -0,0 +1,118 @@
|
||||
import pytest
|
||||
|
||||
import app.tts.piper as piper
|
||||
from app.tts.voices import VOICES
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _fake_piper_synth(monkeypatch):
|
||||
async def _fake_synthesize(self, text):
|
||||
return b"fake-wav-bytes"
|
||||
|
||||
monkeypatch.setattr(piper.PiperTTS, "synthesize", _fake_synthesize)
|
||||
|
||||
|
||||
def _capture_effects_call(monkeypatch):
|
||||
calls = []
|
||||
|
||||
def _fake_apply_effects(wav_bytes, **kwargs):
|
||||
calls.append(kwargs)
|
||||
return wav_bytes
|
||||
|
||||
monkeypatch.setattr(piper, "apply_effects", _fake_apply_effects)
|
||||
return calls
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_zero_instability_keeps_profile_defaults(monkeypatch):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
voice = VOICES["lessac"]
|
||||
|
||||
await piper.synthesize_spirit_voice("hello", voice, {"noise": 0.03, "bitcrush": 0})
|
||||
|
||||
assert calls[0]["noise_level"] == pytest.approx(0.03)
|
||||
assert calls[0]["bitcrush_bits"] == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_instability_increases_noise_level(monkeypatch):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
voice = VOICES["lessac"]
|
||||
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 0}, instability=1.0
|
||||
)
|
||||
|
||||
# Meaningfully louder than the clean baseline, but not blown out.
|
||||
assert calls[0]["noise_level"] > 0.03
|
||||
assert calls[0]["noise_level"] <= 0.1
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_noise_level_scales_monotonically_with_instability(monkeypatch):
|
||||
voice = VOICES["lessac"]
|
||||
noise_levels = []
|
||||
for instability in (0.0, 0.25, 0.5, 0.75, 1.0):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 0}, instability=instability
|
||||
)
|
||||
noise_levels.append(calls[0]["noise_level"])
|
||||
|
||||
assert noise_levels == sorted(noise_levels)
|
||||
assert noise_levels[0] < noise_levels[-1]
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bitcrush_stays_off_below_threshold(monkeypatch):
|
||||
voice = VOICES["lessac"]
|
||||
for instability in (0.0, 0.2, 0.5):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 0}, instability=instability
|
||||
)
|
||||
assert calls[0]["bitcrush_bits"] == 0
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_bitcrush_kicks_in_above_threshold(monkeypatch):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
voice = VOICES["lessac"]
|
||||
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 0}, instability=1.0
|
||||
)
|
||||
|
||||
assert calls[0]["bitcrush_bits"] > 0
|
||||
assert calls[0]["bitcrush_bits"] < 16
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_instability_is_clamped_to_unit_range(monkeypatch):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
voice = VOICES["lessac"]
|
||||
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 0}, instability=5.0
|
||||
)
|
||||
clamped_high = calls[0]
|
||||
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 0}, instability=1.0
|
||||
)
|
||||
at_one = calls[0]
|
||||
|
||||
assert clamped_high == at_one
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_instability_adds_on_top_of_profile_bitcrush(monkeypatch):
|
||||
calls = _capture_effects_call(monkeypatch)
|
||||
voice = VOICES["lessac"]
|
||||
|
||||
await piper.synthesize_spirit_voice(
|
||||
"hello", voice, {"noise": 0.03, "bitcrush": 4}, instability=1.0
|
||||
)
|
||||
|
||||
assert calls[0]["bitcrush_bits"] > 4
|
||||
@@ -31,7 +31,7 @@ class FakeSpiritService:
|
||||
return False
|
||||
|
||||
|
||||
async def _fake_synth(text, voice, profile):
|
||||
async def _fake_synth(text, voice, profile, instability=0.0):
|
||||
return b"RIFFfake wav bytes"
|
||||
|
||||
|
||||
@@ -124,7 +124,8 @@ async def test_question_streams_reply_and_records_history(sync_client, db_sessio
|
||||
_read_until(ws, "session")
|
||||
ws.send_json({"type": "question", "text": "Are you at peace?"})
|
||||
_read_until(ws, "entity") # auto-summoned before answering
|
||||
_read_until(ws, "reply_start")
|
||||
reply_start = _read_until(ws, "reply_start")
|
||||
assert 0.05 <= reply_start["stability"] <= 0.98
|
||||
reply_end = _read_until(ws, "reply_end")
|
||||
assert reply_end["text"] == "I am here."
|
||||
|
||||
|
||||
Reference in New Issue
Block a user