(null)
+ const [checkingSession, setCheckingSession] = useState(true)
+
+ useEffect(() => {
+ me()
+ .then(setUser)
+ .catch(() => setUser(null))
+ .finally(() => setCheckingSession(false))
+ }, [])
+
+ async function handleSubmit(event: React.FormEvent) {
+ event.preventDefault()
+ setError(null)
+ try {
+ if (mode === 'register') {
+ await register(username, password)
+ }
+ const loggedInUser = await login(username, password)
+ setUser(loggedInUser)
+ } catch (err) {
+ setError(err instanceof Error ? err.message : 'Something went wrong')
+ }
+ }
+
+ async function handleLogout() {
+ await logout()
+ setUser(null)
+ }
+
+ if (checkingSession) {
+ return (
+
+ Listening for a signal...
+
+ )
+ }
+
+ return (
+
+ Quantumancy
+ Something is listening on the other side.
+
+ {user ? (
+
+ Contact established as {user.username}.
+
+
+ ) : (
+
+ )}
+
+ )
+}
+
+export default App
+```
+
+- [ ] **Step 5: Run test to verify it passes**
+
+Run: `cd frontend && npm test`
+Expected: PASS (2/2)
+
+- [ ] **Step 6: Build the production bundle**
+
+Run: `cd frontend && npm run build`
+Expected: succeeds, produces `frontend/dist/index.html` and `frontend/dist/assets/*`. Task 2 requires this directory to exist.
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add frontend
+git commit -m "feat: scaffold React frontend with login/register UI"
+```
+
+---
+
+## Task 2: Backend Serves the Frontend
+
+**Files:**
+- Modify: `backend/app/main.py`
+- Test: `backend/tests/test_frontend_serving.py`
+
+**Interfaces:**
+- Consumes: `frontend/dist/` (built in Task 1 — this task's tests will fail if it doesn't exist; that's expected, not a bug in this task).
+- Produces: nothing new for later tasks — this is the terminal "make it visitable" step for the frontend.
+
+- [ ] **Step 1: Write the failing test**
+
+`backend/tests/test_frontend_serving.py`:
+```python
+import pytest
+
+
+@pytest.mark.asyncio
+async def test_unknown_path_serves_spa_index(client):
+ response = await client.get("/some/client/side/route")
+ assert response.status_code == 200
+ assert '' in response.text
+
+
+@pytest.mark.asyncio
+async def test_auth_routes_still_work_alongside_spa_fallback(client):
+ response = await client.post(
+ "/auth/register", json={"username": "frontendcheck", "password": "spookyspooky"}
+ )
+ assert response.status_code == 201
+```
+
+- [ ] **Step 2: Run test to verify it fails**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_frontend_serving.py -v
+```
+Expected: FAIL — `assert 404 == 200` (no catch-all route exists yet, unknown paths 404).
+
+- [ ] **Step 3: Write minimal implementation**
+
+`backend/app/main.py` (replace entirely):
+```python
+from contextlib import asynccontextmanager
+from pathlib import Path
+
+from fastapi import FastAPI
+from fastapi.responses import FileResponse
+from fastapi.staticfiles import StaticFiles
+
+import app.models # noqa: F401 — registers models on Base.metadata before create_all
+from app.db import Base, engine
+from app.routes.auth import router as auth_router
+
+FRONTEND_DIST = Path(__file__).resolve().parent.parent.parent / "frontend" / "dist"
+
+
+@asynccontextmanager
+async def lifespan(app: FastAPI):
+ async with engine.begin() as conn:
+ await conn.run_sync(Base.metadata.create_all)
+ yield
+
+
+app = FastAPI(title="Quantumancy", lifespan=lifespan)
+app.include_router(auth_router)
+
+
+@app.get("/healthz")
+async def healthz():
+ return {"status": "ok"}
+
+
+app.mount("/assets", StaticFiles(directory=FRONTEND_DIST / "assets"), name="frontend-assets")
+
+
+@app.get("/{full_path:path}")
+async def serve_spa(full_path: str):
+ return FileResponse(FRONTEND_DIST / "index.html")
+```
+
+Note: the catch-all `serve_spa` route is registered last, after `/healthz` and the `auth_router` include, so those routes still take precedence — FastAPI matches in registration order.
+
+- [ ] **Step 4: Run test to verify it passes**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_frontend_serving.py -v
+```
+Expected: PASS (2/2)
+
+Then the full suite:
+```bash
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest -v
+```
+Expected: all pass (14 total: 12 from Plan 1 + 2 new).
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add backend/app/main.py backend/tests/test_frontend_serving.py
+git commit -m "feat: serve built frontend from backend with SPA fallback"
+```
+
+---
+
+## Task 3: Ollama Client & Bounded Request Queue
+
+**Files:**
+- Create: `backend/app/llm/__init__.py`
+- Create: `backend/app/llm/client.py`
+- Create: `backend/app/llm/queue.py`
+- Test: `backend/tests/test_llm_queue.py`
+
+**Interfaces:**
+- Consumes: `settings.ollama_base_url` (Plan 1's `app.config`).
+- Produces: `OllamaClient` with async `generate(model: str, prompt: str, system: str | None = None) -> str` in `app.llm.client`; `LLMQueue(max_concurrency: int, max_queue_depth: int)` with async `submit(coro_factory: Callable[[], Awaitable[T]]) -> T` and `class QueueFullError(Exception)` in `app.llm.queue` — Plans 3-6 depend on `LLMQueue.submit` to serialize every spirit-mode LLM call.
+
+- [ ] **Step 1: Write the failing tests**
+
+`backend/tests/test_llm_queue.py`:
+```python
+import asyncio
+
+import pytest
+
+from app.llm.queue import LLMQueue, QueueFullError
+
+
+@pytest.mark.asyncio
+async def test_queue_runs_calls_up_to_concurrency_limit():
+ queue = LLMQueue(max_concurrency=2, max_queue_depth=10)
+ concurrent_count = 0
+ max_observed = 0
+
+ async def slow_call():
+ nonlocal concurrent_count, max_observed
+ concurrent_count += 1
+ max_observed = max(max_observed, concurrent_count)
+ await asyncio.sleep(0.05)
+ concurrent_count -= 1
+ return "done"
+
+ results = await asyncio.gather(*(queue.submit(slow_call) for _ in range(5)))
+
+ assert results == ["done"] * 5
+ assert max_observed == 2
+
+
+@pytest.mark.asyncio
+async def test_queue_raises_when_depth_exceeded():
+ queue = LLMQueue(max_concurrency=1, max_queue_depth=1)
+
+ async def slow_call():
+ await asyncio.sleep(0.1)
+ return "done"
+
+ task1 = asyncio.create_task(queue.submit(slow_call))
+ task2 = asyncio.create_task(queue.submit(slow_call))
+ await asyncio.sleep(0.01) # let task1 start running, task2 start waiting
+
+ with pytest.raises(QueueFullError):
+ await queue.submit(slow_call)
+
+ await asyncio.gather(task1, task2)
+```
+
+- [ ] **Step 2: Run test to verify it fails**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_llm_queue.py -v
+```
+Expected: FAIL with `ModuleNotFoundError: No module named 'app.llm'`
+
+- [ ] **Step 3: Write minimal implementation**
+
+`backend/app/llm/__init__.py`: (empty file)
+
+`backend/app/llm/client.py`:
+```python
+import httpx
+
+from app.config import settings
+
+
+class OllamaClient:
+ def __init__(self, base_url: str | None = None):
+ self._base_url = base_url or settings.ollama_base_url
+
+ async def generate(self, model: str, prompt: str, system: str | None = None) -> str:
+ payload = {"model": model, "prompt": prompt, "stream": False}
+ if system is not None:
+ payload["system"] = system
+
+ async with httpx.AsyncClient(base_url=self._base_url, timeout=120.0) as http_client:
+ response = await http_client.post("/api/generate", json=payload)
+ response.raise_for_status()
+ return response.json()["response"]
+```
+
+`backend/app/llm/queue.py`:
+```python
+import asyncio
+from collections.abc import Awaitable, Callable
+from typing import TypeVar
+
+T = TypeVar("T")
+
+
+class QueueFullError(Exception):
+ pass
+
+
+class LLMQueue:
+ """Bounds concurrent Ollama calls and rejects work once too much is queued."""
+
+ def __init__(self, max_concurrency: int, max_queue_depth: int):
+ self._semaphore = asyncio.Semaphore(max_concurrency)
+ self._max_queue_depth = max_queue_depth
+ self._waiting = 0
+ self._lock = asyncio.Lock()
+
+ async def submit(self, coro_factory: Callable[[], Awaitable[T]]) -> T:
+ async with self._lock:
+ if self._waiting >= self._max_queue_depth:
+ raise QueueFullError("too many seekers right now")
+ self._waiting += 1
+
+ try:
+ async with self._semaphore:
+ return await coro_factory()
+ finally:
+ async with self._lock:
+ self._waiting -= 1
+```
+
+- [ ] **Step 4: Run test to verify it passes**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_llm_queue.py -v
+```
+Expected: PASS (2/2)
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add backend/app/llm backend/tests/test_llm_queue.py
+git commit -m "feat: add Ollama client and bounded request queue"
+```
+
+---
+
+## Task 4: Piper TTS Wrapper & Effects Chain
+
+**Files:**
+- Create: `backend/app/tts/__init__.py`
+- Create: `backend/app/tts/piper.py`
+- Create: `backend/app/tts/effects.py`
+- Modify: `backend/requirements.txt`
+- Test: `backend/tests/test_tts_effects.py`
+
+**Interfaces:**
+- Produces: `PiperTTS(voice_model_path: str)` with `synthesize(text: str) -> bytes` (returns WAV bytes) in `app.tts.piper`; `apply_static_effect(wav_bytes: bytes, noise_level: float = 0.02) -> bytes` in `app.tts.effects` — Plans 3-5 call these to render spirit voices.
+
+Note: `PiperTTS.synthesize` requires a real Piper voice model file and the `piper` CLI, which this task does not install a model for (that's a deployment step for whichever plan first uses a mode with audio). This task's automated tests exercise only `app.tts.effects`, which is pure signal processing with no external dependency — `piper.py`'s wrapper is written and typed correctly but is not covered by an automated test here, since doing so would require downloading a voice model as part of CI/test setup. Flag this honestly rather than faking a test against a mocked subprocess that wouldn't catch real integration issues.
+
+- [ ] **Step 1: Add numpy-based effects dependency check**
+
+`numpy` is already in `backend/requirements.txt` from Plan 1 (pre-installed for this purpose) — confirm it's there, no new entry needed. Add `piper-tts` for the CLI/runtime:
+
+Append to `backend/requirements.txt`:
+```
+piper-tts==1.2.0
+```
+
+Run: `cd backend && source venv/bin/activate && pip install -r requirements.txt`
+
+- [ ] **Step 2: Write the failing test**
+
+`backend/tests/test_tts_effects.py`:
+```python
+import struct
+import wave
+from io import BytesIO
+
+from app.tts.effects import apply_static_effect
+
+
+def _make_silent_wav(duration_seconds: float = 0.1, sample_rate: int = 22050) -> bytes:
+ num_samples = int(duration_seconds * sample_rate)
+ buffer = BytesIO()
+ with wave.open(buffer, "wb") as wav_file:
+ wav_file.setnchannels(1)
+ wav_file.setsampwidth(2)
+ wav_file.setframerate(sample_rate)
+ wav_file.writeframes(struct.pack(f"<{num_samples}h", *([0] * num_samples)))
+ return buffer.getvalue()
+
+
+def test_apply_static_effect_returns_valid_wav_of_same_duration():
+ original = _make_silent_wav()
+ processed = apply_static_effect(original)
+
+ with wave.open(BytesIO(original)) as original_wav:
+ original_frames = original_wav.getnframes()
+ original_rate = original_wav.getframerate()
+
+ with wave.open(BytesIO(processed)) as processed_wav:
+ assert processed_wav.getnframes() == original_frames
+ assert processed_wav.getframerate() == original_rate
+ assert processed_wav.getnchannels() == 1
+
+
+def test_apply_static_effect_actually_adds_noise():
+ original = _make_silent_wav()
+ processed = apply_static_effect(original, noise_level=0.5)
+
+ with wave.open(BytesIO(processed)) as processed_wav:
+ frames = processed_wav.readframes(processed_wav.getnframes())
+
+ # A silent input run through noise injection should no longer be all-zero.
+ assert any(byte != 0 for byte in frames)
+```
+
+- [ ] **Step 3: Run test to verify it fails**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_tts_effects.py -v
+```
+Expected: FAIL with `ModuleNotFoundError: No module named 'app.tts'`
+
+- [ ] **Step 4: Write minimal implementation**
+
+`backend/app/tts/__init__.py`: (empty file)
+
+`backend/app/tts/effects.py`:
+```python
+import wave
+from io import BytesIO
+
+import numpy as np
+
+
+def apply_static_effect(wav_bytes: bytes, noise_level: float = 0.02) -> bytes:
+ """Adds white noise to a mono 16-bit PCM WAV, simulating spirit-box static."""
+ with wave.open(BytesIO(wav_bytes)) as wav_in:
+ params = wav_in.getparams()
+ frames = wav_in.readframes(wav_in.getnframes())
+
+ samples = np.frombuffer(frames, dtype=np.int16).astype(np.float32)
+ noise = np.random.normal(0, noise_level * 32767, size=samples.shape)
+ noisy_samples = np.clip(samples + noise, -32768, 32767).astype(np.int16)
+
+ output = BytesIO()
+ with wave.open(output, "wb") as wav_out:
+ wav_out.setparams(params)
+ wav_out.writeframes(noisy_samples.tobytes())
+ return output.getvalue()
+```
+
+`backend/app/tts/piper.py`:
+```python
+import subprocess
+
+
+class PiperTTS:
+ """Wraps the `piper` CLI to synthesize speech locally, no cloud calls."""
+
+ def __init__(self, voice_model_path: str):
+ self._voice_model_path = voice_model_path
+
+ def synthesize(self, text: str) -> bytes:
+ result = subprocess.run(
+ ["piper", "--model", self._voice_model_path, "--output-raw"],
+ input=text.encode("utf-8"),
+ capture_output=True,
+ check=True,
+ )
+ return result.stdout
+```
+
+- [ ] **Step 5: Run test to verify it passes**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_tts_effects.py -v
+```
+Expected: PASS (2/2)
+
+- [ ] **Step 6: Commit**
+
+```bash
+git add backend/app/tts backend/requirements.txt backend/tests/test_tts_effects.py
+git commit -m "feat: add Piper TTS wrapper and static-noise effects chain"
+```
+
+---
+
+## Task 5: WebSocket Session Channel
+
+**Files:**
+- Create: `backend/app/models/contact_session.py`
+- Modify: `backend/app/models/__init__.py`
+- Create: `backend/app/ws.py`
+- Modify: `backend/app/main.py`
+- Test: `backend/tests/test_ws_session.py`
+
+**Interfaces:**
+- Consumes: `get_current_user` (Plan 1's `app.deps`), `Base`/`get_db` (Plan 1's `app.db`).
+- Produces: `ContactSession` model (`id`, `user_id`, `mode`, `started_at`, `ended_at`) in `app.models.contact_session`; a `WS /ws/session` endpoint — Plans 3-5 build their per-mode message handling on top of this connection.
+
+- [ ] **Step 1: Write the failing test**
+
+`backend/tests/test_ws_session.py`:
+```python
+import pytest
+from sqlalchemy import select
+
+from app.models.contact_session import ContactSession
+
+
+@pytest.mark.asyncio
+async def test_websocket_requires_authentication(client):
+ with pytest.raises(Exception):
+ async with client.websocket_connect("/ws/session") as ws:
+ await ws.receive_json()
+
+
+@pytest.mark.asyncio
+async def test_websocket_ping_pong_and_session_lifecycle(client, db_session):
+ await client.post("/auth/register", json={"username": "wsmedium", "password": "spookyspooky"})
+ await client.post("/auth/login", json={"username": "wsmedium", "password": "spookyspooky"})
+
+ with client.websocket_connect("/ws/session") as ws:
+ ws.send_json({"type": "ping"})
+ response = ws.receive_json()
+ assert response == {"type": "pong"}
+
+ sessions = (await db_session.execute(select(ContactSession))).scalars().all()
+ assert len(sessions) == 1
+ assert sessions[0].ended_at is None
+
+ sessions = (await db_session.execute(select(ContactSession))).scalars().all()
+ assert sessions[0].ended_at is not None
+```
+
+Note: `httpx.AsyncClient` (used by the existing async `client` fixture) does not support WebSocket testing — `websocket_connect` is a `starlette.testclient.TestClient` (sync) method. This test uses a second, differently-named fixture, `sync_client`, plus a `db_session` fixture for direct DB assertions after the socket closes. **Do not modify or shadow the existing async `client` fixture** — add these as new, additional fixtures alongside it.
+
+Replace `backend/tests/test_ws_session.py`'s contents with this final version (uses `sync_client` throughout, since `TestClient` also handles regular HTTP calls, so there's no need to mix it with the async `client` fixture in this file):
+
+```python
+import pytest
+from sqlalchemy import select
+
+from app.models.contact_session import ContactSession
+
+
+def test_websocket_requires_authentication(sync_client):
+ with pytest.raises(Exception):
+ with sync_client.websocket_connect("/ws/session"):
+ pass
+
+
+@pytest.mark.asyncio
+async def test_websocket_ping_pong_and_session_lifecycle(sync_client, db_session):
+ sync_client.post("/auth/register", json={"username": "wsmedium", "password": "spookyspooky"})
+ sync_client.post("/auth/login", json={"username": "wsmedium", "password": "spookyspooky"})
+
+ with sync_client.websocket_connect("/ws/session") as ws:
+ ws.send_json({"type": "ping"})
+ response = ws.receive_json()
+ assert response == {"type": "pong"}
+
+ sessions = (await db_session.execute(select(ContactSession))).scalars().all()
+ assert len(sessions) == 1
+ assert sessions[0].ended_at is None
+
+ sessions = (await db_session.execute(select(ContactSession))).scalars().all()
+ assert sessions[0].ended_at is not None
+```
+
+Add these two new fixtures to `backend/tests/conftest.py` (append — do not touch the existing async `client`/`_reset_db` fixtures):
+```python
+from fastapi.testclient import TestClient
+
+
+@pytest.fixture
+def sync_client():
+ return TestClient(app)
+
+
+@pytest_asyncio.fixture
+async def db_session():
+ async with TestSessionLocal() as session:
+ yield session
+```
+
+`TestClient` and `pytest` need importing at the top of `conftest.py` — `pytest` itself isn't imported yet there (only `pytest_asyncio` is), so add `import pytest` alongside the existing imports.
+
+- [ ] **Step 2: Run test to verify it fails**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_ws_session.py -v
+```
+Expected: FAIL — connection to `/ws/session` fails outright (no route registered), or `ModuleNotFoundError` for `app.models.contact_session`.
+
+- [ ] **Step 3: Write minimal implementation**
+
+`backend/app/models/contact_session.py`:
+```python
+import uuid
+from datetime import datetime, timezone
+
+from sqlalchemy import DateTime, ForeignKey, String
+from sqlalchemy.orm import Mapped, mapped_column
+
+from app.db import Base
+
+
+class ContactSession(Base):
+ __tablename__ = "contact_sessions"
+
+ id: Mapped[uuid.UUID] = mapped_column(primary_key=True, default=uuid.uuid4)
+ user_id: Mapped[uuid.UUID] = mapped_column(ForeignKey("users.id"))
+ mode: Mapped[str] = mapped_column(String(32), default="unknown")
+ started_at: Mapped[datetime] = mapped_column(
+ DateTime(timezone=True), default=lambda: datetime.now(timezone.utc)
+ )
+ ended_at: Mapped[datetime | None] = mapped_column(DateTime(timezone=True), nullable=True)
+```
+
+`backend/app/models/__init__.py` (replace entirely):
+```python
+from app.models.auth_session import AuthSession
+from app.models.contact_session import ContactSession
+from app.models.user import User
+
+__all__ = ["User", "AuthSession", "ContactSession"]
+```
+
+`backend/app/ws.py`:
+```python
+from datetime import datetime, timezone
+
+from fastapi import APIRouter, WebSocket, WebSocketDisconnect
+from sqlalchemy import select
+
+from app.db import async_session_maker
+from app.deps import SESSION_COOKIE_NAME
+from app.models.auth_session import AuthSession, hash_token
+from app.models.contact_session import ContactSession
+
+router = APIRouter()
+
+
+async def _authenticate(websocket: WebSocket):
+ raw_token = websocket.cookies.get(SESSION_COOKIE_NAME)
+ if raw_token is None:
+ return None
+
+ token_hash = hash_token(raw_token)
+ async with async_session_maker() as db:
+ result = await db.execute(select(AuthSession).where(AuthSession.token_hash == token_hash))
+ session = result.scalar_one_or_none()
+ if session is None or session.expires_at < datetime.now(timezone.utc):
+ return None
+ return session.user_id
+
+
+@router.websocket("/ws/session")
+async def session_socket(websocket: WebSocket):
+ user_id = await _authenticate(websocket)
+ if user_id is None:
+ await websocket.close(code=4401)
+ return
+
+ await websocket.accept()
+
+ async with async_session_maker() as db:
+ contact_session = ContactSession(user_id=user_id)
+ db.add(contact_session)
+ await db.commit()
+ await db.refresh(contact_session)
+
+ try:
+ while True:
+ message = await websocket.receive_json()
+ if message.get("type") == "ping":
+ await websocket.send_json({"type": "pong"})
+ except WebSocketDisconnect:
+ pass
+ finally:
+ async with async_session_maker() as db:
+ session_to_close = await db.get(ContactSession, contact_session.id)
+ session_to_close.ended_at = datetime.now(timezone.utc)
+ await db.commit()
+```
+
+`backend/app/main.py` — add the WebSocket router (insert alongside the existing `auth_router` include, before the catch-all SPA route):
+```python
+from app.ws import router as ws_router
+...
+app.include_router(auth_router)
+app.include_router(ws_router)
+```
+
+- [ ] **Step 4: Run test to verify it passes**
+
+Run:
+```bash
+cd backend && source venv/bin/activate
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest tests/test_ws_session.py -v
+```
+Expected: PASS (2/2)
+
+Then the full suite:
+```bash
+DATABASE_URL=postgresql+asyncpg://quantumancy:quantumancy@localhost:5432/quantumancy_test \
+OLLAMA_BASE_URL=http://10.30.20.107:11434 \
+SESSION_SECRET=test-secret \
+python -m pytest -v
+```
+Expected: all pass (20 total: 14 from Tasks 1-2 range + 2 from Task 3 + 2 from Task 4 + 2 from Task 5).
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add backend/app/models/contact_session.py backend/app/models/__init__.py backend/app/ws.py backend/app/main.py backend/tests/test_ws_session.py backend/tests/conftest.py
+git commit -m "feat: add WebSocket session channel with ContactSession lifecycle"
+```
+
+---
+
+## Self-Review
+
+**Spec coverage:** Frontend React/Vite SPA served by backend (spec §2) → Tasks 1-2. Ollama request queue (spec §2 "known operational constraint") → Task 3. Local Piper TTS (spec §6) → Task 4 (effects chain fully tested; `piper.py`'s subprocess wrapper is honestly flagged as untested pending a real voice model, deferred to whichever Plan 3-5 task first plays audio). WebSocket per-session channel implied by spec §3's "streamed to the frontend as they occur" real-time requirement → Task 5. Rate limiting enforcement onto these endpoints is explicitly NOT in this plan — no LLM-triggering HTTP/WS endpoint with real spirit-mode content exists yet; that lands in Plan 3+ alongside the `RateLimiter` wiring.
+
+**Placeholder scan:** none — every step has complete, runnable code, and the one deliberate test gap (Piper's subprocess wrapper) is explicitly justified rather than faked with a hollow mock.
+
+**Type consistency:** `LLMQueue.submit(coro_factory)` signature matches its two test usages (`queue.submit(slow_call)` passing a zero-arg async callable). `OllamaClient.generate` signature (`model, prompt, system=None`) is what Plans 3-5 will call. `ContactSession` fields (`id`, `user_id`, `mode`, `started_at`, `ended_at`) match the spec's `contact_sessions` naming resolution from Plan 1. `apply_static_effect(wav_bytes, noise_level=0.02)` signature matches both test calls (one default, one explicit `noise_level=0.5`).
+
+---
+
+Plan complete and saved to `docs/superpowers/plans/2026-07-20-quantumancy-02-frontend-llm-pipeline.md`. Two execution options:
+
+**1. Subagent-Driven (recommended)** - I dispatch a fresh subagent per task, review between tasks, fast iteration
+
+**2. Inline Execution** - Execute tasks in this session using executing-plans, batch execution with checkpoints
+
+**Which approach?**