Files
qtalker---/frontend/src/lib/camera.test.ts
Indiana d37bb71e5d feat: the lens — the camera as a channel the dead look through
A new séance mode. The seeker opens their camera, presses "let it look",
and the entity speaks about what is ACTUALLY in the room — the configured
chat model (minicpm-v4.5:8b) is vision-capable, so this is real perception,
not invented description. Same principle as every other channel here: real
measurement first, interpretation second.

Verified live end-to-end through the real WebSocket: given a synthetic room
(pale doorway, red flame on dark boards), "Bessie L. Carter" reported the
gray rectangle and red square on a dark surface with faint shadows, then
misread it as her pen feeling heavy the night before Mr. Edgerton's birdseed
arrived. Accuracy followed by wrongness, which is the whole effect.

Privacy is the load-bearing design constraint, not a footnote:
- "Camera open" and "the entity saw something" are deliberately separate
  states. Opening the lens transmits NOTHING; only an explicit press sends
  one still. There is no timer and no background capture path.
- Frames are downscaled to 768px and JPEG-compressed client-side, then
  passed to the model and dropped. Never written to disk, never logged,
  never attached to an event row — only the resulting utterance is stored,
  exactly like any other thing a spirit says.
- The prompt forbids describing faces or guessing anyone's identity, age or
  appearance; a person present is spoken of only as a presence.
- A closed lens is covered by an opaque veil in the UI, so there is never
  ambiguity about whether the camera is live.

Robustness:
- CameraEye carries the same generation guard the EVP listener needed:
  closing during the permission prompt releases the late-arriving stream
  instead of letting the camera go live after teardown.
- Failures are classified (denied / insecure / absent / busy / unknown)
  rather than always blaming the seeker for a refusal.
- Scrying is the heaviest request this app makes of a CPU-only Ollama box,
  so it gets the tightest limiter of any channel (4/min/user, 8/min/IP).
- Frames are size-capped BEFORE reaching the queue, and a vision failure
  emits an error frame instead of killing the socket — both covered by
  tests asserting the model was never called.

10 new frontend tests, 5 new backend tests. 385 frontend + backend suites
pass; i18n parity holds across both languages.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-07-30 02:08:02 +00:00

168 lines
5.2 KiB
TypeScript

import { afterEach, describe, expect, it, vi } from 'vitest'
import {
CAPTURE_MAX_EDGE,
CameraEye,
classifyFailure,
isSupported,
toBase64,
} from './camera'
afterEach(() => {
vi.unstubAllGlobals()
})
/** getUserMedia whose resolution we control, with track-stop spies. */
function stubCamera() {
const stopped: string[] = []
const tracks = [{ stop: () => stopped.push('video') }]
let release: (() => void) | null = null
const pending = new Promise<MediaStream>((resolve) => {
release = () => resolve({ getTracks: () => tracks } as unknown as MediaStream)
})
vi.stubGlobal('navigator', { mediaDevices: { getUserMedia: () => pending } })
return { stopped, release: () => release!() }
}
function fakeVideo(w = 1920, h = 1080) {
return {
videoWidth: w,
videoHeight: h,
srcObject: null as unknown,
muted: false,
playsInline: false,
play: () => Promise.resolve(),
} as unknown as HTMLVideoElement
}
describe('classifyFailure', () => {
it('distinguishes a real refusal from every other cause', () => {
vi.stubGlobal('navigator', { mediaDevices: { getUserMedia: () => undefined } })
expect(classifyFailure(new DOMException('x', 'NotAllowedError'))).toBe('denied')
expect(classifyFailure(new DOMException('x', 'SecurityError'))).toBe('insecure')
expect(classifyFailure(new DOMException('x', 'NotFoundError'))).toBe('absent')
expect(classifyFailure(new DOMException('x', 'NotReadableError'))).toBe('busy')
// An unrecognised error must not be reported as a refusal the seeker made.
expect(classifyFailure(new Error('who knows'))).toBe('unknown')
})
it('reports an unsupported context as insecure rather than denied', () => {
vi.stubGlobal('navigator', {})
expect(isSupported()).toBe(false)
expect(classifyFailure(new DOMException('x', 'NotAllowedError'))).toBe('insecure')
})
})
describe('toBase64', () => {
it('strips the data-URL prefix Ollama does not want', () => {
expect(toBase64('data:image/jpeg;base64,AAAA')).toBe('AAAA')
})
it('passes through a string that is already bare base64', () => {
expect(toBase64('AAAA')).toBe('AAAA')
})
})
describe('CameraEye lifecycle', () => {
it('releases the camera when closed while the permission prompt is open', async () => {
// The hazard: the stream is only assigned after the await, so a close()
// during the prompt would otherwise release nothing and the camera would
// go live *after* teardown, leaving the recording light on.
const { stopped, release } = stubCamera()
const eye = new CameraEye()
const video = fakeVideo()
const opening = eye.open(video)
eye.close() // seeker switches mode mid-prompt
release()
await opening
expect(stopped).toEqual(['video'])
expect(eye.isOpen).toBe(false)
})
it('opens normally when nobody interrupts', async () => {
const { stopped, release } = stubCamera()
const eye = new CameraEye()
const video = fakeVideo()
const opening = eye.open(video)
release()
await opening
expect(eye.isOpen).toBe(true)
expect(stopped).toEqual([])
// iOS Safari refuses to start a stream without both of these.
expect(video.muted).toBe(true)
expect(video.playsInline).toBe(true)
eye.close()
expect(stopped).toEqual(['video'])
expect(eye.isOpen).toBe(false)
})
it('captures nothing when the lens is closed', () => {
const eye = new CameraEye()
expect(eye.capture(fakeVideo())).toBeNull()
})
it('captures nothing before video metadata arrives', async () => {
const { release } = stubCamera()
const eye = new CameraEye()
const video = fakeVideo(0, 0) // dimensions not known yet
const opening = eye.open(video)
release()
await opening
expect(eye.capture(video)).toBeNull()
})
it('downscales a large frame so the payload stays small', async () => {
const { release } = stubCamera()
const eye = new CameraEye()
const video = fakeVideo(1920, 1080)
// Record what the canvas was sized to.
const sizes: Array<{ w: number; h: number }> = []
const canvas = {
width: 0,
height: 0,
getContext: () => ({ drawImage: () => undefined }),
toDataURL: () => {
sizes.push({ w: canvas.width, h: canvas.height })
return 'data:image/jpeg;base64,ZZZZ'
},
}
vi.stubGlobal('document', { createElement: () => canvas })
const opening = eye.open(video)
release()
await opening
expect(eye.capture(video)).toBe('data:image/jpeg;base64,ZZZZ')
expect(sizes).toHaveLength(1)
// Longest edge clamped, aspect ratio preserved.
expect(sizes[0].w).toBe(CAPTURE_MAX_EDGE)
expect(sizes[0].h).toBe(Math.round((1080 / 1920) * CAPTURE_MAX_EDGE))
})
it('never upscales a frame smaller than the cap', async () => {
const { release } = stubCamera()
const eye = new CameraEye()
const video = fakeVideo(320, 240)
const canvas = {
width: 0,
height: 0,
getContext: () => ({ drawImage: () => undefined }),
toDataURL: () => 'data:image/jpeg;base64,S',
}
vi.stubGlobal('document', { createElement: () => canvas })
const opening = eye.open(video)
release()
await opening
eye.capture(video)
expect(canvas.width).toBe(320)
expect(canvas.height).toBe(240)
})
})