A new séance mode. The seeker opens their camera, presses "let it look", and the entity speaks about what is ACTUALLY in the room — the configured chat model (minicpm-v4.5:8b) is vision-capable, so this is real perception, not invented description. Same principle as every other channel here: real measurement first, interpretation second. Verified live end-to-end through the real WebSocket: given a synthetic room (pale doorway, red flame on dark boards), "Bessie L. Carter" reported the gray rectangle and red square on a dark surface with faint shadows, then misread it as her pen feeling heavy the night before Mr. Edgerton's birdseed arrived. Accuracy followed by wrongness, which is the whole effect. Privacy is the load-bearing design constraint, not a footnote: - "Camera open" and "the entity saw something" are deliberately separate states. Opening the lens transmits NOTHING; only an explicit press sends one still. There is no timer and no background capture path. - Frames are downscaled to 768px and JPEG-compressed client-side, then passed to the model and dropped. Never written to disk, never logged, never attached to an event row — only the resulting utterance is stored, exactly like any other thing a spirit says. - The prompt forbids describing faces or guessing anyone's identity, age or appearance; a person present is spoken of only as a presence. - A closed lens is covered by an opaque veil in the UI, so there is never ambiguity about whether the camera is live. Robustness: - CameraEye carries the same generation guard the EVP listener needed: closing during the permission prompt releases the late-arriving stream instead of letting the camera go live after teardown. - Failures are classified (denied / insecure / absent / busy / unknown) rather than always blaming the seeker for a refusal. - Scrying is the heaviest request this app makes of a CPU-only Ollama box, so it gets the tightest limiter of any channel (4/min/user, 8/min/IP). - Frames are size-capped BEFORE reaching the queue, and a vision failure emits an error frame instead of killing the socket — both covered by tests asserting the model was never called. 10 new frontend tests, 5 new backend tests. 385 frontend + backend suites pass; i18n parity holds across both languages. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
168 lines
5.2 KiB
TypeScript
168 lines
5.2 KiB
TypeScript
import { afterEach, describe, expect, it, vi } from 'vitest'
|
|
import {
|
|
CAPTURE_MAX_EDGE,
|
|
CameraEye,
|
|
classifyFailure,
|
|
isSupported,
|
|
toBase64,
|
|
} from './camera'
|
|
|
|
afterEach(() => {
|
|
vi.unstubAllGlobals()
|
|
})
|
|
|
|
/** getUserMedia whose resolution we control, with track-stop spies. */
|
|
function stubCamera() {
|
|
const stopped: string[] = []
|
|
const tracks = [{ stop: () => stopped.push('video') }]
|
|
let release: (() => void) | null = null
|
|
const pending = new Promise<MediaStream>((resolve) => {
|
|
release = () => resolve({ getTracks: () => tracks } as unknown as MediaStream)
|
|
})
|
|
vi.stubGlobal('navigator', { mediaDevices: { getUserMedia: () => pending } })
|
|
return { stopped, release: () => release!() }
|
|
}
|
|
|
|
function fakeVideo(w = 1920, h = 1080) {
|
|
return {
|
|
videoWidth: w,
|
|
videoHeight: h,
|
|
srcObject: null as unknown,
|
|
muted: false,
|
|
playsInline: false,
|
|
play: () => Promise.resolve(),
|
|
} as unknown as HTMLVideoElement
|
|
}
|
|
|
|
describe('classifyFailure', () => {
|
|
it('distinguishes a real refusal from every other cause', () => {
|
|
vi.stubGlobal('navigator', { mediaDevices: { getUserMedia: () => undefined } })
|
|
expect(classifyFailure(new DOMException('x', 'NotAllowedError'))).toBe('denied')
|
|
expect(classifyFailure(new DOMException('x', 'SecurityError'))).toBe('insecure')
|
|
expect(classifyFailure(new DOMException('x', 'NotFoundError'))).toBe('absent')
|
|
expect(classifyFailure(new DOMException('x', 'NotReadableError'))).toBe('busy')
|
|
// An unrecognised error must not be reported as a refusal the seeker made.
|
|
expect(classifyFailure(new Error('who knows'))).toBe('unknown')
|
|
})
|
|
|
|
it('reports an unsupported context as insecure rather than denied', () => {
|
|
vi.stubGlobal('navigator', {})
|
|
expect(isSupported()).toBe(false)
|
|
expect(classifyFailure(new DOMException('x', 'NotAllowedError'))).toBe('insecure')
|
|
})
|
|
})
|
|
|
|
describe('toBase64', () => {
|
|
it('strips the data-URL prefix Ollama does not want', () => {
|
|
expect(toBase64('data:image/jpeg;base64,AAAA')).toBe('AAAA')
|
|
})
|
|
|
|
it('passes through a string that is already bare base64', () => {
|
|
expect(toBase64('AAAA')).toBe('AAAA')
|
|
})
|
|
})
|
|
|
|
describe('CameraEye lifecycle', () => {
|
|
it('releases the camera when closed while the permission prompt is open', async () => {
|
|
// The hazard: the stream is only assigned after the await, so a close()
|
|
// during the prompt would otherwise release nothing and the camera would
|
|
// go live *after* teardown, leaving the recording light on.
|
|
const { stopped, release } = stubCamera()
|
|
const eye = new CameraEye()
|
|
const video = fakeVideo()
|
|
|
|
const opening = eye.open(video)
|
|
eye.close() // seeker switches mode mid-prompt
|
|
release()
|
|
await opening
|
|
|
|
expect(stopped).toEqual(['video'])
|
|
expect(eye.isOpen).toBe(false)
|
|
})
|
|
|
|
it('opens normally when nobody interrupts', async () => {
|
|
const { stopped, release } = stubCamera()
|
|
const eye = new CameraEye()
|
|
const video = fakeVideo()
|
|
|
|
const opening = eye.open(video)
|
|
release()
|
|
await opening
|
|
|
|
expect(eye.isOpen).toBe(true)
|
|
expect(stopped).toEqual([])
|
|
// iOS Safari refuses to start a stream without both of these.
|
|
expect(video.muted).toBe(true)
|
|
expect(video.playsInline).toBe(true)
|
|
|
|
eye.close()
|
|
expect(stopped).toEqual(['video'])
|
|
expect(eye.isOpen).toBe(false)
|
|
})
|
|
|
|
it('captures nothing when the lens is closed', () => {
|
|
const eye = new CameraEye()
|
|
expect(eye.capture(fakeVideo())).toBeNull()
|
|
})
|
|
|
|
it('captures nothing before video metadata arrives', async () => {
|
|
const { release } = stubCamera()
|
|
const eye = new CameraEye()
|
|
const video = fakeVideo(0, 0) // dimensions not known yet
|
|
const opening = eye.open(video)
|
|
release()
|
|
await opening
|
|
expect(eye.capture(video)).toBeNull()
|
|
})
|
|
|
|
it('downscales a large frame so the payload stays small', async () => {
|
|
const { release } = stubCamera()
|
|
const eye = new CameraEye()
|
|
const video = fakeVideo(1920, 1080)
|
|
|
|
// Record what the canvas was sized to.
|
|
const sizes: Array<{ w: number; h: number }> = []
|
|
const canvas = {
|
|
width: 0,
|
|
height: 0,
|
|
getContext: () => ({ drawImage: () => undefined }),
|
|
toDataURL: () => {
|
|
sizes.push({ w: canvas.width, h: canvas.height })
|
|
return 'data:image/jpeg;base64,ZZZZ'
|
|
},
|
|
}
|
|
vi.stubGlobal('document', { createElement: () => canvas })
|
|
|
|
const opening = eye.open(video)
|
|
release()
|
|
await opening
|
|
|
|
expect(eye.capture(video)).toBe('data:image/jpeg;base64,ZZZZ')
|
|
expect(sizes).toHaveLength(1)
|
|
// Longest edge clamped, aspect ratio preserved.
|
|
expect(sizes[0].w).toBe(CAPTURE_MAX_EDGE)
|
|
expect(sizes[0].h).toBe(Math.round((1080 / 1920) * CAPTURE_MAX_EDGE))
|
|
})
|
|
|
|
it('never upscales a frame smaller than the cap', async () => {
|
|
const { release } = stubCamera()
|
|
const eye = new CameraEye()
|
|
const video = fakeVideo(320, 240)
|
|
const canvas = {
|
|
width: 0,
|
|
height: 0,
|
|
getContext: () => ({ drawImage: () => undefined }),
|
|
toDataURL: () => 'data:image/jpeg;base64,S',
|
|
}
|
|
vi.stubGlobal('document', { createElement: () => canvas })
|
|
|
|
const opening = eye.open(video)
|
|
release()
|
|
await opening
|
|
eye.capture(video)
|
|
|
|
expect(canvas.width).toBe(320)
|
|
expect(canvas.height).toBe(240)
|
|
})
|
|
})
|