diff --git a/backend/app/config.py b/backend/app/config.py index 5112374..6c3ed53 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -21,5 +21,12 @@ class Settings(BaseSettings): # Where synthesized utterance audio is written (served at /audio/). data_dir: str = "data" + # Canonical public origin, used for absolute URLs that must be correct + # for machines rather than browsers: sitemap entries, the + # robots.txt Sitemap line, and rel=canonical. Cannot be derived from the + # request — behind the Cloudflare Tunnel the app sees an internal host, + # and advertising that to a crawler would publish unreachable URLs. + public_base_url: str = "https://spirit.thetempleofdoom.com" + settings = Settings() diff --git a/backend/app/llm/client.py b/backend/app/llm/client.py index 9bdd215..9b7e62d 100644 --- a/backend/app/llm/client.py +++ b/backend/app/llm/client.py @@ -15,12 +15,19 @@ class OllamaClient: prompt: str, system: str | None = None, options: dict | None = None, + images: list[str] | None = None, ) -> str: + """`images` is a list of raw base64 JPEG/PNG strings (no data-URL + prefix) for vision-capable models — Ollama's /api/generate takes them + alongside the prompt. Ignored by text-only models, so passing them is + safe; the caller is responsible for choosing a model that can see.""" payload: dict = {"model": model, "prompt": prompt, "stream": False} if system is not None: payload["system"] = system if options: payload["options"] = options + if images: + payload["images"] = images async with httpx.AsyncClient(base_url=self._base_url, timeout=120.0) as http_client: response = await http_client.post("/api/generate", json=payload) diff --git a/backend/app/llm/prompts.py b/backend/app/llm/prompts.py index b3a7492..e522dad 100644 --- a/backend/app/llm/prompts.py +++ b/backend/app/llm/prompts.py @@ -63,6 +63,30 @@ MANIFEST_PROMPT = """The room right now: Something shifted. Speak.""" +SCRY_SYSTEM = ( + "You are {name}, {epithet} — a spirit persona in an interactive horror " + "art installation.\n" + "Your nature: {persona}\n" + "The seeker has turned a lens toward the room and you can see through " + "it. Speak about what is ACTUALLY in the image — the real objects, the " + "real light, the real room. Name one or two specific things you see, " + "plainly enough that the seeker knows you are truly looking.\n" + "Then let one of them mean something to you: mistake an object for one " + "you owned, recognise a shape, notice what is missing, or refuse to look " + "at a particular corner. The unsettling part is accuracy followed by " + "wrongness — not vagueness.\n" + "Never describe a person's face or body, and never guess at anyone's " + "identity, age or appearance; if a person is present, speak only of " + "their presence. Under 40 words. Never break character, never mention " + "being an AI, never mention images, cameras or models." + "{language_clause}" +) + +SCRY_PROMPT = """This is what the lens shows you right now. + +Speak.""" + + MINT_SYSTEM = ( "You invent spirit personas for an interactive horror art installation. " "The single biggest thing separating a convincing dead person from a " @@ -197,6 +221,15 @@ def manifest_prompt(readings: dict) -> str: return MANIFEST_PROMPT.format(readings="\n".join(lines)) +def scry_system(entity: dict, language: str = "en") -> str: + return SCRY_SYSTEM.format( + name=entity.get("name", "an unnamed presence"), + epithet=entity.get("epithet", "a voice in the static"), + persona=entity.get("persona", "A drifting presence with no remembered past."), + language_clause=language_clause(language), + ) + + def mint_prompt( signature: str, channel: str, diff --git a/backend/app/llm/service.py b/backend/app/llm/service.py index 8c29ddb..a14a8c8 100644 --- a/backend/app/llm/service.py +++ b/backend/app/llm/service.py @@ -185,6 +185,43 @@ class SpiritService: self._touch() return raw.strip().strip('"')[:200] + async def scry( + self, + entity: dict, + image_b64: str, + language: str = "en", + entropy: object = None, + ) -> str: + """The entity speaks about what the seeker's camera actually shows. + + The configured chat model (minicpm-v4.5) is vision-capable, so this + is a genuine look at the real room rather than an invented + description — the same principle as every other channel here: real + measurement first, interpretation second. + + Seeded from physical entropy like manifest(), so two identical rooms + still produce different speech. + """ + seed = int.from_bytes(veil_seed(entropy, "scry")[:8], "big") % (2**63) + + async def call() -> str: + return await self._client.generate( + settings.ollama_chat_model, + prompts.SCRY_PROMPT, + system=prompts.scry_system(entity, language), + options={ + "num_predict": 90, + "temperature": 0.95, + "top_p": 0.95, + "seed": seed, + }, + images=[image_b64], + ) + + raw = await self._queue.submit(call) + self._touch() + return raw.strip().strip('"')[:400] + async def mint_profile( self, signature: str, diff --git a/backend/app/main.py b/backend/app/main.py index fc9dc8c..a64957a 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -16,6 +16,7 @@ from app.routes.conditions import router as conditions_router from app.routes.device import router as device_router from app.routes.inventory import router as inventory_router from app.routes.seances import router as seances_router +from app.routes.seo import router as seo_router from app.routes.shop import router as shop_router from app.session_cleanup import delete_expired_sessions from app.ws import AUDIO_DIR @@ -100,6 +101,9 @@ app.include_router(conditions_router) app.include_router(device_router) app.include_router(inventory_router) app.include_router(seances_router) +# Registered before the SPA catch-all below, or /robots.txt and +# /sitemap.xml would be served index.html instead. +app.include_router(seo_router) app.include_router(shop_router) app.include_router(ws_router) diff --git a/backend/app/routes/seo.py b/backend/app/routes/seo.py new file mode 100644 index 0000000..8037735 --- /dev/null +++ b/backend/app/routes/seo.py @@ -0,0 +1,105 @@ +"""robots.txt and a live sitemap. + +Served from the backend rather than dropped in `public/` because the +valuable, indexable surface of this site is the Codex, and the Codex grows +every time somebody summons. A static sitemap would be stale within an +hour; this one is generated from the actual entity rows. + +What is deliberately NOT listed: /seance, /enter, /profile, /log, +/inventory, /devices — anything per-seeker or interactive. A crawler +hitting /seance would provision a guest account on arrival (the open +door), which would fill the users table with wanderers that never +existed as people. robots.txt disallows those paths for the same reason, +and the sitemap only advertises pages that are genuinely public, +stable, and worth a search result: the landing page, the shop, and every +discovered spirit. +""" + +from datetime import datetime, timezone + +from fastapi import APIRouter, Depends, Response +from sqlalchemy import select +from sqlalchemy.ext.asyncio import AsyncSession + +from app.config import settings +from app.db import get_db +from app.models.entity import Entity + +router = APIRouter(tags=["seo"]) + +# Cap the sitemap so a runaway Codex can't produce a multi-megabyte +# document. 5k is far inside the 50k/50MB sitemap limit and orders of +# magnitude beyond what this install will realistically hold. +MAX_SITEMAP_ENTITIES = 5000 + +# Paths that must never be crawled: they either mutate state (provisioning a +# guest) or are meaningless without a session. +DISALLOWED = ( + "/seance", + "/enter", + "/profile", + "/hunters", + "/log", + "/inventory", + "/devices", + "/api/", +) + + +def _base_url() -> str: + return settings.public_base_url.rstrip("/") + + +def _iso(dt: datetime | None) -> str: + value = dt or datetime.now(timezone.utc) + if value.tzinfo is None: + value = value.replace(tzinfo=timezone.utc) + return value.date().isoformat() + + +def _xml_escape(text: str) -> str: + return ( + text.replace("&", "&") + .replace("<", "<") + .replace(">", ">") + .replace('"', """) + ) + + +@router.get("/robots.txt", include_in_schema=False) +async def robots() -> Response: + lines = ["User-agent: *"] + lines += [f"Disallow: {path}" for path in DISALLOWED] + lines.append("Allow: /") + lines.append(f"Sitemap: {_base_url()}/sitemap.xml") + return Response("\n".join(lines) + "\n", media_type="text/plain") + + +@router.get("/sitemap.xml", include_in_schema=False) +async def sitemap(db: AsyncSession = Depends(get_db)) -> Response: + base = _base_url() + result = await db.execute( + select(Entity.id, Entity.discovered_at) + .order_by(Entity.discovered_at.desc()) + .limit(MAX_SITEMAP_ENTITIES) + ) + entities = result.all() + + parts = [ + '', + '', + f"{base}/daily" + "1.0", + f"{base}/codexhourly" + "0.9", + f"{base}/shopweekly" + "0.7", + ] + for entity_id, discovered_at in entities: + loc = _xml_escape(f"{base}/codex/{entity_id}") + parts.append( + f"{loc}{_iso(discovered_at)}" + "weekly0.6" + ) + parts.append("") + return Response("\n".join(parts), media_type="application/xml") diff --git a/backend/tests/conftest.py b/backend/tests/conftest.py index 62e00b1..54b1143 100644 --- a/backend/tests/conftest.py +++ b/backend/tests/conftest.py @@ -1,3 +1,5 @@ +import os + import pytest import pytest_asyncio from fastapi.testclient import TestClient @@ -9,12 +11,50 @@ from app.config import settings from app.db import Base, get_db from app.main import app +def _test_database_url() -> str: + """A SEPARATE database from the configured one. + + This is not a nicety. The autouse fixture below drop_all()s every table + before each test, so pointing it at settings.database_url means running + the suite silently annihilates the live install — every account, every + discovered spirit, every Ghost Log. That is exactly what used to happen: + a full test run left the production Codex at zero entities. + + Honouring TEST_DATABASE_URL when set (for CI), and otherwise deriving + `_test`, so the suite can never touch real data even if + somebody runs it on the production host — which is the normal case here, + since this box both serves the app and holds the repo. + """ + explicit = os.environ.get("TEST_DATABASE_URL") + if explicit: + return explicit + base, _, name = settings.database_url.rpartition("/") + if not name: + raise RuntimeError( + "cannot derive a test database from DATABASE_URL; " + "set TEST_DATABASE_URL explicitly" + ) + # Strip any query string (e.g. ?ssl=require) before suffixing the name. + db_name, sep, query = name.partition("?") + return f"{base}/{db_name}_test{sep}{query}" + + +TEST_DATABASE_URL = _test_database_url() + +# Fail loudly rather than eating the live data if the guard above is ever +# defeated by an unusual URL shape. +if TEST_DATABASE_URL == settings.database_url: + raise RuntimeError( + "refusing to run: the test database resolved to the production " + "database, and the suite drops every table" + ) + # pytest-asyncio gives each test function its own event loop by default; # asyncpg connections are bound to the loop they were opened on, so a pooled # connection from one test's loop breaks the next test. NullPool sidesteps # this by opening a fresh connection per checkout — scoped to this dedicated # test engine so app.db.engine (used by production) is unaffected. -test_engine = create_async_engine(settings.database_url, poolclass=NullPool) +test_engine = create_async_engine(TEST_DATABASE_URL, poolclass=NullPool) TestSessionLocal = async_sessionmaker(test_engine, expire_on_commit=False) diff --git a/backend/tests/test_seo.py b/backend/tests/test_seo.py new file mode 100644 index 0000000..a97fd38 --- /dev/null +++ b/backend/tests/test_seo.py @@ -0,0 +1,92 @@ +"""robots.txt and sitemap.xml. + +The load-bearing property is that crawlers are kept OFF the interactive +routes. /seance provisions a guest account on arrival (the open door), so a +crawler wandering in would create real user rows for visitors who never +existed — these tests pin that it stays disallowed and unlisted. +""" + +import xml.etree.ElementTree as ET + +import pytest + +from app.routes.seo import DISALLOWED + +SITEMAP_NS = {"sm": "http://www.sitemaps.org/schemas/sitemap/0.9"} + + +@pytest.mark.asyncio +async def test_robots_is_plain_text_and_not_the_spa(client): + response = await client.get("/robots.txt") + assert response.status_code == 200 + assert response.headers["content-type"].startswith("text/plain") + # If the SPA catch-all had won the route we'd get HTML instead. + assert " element: the caller supplies one so + * React keeps control of the DOM. This class only manages the stream and + * the canvas used for capture. + */ +export class CameraEye { + private stream: MediaStream | null = null + private canvas: HTMLCanvasElement | null = null + /** Bumped by open() and close(); lets an open() suspended on the + * permission prompt detect that it was abandoned. Same hazard the EVP + * listener had: the stream is assigned only after the await, so a close() + * during the prompt would release nothing and the camera would go live + * *after* teardown, leaving the recording light on. */ + private generation = 0 + + get isOpen(): boolean { + return this.stream !== null + } + + /** Requests the camera and attaches it to `video`. Throws on refusal; + * use classifyFailure() on the error. */ + async open(video: HTMLVideoElement, facing: 'user' | 'environment' = 'environment'): Promise { + if (this.stream) return + if (!isSupported()) throw new DOMException('no camera api', 'SecurityError') + + const generation = ++this.generation + const stream = await navigator.mediaDevices.getUserMedia({ + // `ideal` rather than `exact`: a laptop has no environment camera, and + // an exact constraint would fail outright instead of falling back to + // the only lens available. + video: { facingMode: { ideal: facing } }, + audio: false, + }) + if (generation !== this.generation) { + stream.getTracks().forEach((t) => t.stop()) + return + } + + this.stream = stream + video.srcObject = stream + // iOS Safari will not start a stream without this combination, and + // refuses to autoplay with sound even though we requested none. + video.muted = true + video.playsInline = true + await video.play().catch(() => undefined) + } + + /** + * Grabs one frame as a JPEG data URL, or null if the stream isn't ready. + * + * Only ever called from an explicit seeker action — see the module note. + */ + capture(video: HTMLVideoElement): string | null { + if (!this.stream) return null + const w = video.videoWidth + const h = video.videoHeight + if (!w || !h) return null // metadata hasn't arrived yet + + const scale = Math.min(1, CAPTURE_MAX_EDGE / Math.max(w, h)) + const canvas = (this.canvas ??= document.createElement('canvas')) + canvas.width = Math.max(1, Math.round(w * scale)) + canvas.height = Math.max(1, Math.round(h * scale)) + const ctx = canvas.getContext('2d') + if (!ctx) return null + ctx.drawImage(video, 0, 0, canvas.width, canvas.height) + try { + return canvas.toDataURL('image/jpeg', CAPTURE_QUALITY) + } catch { + // Tainted canvas shouldn't be possible for a same-origin camera + // stream, but a failed capture must not take the séance down. + return null + } + } + + close(): void { + this.generation++ + this.stream?.getTracks().forEach((t) => t.stop()) + this.stream = null + } +} + +/** Strips the `data:image/jpeg;base64,` prefix — Ollama wants raw base64. */ +export function toBase64(dataUrl: string): string { + const comma = dataUrl.indexOf(',') + return comma === -1 ? dataUrl : dataUrl.slice(comma + 1) +} diff --git a/frontend/src/lib/pageMeta.ts b/frontend/src/lib/pageMeta.ts new file mode 100644 index 0000000..244093d --- /dev/null +++ b/frontend/src/lib/pageMeta.ts @@ -0,0 +1,111 @@ +// Per-route document metadata. +// +// This is a client-rendered SPA, so every route ships the same static +// , description and canonical from index.html. That is fine for +// humans and bad for search: the genuinely valuable, unique content here is +// the Codex — every spirit is a distinct, LLM-written biography — and a +// crawler that renders the page still sees one shared title for all of +// them, so they compete with each other instead of ranking. +// +// Google does execute JS and picks up title/meta/canonical mutated after +// load, so patching the head per route is enough to make each entity page +// its own result. This deliberately stays a tiny imperative helper rather +// than pulling in react-helmet: three tags, no provider, no dependency. +// +// Not a substitute for SSR. If indexing ever needs to be bulletproof +// (crawlers that don't run JS, richer previews), the honest fix is +// prerendering /codex/:id server-side — this closes most of the gap for a +// fraction of the work. + +import { useEffect } from 'react' + +const SITE_NAME = 'Quantumancy' + +/** Must match backend Settings.public_base_url — absolute canonical URLs + * have to point at the public origin, not whatever host served the app. */ +export const PUBLIC_ORIGIN = 'https://spirit.thetempleofdoom.com' + +function setMeta(selector: string, attr: string, value: string): void { + let el = document.head.querySelector<HTMLMetaElement>(selector) + if (!el) { + el = document.createElement('meta') + const [, name] = selector.match(/\[(?:name|property)="([^"]+)"\]/) ?? [] + if (!name) return + el.setAttribute(selector.includes('property=') ? 'property' : 'name', name) + document.head.appendChild(el) + } + el.setAttribute(attr, value) +} + +function setCanonical(href: string): void { + let link = document.head.querySelector<HTMLLinkElement>('link[rel="canonical"]') + if (!link) { + link = document.createElement('link') + link.rel = 'canonical' + document.head.appendChild(link) + } + link.href = href +} + +export type PageMeta = { + /** Page-specific part of the title; SITE_NAME is appended. Omit on the + * landing page, which owns the bare brand title. */ + title?: string + description?: string + /** Path only, e.g. "/codex/abc". Resolved against PUBLIC_ORIGIN. */ + path?: string + /** JSON-LD to publish for this page, if any. */ + structuredData?: Record<string, unknown> +} + +const STRUCTURED_DATA_ID = 'qm-structured-data' + +function setStructuredData(data: Record<string, unknown> | undefined): void { + const existing = document.getElementById(STRUCTURED_DATA_ID) + if (!data) { + existing?.remove() + return + } + const script = + (existing as HTMLScriptElement | null) ?? document.createElement('script') + script.id = STRUCTURED_DATA_ID + script.setAttribute('type', 'application/ld+json') + script.textContent = JSON.stringify(data) + if (!existing) document.head.appendChild(script) +} + +/** + * Applies page metadata for as long as the component is mounted. + * + * Nothing is restored on unmount: the next route's own hook overwrites it, + * and restoring would briefly flash the previous page's title during + * navigation. Values are stringified defensively because entity personas + * come from an LLM and can contain anything. + */ +export function usePageMeta({ title, description, path, structuredData }: PageMeta): void { + // Callers build structuredData inline, so it is a fresh object on every + // render. Depending on it directly would re-run this effect (and rewrite + // the head) on every render; serializing gives it value semantics. + const structuredKey = structuredData ? JSON.stringify(structuredData) : '' + + useEffect(() => { + document.title = title ? `${title} · ${SITE_NAME}` : SITE_NAME + if (description) { + const clean = description.replace(/\s+/g, ' ').trim().slice(0, 300) + setMeta('meta[name="description"]', 'content', clean) + setMeta('meta[property="og:description"]', 'content', clean) + setMeta('meta[name="twitter:description"]', 'content', clean) + } + const ogTitle = title ? `${title} · ${SITE_NAME}` : SITE_NAME + setMeta('meta[property="og:title"]', 'content', ogTitle) + setMeta('meta[name="twitter:title"]', 'content', ogTitle) + if (path) { + const url = `${PUBLIC_ORIGIN}${path}` + setCanonical(url) + setMeta('meta[property="og:url"]', 'content', url) + } + setStructuredData(structuredData) + // structuredKey stands in for structuredData; see above. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [title, description, path, structuredKey]) +} diff --git a/frontend/src/pages/CodexEntityPage.tsx b/frontend/src/pages/CodexEntityPage.tsx index 38354f3..be871bd 100644 --- a/frontend/src/pages/CodexEntityPage.tsx +++ b/frontend/src/pages/CodexEntityPage.tsx @@ -4,6 +4,7 @@ import { useEffect, useState, type CSSProperties } from 'react' import { Link, useParams } from 'react-router-dom' import { useTranslation } from 'react-i18next' +import { PUBLIC_ORIGIN, usePageMeta } from '../lib/pageMeta' import type { CodexEntityDetail } from '../lib/types' import './CodexEntityPage.css' @@ -38,6 +39,30 @@ export function CodexEntityPage() { const { id } = useParams<{ id: string }>() const [state, setState] = useState<DetailState>({ status: 'loading' }) + // Every spirit is a distinct, unique page — give each one its own title, + // description and canonical URL so they rank individually instead of all + // sharing the app's single static title. The persona doubles as the + // description: it is the actual unique prose a searcher would be looking + // for. Marked up as a fictional Person so previews and rich results can + // never imply this is a record of a real dead human. + const detail = state.status === 'ready' ? state.entity : null + usePageMeta({ + title: detail ? `${detail.name} — ${detail.epithet}` : undefined, + description: detail?.persona, + path: id ? `/codex/${id}` : undefined, + structuredData: detail + ? { + '@context': 'https://schema.org', + '@type': 'Person', + name: detail.name, + alternateName: detail.epithet, + description: detail.persona, + additionalType: 'https://schema.org/FictionalCharacter', + url: `${PUBLIC_ORIGIN}/codex/${id}`, + } + : undefined, + }) + useEffect(() => { if (!id) { setState({ status: 'notfound' }) diff --git a/frontend/src/pages/CodexPage.tsx b/frontend/src/pages/CodexPage.tsx index 01075b9..40db548 100644 --- a/frontend/src/pages/CodexPage.tsx +++ b/frontend/src/pages/CodexPage.tsx @@ -4,6 +4,7 @@ import { useEffect, useState, type CSSProperties } from 'react' import { Link } from 'react-router-dom' import { useTranslation } from 'react-i18next' +import { usePageMeta } from '../lib/pageMeta' import type { CodexEntity, CodexListResponse, Rarity, SortOrder } from '../lib/types' import './CodexPage.css' @@ -25,6 +26,12 @@ function formatDate(iso: string, locale: string): string { export function CodexPage() { const { t, i18n } = useTranslation() + usePageMeta({ + title: 'The Codex of Contacted Spirits', + description: + 'Every spirit ever reached through Quantumancy — names, epithets, and the lives they left unfinished, recorded as each was contacted.', + path: '/codex', + }) const [rarity, setRarity] = useState<RarityFilter>('all') const [sort, setSort] = useState<SortOrder>('recent') const [entities, setEntities] = useState<CodexEntity[]>([]) diff --git a/frontend/src/pages/LandingPage.tsx b/frontend/src/pages/LandingPage.tsx index 7c3e874..4af7aff 100644 --- a/frontend/src/pages/LandingPage.tsx +++ b/frontend/src/pages/LandingPage.tsx @@ -8,6 +8,7 @@ import { useTranslation } from 'react-i18next' import { useAuth } from '../state/auth' import { GhostGlyph } from '../components/GhostGlyph' import type { CodexEntity, CodexListResponse, Mode, Stats } from '../lib/types' +import { PUBLIC_ORIGIN, usePageMeta } from '../lib/pageMeta' import './LandingPage.css' const STATS_REFRESH_MS = 15_000 @@ -192,6 +193,24 @@ export function LandingPage() { const { user } = useAuth() const navigate = useNavigate() + // The landing page keeps the bare brand title (no suffix) but still needs + // a canonical URL, and WebApplication markup so a search result can show + // what this actually is rather than guessing from the copy. + usePageMeta({ + path: '/', + structuredData: { + '@context': 'https://schema.org', + '@type': 'WebApplication', + name: 'Quantumancy', + url: PUBLIC_ORIGIN, + applicationCategory: 'EntertainmentApplication', + operatingSystem: 'Any modern browser', + description: + "A self-hosted séance. Talk to spirits through your microphone, an RTL-SDR dongle, your network's jitter, your phone's motion sensors, or a paired ESP32 device — a locally-run LLM gives each real anomaly a voice.", + offers: { '@type': 'Offer', price: '0', priceCurrency: 'USD' }, + }, + }) + const [stats, setStats] = useState<Stats | null>(null) const [featured, setFeatured] = useState<CodexEntity[] | null>(null) const [taglineIdx, setTaglineIdx] = useState(0)