From 16c8bae00d16418b8d7ec26aa238846e3a0e50b6 Mon Sep 17 00:00:00 2001 From: Indiana Date: Thu, 30 Jul 2026 01:44:01 +0000 Subject: [PATCH] fix: the test suite was destroying the production database MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The worst bug of the session. tests/conftest.py built its engine from settings.database_url — the live database — and an autouse fixture calls drop_all() before EVERY test. So every backend run silently annihilated the real install: accounts, discovered spirits, Ghost Logs, devices, all of it. Found it because /sitemap.xml listed zero entities minutes after I had watched live séances mint real ones. Tests now use TEST_DATABASE_URL, or `_test` derived from it, and refuse to start at all if that ever resolves back to the production URL — this box both serves the app and holds the repo, so "don't run tests in prod" is not a workable guard. Proven: inserted a canary row into production, ran 50 tests, canary survived. Before this it would have been dropped. Also in this commit: SEO (routes/seo.py, lib/pageMeta.ts) - Live /sitemap.xml generated from real entity rows, and /robots.txt, both registered BEFORE the SPA catch-all or they'd be served index.html. Crawlers are disallowed from /seance specifically because the open door provisions a guest on arrival — a crawler would fill the users table with wanderers who never existed. - Per-route , description, canonical and JSON-LD. The Codex is the indexable asset here (every spirit is unique long-form prose) and all of it previously shared one static title, so entities competed with each other instead of ranking. Entities are marked up as fictional Persons so a rich result can never imply a record of a real dead human. - public_base_url setting: absolute URLs for crawlers can't be derived from the request, since behind the tunnel the app only sees an internal host. Camera channel, first half (lib/camera.ts, llm scry path) - OllamaClient.generate() now accepts `images`; the configured chat model (minicpm-v4.5:8b) is vision-capable, so the entity can speak about what the seeker's camera actually shows. Verified against a synthetic room image: it named the pale column and the small red cube, then misread them as oak in a farmhouse parlor — real perception, in character. - Frames are captured only on an explicit act, downscaled to 768px and JPEG-compressed, never stored, and the prompt forbids describing faces or guessing identity. CameraEye carries the same generation guard as the EVP listener so closing during the permission prompt can't leave the camera live after teardown. 338 backend tests pass; 375 frontend; i18n parity holds. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com> --- backend/app/config.py | 7 ++ backend/app/llm/client.py | 7 ++ backend/app/llm/prompts.py | 33 ++++++ backend/app/llm/service.py | 37 +++++++ backend/app/main.py | 4 + backend/app/routes/seo.py | 105 ++++++++++++++++++ backend/tests/conftest.py | 42 ++++++- backend/tests/test_seo.py | 92 ++++++++++++++++ frontend/src/lib/camera.ts | 145 +++++++++++++++++++++++++ frontend/src/lib/pageMeta.ts | 111 +++++++++++++++++++ frontend/src/pages/CodexEntityPage.tsx | 25 +++++ frontend/src/pages/CodexPage.tsx | 7 ++ frontend/src/pages/LandingPage.tsx | 19 ++++ 13 files changed, 633 insertions(+), 1 deletion(-) create mode 100644 backend/app/routes/seo.py create mode 100644 backend/tests/test_seo.py create mode 100644 frontend/src/lib/camera.ts create mode 100644 frontend/src/lib/pageMeta.ts diff --git a/backend/app/config.py b/backend/app/config.py index 5112374..6c3ed53 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -21,5 +21,12 @@ class Settings(BaseSettings): # Where synthesized utterance audio is written (served at /audio/). data_dir: str = "data" + # Canonical public origin, used for absolute URLs that must be correct + # for machines rather than browsers: sitemap <loc> entries, the + # robots.txt Sitemap line, and rel=canonical. Cannot be derived from the + # request — behind the Cloudflare Tunnel the app sees an internal host, + # and advertising that to a crawler would publish unreachable URLs. + public_base_url: str = "https://spirit.thetempleofdoom.com" + settings = Settings() diff --git a/backend/app/llm/client.py b/backend/app/llm/client.py index 9bdd215..9b7e62d 100644 --- a/backend/app/llm/client.py +++ b/backend/app/llm/client.py @@ -15,12 +15,19 @@ class OllamaClient: prompt: str, system: str | None = None, options: dict | None = None, + images: list[str] | None = None, ) -> str: + """`images` is a list of raw base64 JPEG/PNG strings (no data-URL + prefix) for vision-capable models — Ollama's /api/generate takes them + alongside the prompt. Ignored by text-only models, so passing them is + safe; the caller is responsible for choosing a model that can see.""" payload: dict = {"model": model, "prompt": prompt, "stream": False} if system is not None: payload["system"] = system if options: payload["options"] = options + if images: + payload["images"] = images async with httpx.AsyncClient(base_url=self._base_url, timeout=120.0) as http_client: response = await http_client.post("/api/generate", json=payload) diff --git a/backend/app/llm/prompts.py b/backend/app/llm/prompts.py index b3a7492..e522dad 100644 --- a/backend/app/llm/prompts.py +++ b/backend/app/llm/prompts.py @@ -63,6 +63,30 @@ MANIFEST_PROMPT = """The room right now: Something shifted. Speak.""" +SCRY_SYSTEM = ( + "You are {name}, {epithet} — a spirit persona in an interactive horror " + "art installation.\n" + "Your nature: {persona}\n" + "The seeker has turned a lens toward the room and you can see through " + "it. Speak about what is ACTUALLY in the image — the real objects, the " + "real light, the real room. Name one or two specific things you see, " + "plainly enough that the seeker knows you are truly looking.\n" + "Then let one of them mean something to you: mistake an object for one " + "you owned, recognise a shape, notice what is missing, or refuse to look " + "at a particular corner. The unsettling part is accuracy followed by " + "wrongness — not vagueness.\n" + "Never describe a person's face or body, and never guess at anyone's " + "identity, age or appearance; if a person is present, speak only of " + "their presence. Under 40 words. Never break character, never mention " + "being an AI, never mention images, cameras or models." + "{language_clause}" +) + +SCRY_PROMPT = """This is what the lens shows you right now. + +Speak.""" + + MINT_SYSTEM = ( "You invent spirit personas for an interactive horror art installation. " "The single biggest thing separating a convincing dead person from a " @@ -197,6 +221,15 @@ def manifest_prompt(readings: dict) -> str: return MANIFEST_PROMPT.format(readings="\n".join(lines)) +def scry_system(entity: dict, language: str = "en") -> str: + return SCRY_SYSTEM.format( + name=entity.get("name", "an unnamed presence"), + epithet=entity.get("epithet", "a voice in the static"), + persona=entity.get("persona", "A drifting presence with no remembered past."), + language_clause=language_clause(language), + ) + + def mint_prompt( signature: str, channel: str, diff --git a/backend/app/llm/service.py b/backend/app/llm/service.py index 8c29ddb..a14a8c8 100644 --- a/backend/app/llm/service.py +++ b/backend/app/llm/service.py @@ -185,6 +185,43 @@ class SpiritService: self._touch() return raw.strip().strip('"')[:200] + async def scry( + self, + entity: dict, + image_b64: str, + language: str = "en", + entropy: object = None, + ) -> str: + """The entity speaks about what the seeker's camera actually shows. + + The configured chat model (minicpm-v4.5) is vision-capable, so this + is a genuine look at the real room rather than an invented + description — the same principle as every other channel here: real + measurement first, interpretation second. + + Seeded from physical entropy like manifest(), so two identical rooms + still produce different speech. + """ + seed = int.from_bytes(veil_seed(entropy, "scry")[:8], "big") % (2**63) + + async def call() -> str: + return await self._client.generate( + settings.ollama_chat_model, + prompts.SCRY_PROMPT, + system=prompts.scry_system(entity, language), + options={ + "num_predict": 90, + "temperature": 0.95, + "top_p": 0.95, + "seed": seed, + }, + images=[image_b64], + ) + + raw = await self._queue.submit(call) + self._touch() + return raw.strip().strip('"')[:400] + async def mint_profile( self, signature: str, diff --git a/backend/app/main.py b/backend/app/main.py index fc9dc8c..a64957a 100644 --- a/backend/app/main.py +++ b/backend/app/main.py @@ -16,6 +16,7 @@ from app.routes.conditions import router as conditions_router from app.routes.device import router as device_router from app.routes.inventory import router as inventory_router from app.routes.seances import router as seances_router +from app.routes.seo import router as seo_router from app.routes.shop import router as shop_router from app.session_cleanup import delete_expired_sessions from app.ws import AUDIO_DIR @@ -100,6 +101,9 @@ app.include_router(conditions_router) app.include_router(device_router) app.include_router(inventory_router) app.include_router(seances_router) +# Registered before the SPA catch-all below, or /robots.txt and +# /sitemap.xml would be served index.html instead. +app.include_router(seo_router) app.include_router(shop_router) app.include_router(ws_router) diff --git a/backend/app/routes/seo.py b/backend/app/routes/seo.py new file mode 100644 index 0000000..8037735 --- /dev/null +++ b/backend/app/routes/seo.py @@ -0,0 +1,105 @@ +"""robots.txt and a live sitemap. + +Served from the backend rather than dropped in `public/` because the +valuable, indexable surface of this site is the Codex, and the Codex grows +every time somebody summons. A static sitemap would be stale within an +hour; this one is generated from the actual entity rows. + +What is deliberately NOT listed: /seance, /enter, /profile, /log, +/inventory, /devices — anything per-seeker or interactive. A crawler +hitting /seance would provision a guest account on arrival (the open +door), which would fill the users table with wanderers that never +existed as people. robots.txt disallows those paths for the same reason, +and the sitemap only advertises pages that are genuinely public, +stable, and worth a search result: the landing page, the shop, and every +discovered spirit. +""" + +from datetime import datetime, timezone + +from fastapi import APIRouter, Depends, Response +from sqlalchemy import select +from sqlalchemy.ext.asyncio import AsyncSession + +from app.config import settings +from app.db import get_db +from app.models.entity import Entity + +router = APIRouter(tags=["seo"]) + +# Cap the sitemap so a runaway Codex can't produce a multi-megabyte +# document. 5k is far inside the 50k/50MB sitemap limit and orders of +# magnitude beyond what this install will realistically hold. +MAX_SITEMAP_ENTITIES = 5000 + +# Paths that must never be crawled: they either mutate state (provisioning a +# guest) or are meaningless without a session. +DISALLOWED = ( + "/seance", + "/enter", + "/profile", + "/hunters", + "/log", + "/inventory", + "/devices", + "/api/", +) + + +def _base_url() -> str: + return settings.public_base_url.rstrip("/") + + +def _iso(dt: datetime | None) -> str: + value = dt or datetime.now(timezone.utc) + if value.tzinfo is None: + value = value.replace(tzinfo=timezone.utc) + return value.date().isoformat() + + +def _xml_escape(text: str) -> str: + return ( + text.replace("&", "&") + .replace("<", "<") + .replace(">", ">") + .replace('"', """) + ) + + +@router.get("/robots.txt", include_in_schema=False) +async def robots() -> Response: + lines = ["User-agent: *"] + lines += [f"Disallow: {path}" for path in DISALLOWED] + lines.append("Allow: /") + lines.append(f"Sitemap: {_base_url()}/sitemap.xml") + return Response("\n".join(lines) + "\n", media_type="text/plain") + + +@router.get("/sitemap.xml", include_in_schema=False) +async def sitemap(db: AsyncSession = Depends(get_db)) -> Response: + base = _base_url() + result = await db.execute( + select(Entity.id, Entity.discovered_at) + .order_by(Entity.discovered_at.desc()) + .limit(MAX_SITEMAP_ENTITIES) + ) + entities = result.all() + + parts = [ + '<?xml version="1.0" encoding="UTF-8"?>', + '<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">', + f"<url><loc>{base}/</loc><changefreq>daily</changefreq>" + "<priority>1.0</priority></url>", + f"<url><loc>{base}/codex</loc><changefreq>hourly</changefreq>" + "<priority>0.9</priority></url>", + f"<url><loc>{base}/shop</loc><changefreq>weekly</changefreq>" + "<priority>0.7</priority></url>", + ] + for entity_id, discovered_at in entities: + loc = _xml_escape(f"{base}/codex/{entity_id}") + parts.append( + f"<url><loc>{loc}</loc><lastmod>{_iso(discovered_at)}</lastmod>" + "<changefreq>weekly</changefreq><priority>0.6</priority></url>" + ) + parts.append("</urlset>") + return Response("\n".join(parts), media_type="application/xml") diff --git a/backend/tests/conftest.py b/backend/tests/conftest.py index 62e00b1..54b1143 100644 --- a/backend/tests/conftest.py +++ b/backend/tests/conftest.py @@ -1,3 +1,5 @@ +import os + import pytest import pytest_asyncio from fastapi.testclient import TestClient @@ -9,12 +11,50 @@ from app.config import settings from app.db import Base, get_db from app.main import app +def _test_database_url() -> str: + """A SEPARATE database from the configured one. + + This is not a nicety. The autouse fixture below drop_all()s every table + before each test, so pointing it at settings.database_url means running + the suite silently annihilates the live install — every account, every + discovered spirit, every Ghost Log. That is exactly what used to happen: + a full test run left the production Codex at zero entities. + + Honouring TEST_DATABASE_URL when set (for CI), and otherwise deriving + `<configured-db>_test`, so the suite can never touch real data even if + somebody runs it on the production host — which is the normal case here, + since this box both serves the app and holds the repo. + """ + explicit = os.environ.get("TEST_DATABASE_URL") + if explicit: + return explicit + base, _, name = settings.database_url.rpartition("/") + if not name: + raise RuntimeError( + "cannot derive a test database from DATABASE_URL; " + "set TEST_DATABASE_URL explicitly" + ) + # Strip any query string (e.g. ?ssl=require) before suffixing the name. + db_name, sep, query = name.partition("?") + return f"{base}/{db_name}_test{sep}{query}" + + +TEST_DATABASE_URL = _test_database_url() + +# Fail loudly rather than eating the live data if the guard above is ever +# defeated by an unusual URL shape. +if TEST_DATABASE_URL == settings.database_url: + raise RuntimeError( + "refusing to run: the test database resolved to the production " + "database, and the suite drops every table" + ) + # pytest-asyncio gives each test function its own event loop by default; # asyncpg connections are bound to the loop they were opened on, so a pooled # connection from one test's loop breaks the next test. NullPool sidesteps # this by opening a fresh connection per checkout — scoped to this dedicated # test engine so app.db.engine (used by production) is unaffected. -test_engine = create_async_engine(settings.database_url, poolclass=NullPool) +test_engine = create_async_engine(TEST_DATABASE_URL, poolclass=NullPool) TestSessionLocal = async_sessionmaker(test_engine, expire_on_commit=False) diff --git a/backend/tests/test_seo.py b/backend/tests/test_seo.py new file mode 100644 index 0000000..a97fd38 --- /dev/null +++ b/backend/tests/test_seo.py @@ -0,0 +1,92 @@ +"""robots.txt and sitemap.xml. + +The load-bearing property is that crawlers are kept OFF the interactive +routes. /seance provisions a guest account on arrival (the open door), so a +crawler wandering in would create real user rows for visitors who never +existed — these tests pin that it stays disallowed and unlisted. +""" + +import xml.etree.ElementTree as ET + +import pytest + +from app.routes.seo import DISALLOWED + +SITEMAP_NS = {"sm": "http://www.sitemaps.org/schemas/sitemap/0.9"} + + +@pytest.mark.asyncio +async def test_robots_is_plain_text_and_not_the_spa(client): + response = await client.get("/robots.txt") + assert response.status_code == 200 + assert response.headers["content-type"].startswith("text/plain") + # If the SPA catch-all had won the route we'd get HTML instead. + assert "<!doctype html" not in response.text.lower() + assert response.text.startswith("User-agent: *") + + +@pytest.mark.asyncio +async def test_robots_disallows_every_interactive_route(client): + body = await client.get("/robots.txt") + text = body.text + for path in DISALLOWED: + assert f"Disallow: {path}" in text + # The séance is the critical one: crawling it mints guest accounts. + assert "Disallow: /seance" in text + + +@pytest.mark.asyncio +async def test_robots_advertises_an_absolute_sitemap_url(client): + text = (await client.get("/robots.txt")).text + line = next(ln for ln in text.splitlines() if ln.startswith("Sitemap:")) + url = line.split(" ", 1)[1] + # Must be absolute and public — behind the tunnel the app sees an + # internal host, and a relative or internal URL is useless to a crawler. + assert url.startswith("https://") + assert url.endswith("/sitemap.xml") + + +@pytest.mark.asyncio +async def test_sitemap_is_valid_xml_with_the_public_pages(client): + response = await client.get("/sitemap.xml") + assert response.status_code == 200 + assert "xml" in response.headers["content-type"] + root = ET.fromstring(response.text) + locs = [el.text for el in root.findall(".//sm:loc", SITEMAP_NS)] + assert any(loc.endswith("/") for loc in locs) + assert any(loc.endswith("/codex") for loc in locs) + + +@pytest.mark.asyncio +async def test_sitemap_never_lists_an_interactive_route(client): + root = ET.fromstring((await client.get("/sitemap.xml")).text) + locs = [el.text for el in root.findall(".//sm:loc", SITEMAP_NS)] + for loc in locs: + for path in DISALLOWED: + assert not loc.endswith(path.rstrip("/")), f"{loc} advertises {path}" + + +@pytest.mark.asyncio +async def test_sitemap_lists_discovered_spirits(client, db_session): + from app.models.entity import Entity + + entity = Entity( + name="Sitemap Test Spirit", + epithet="the Indexed", + persona="A spirit that exists to be crawled.", + signature="sitemap-sig-1", + ) + db_session.add(entity) + await db_session.commit() + await db_session.refresh(entity) + + root = ET.fromstring((await client.get("/sitemap.xml")).text) + locs = [el.text for el in root.findall(".//sm:loc", SITEMAP_NS)] + assert any(loc.endswith(f"/codex/{entity.id}") for loc in locs) + + +@pytest.mark.asyncio +async def test_sitemap_is_valid_with_an_empty_codex(client): + # A brand-new install must still serve a well-formed sitemap. + root = ET.fromstring((await client.get("/sitemap.xml")).text) + assert root.tag.endswith("urlset") diff --git a/frontend/src/lib/camera.ts b/frontend/src/lib/camera.ts new file mode 100644 index 0000000..a7ba3d7 --- /dev/null +++ b/frontend/src/lib/camera.ts @@ -0,0 +1,145 @@ +// The camera as a channel — letting the dead look through the seeker's lens. +// +// Why this is different from every other channel here: the microphone, the +// magnetometer, the RTL-SDR and the ESP32 all produce *numbers* that the +// app interprets. A camera produces a scene, and the chat model already +// configured for this install (minicpm-v4.5:8b) is a vision model — so the +// entity can be shown the room and speak about what is actually in it. +// Nothing is simulated: the frame is a real photograph of wherever the +// seeker is standing. +// +// PRIVACY, and why this module is shaped the way it is: +// - Frames are only ever captured on an explicit, deliberate act (the +// seeker pressing "let it look"). There is no timer, no background +// loop, and no way for this module to grab a frame on its own. +// - A frame is downscaled and JPEG-compressed before it leaves the +// browser, both to keep the request small and to strip incidental +// detail the model does not need. +// - Nothing is stored. The data URL is handed to the caller, sent once, +// and dropped; there is no cache and no history. +// - The live preview never leaves the page unless a frame is captured, +// so "camera on" and "the ghost saw something" stay separate states. +// +// Platform reality: getUserMedia needs a secure context (https or +// localhost). It works on iOS Safari, unlike WebUSB/Web Bluetooth — so this +// is one of the richest channels available on an iPhone. + +/** Longest edge of a captured frame, px. Vision models see plenty at this + * size, and it keeps a base64 payload well inside a normal request body — + * a full 1080p still would be megabytes of JSON. */ +export const CAPTURE_MAX_EDGE = 768 + +/** JPEG quality for captures. 0.72 keeps faces and text legible while + * roughly halving the payload versus 0.9. */ +export const CAPTURE_QUALITY = 0.72 + +export type CameraFailure = 'denied' | 'insecure' | 'absent' | 'busy' | 'unknown' + +export function isSupported(): boolean { + return ( + typeof navigator !== 'undefined' && + typeof navigator.mediaDevices?.getUserMedia === 'function' + ) +} + +/** Maps a getUserMedia rejection to a cause, so the UI can say something + * true instead of always blaming the seeker for refusing. Mirrors the + * classification the EVP panel uses. */ +export function classifyFailure(err: unknown): CameraFailure { + if (!isSupported()) return 'insecure' + const name = err instanceof DOMException ? err.name : '' + if (name === 'NotAllowedError' || name === 'PermissionDeniedError') return 'denied' + if (name === 'SecurityError') return 'insecure' + if (name === 'NotFoundError' || name === 'OverconstrainedError') return 'absent' + if (name === 'NotReadableError' || name === 'AbortError') return 'busy' + return 'unknown' +} + +/** + * Owns one camera stream and can capture single stills from it. + * + * Deliberately does NOT own a <video> element: the caller supplies one so + * React keeps control of the DOM. This class only manages the stream and + * the canvas used for capture. + */ +export class CameraEye { + private stream: MediaStream | null = null + private canvas: HTMLCanvasElement | null = null + /** Bumped by open() and close(); lets an open() suspended on the + * permission prompt detect that it was abandoned. Same hazard the EVP + * listener had: the stream is assigned only after the await, so a close() + * during the prompt would release nothing and the camera would go live + * *after* teardown, leaving the recording light on. */ + private generation = 0 + + get isOpen(): boolean { + return this.stream !== null + } + + /** Requests the camera and attaches it to `video`. Throws on refusal; + * use classifyFailure() on the error. */ + async open(video: HTMLVideoElement, facing: 'user' | 'environment' = 'environment'): Promise<void> { + if (this.stream) return + if (!isSupported()) throw new DOMException('no camera api', 'SecurityError') + + const generation = ++this.generation + const stream = await navigator.mediaDevices.getUserMedia({ + // `ideal` rather than `exact`: a laptop has no environment camera, and + // an exact constraint would fail outright instead of falling back to + // the only lens available. + video: { facingMode: { ideal: facing } }, + audio: false, + }) + if (generation !== this.generation) { + stream.getTracks().forEach((t) => t.stop()) + return + } + + this.stream = stream + video.srcObject = stream + // iOS Safari will not start a stream without this combination, and + // refuses to autoplay with sound even though we requested none. + video.muted = true + video.playsInline = true + await video.play().catch(() => undefined) + } + + /** + * Grabs one frame as a JPEG data URL, or null if the stream isn't ready. + * + * Only ever called from an explicit seeker action — see the module note. + */ + capture(video: HTMLVideoElement): string | null { + if (!this.stream) return null + const w = video.videoWidth + const h = video.videoHeight + if (!w || !h) return null // metadata hasn't arrived yet + + const scale = Math.min(1, CAPTURE_MAX_EDGE / Math.max(w, h)) + const canvas = (this.canvas ??= document.createElement('canvas')) + canvas.width = Math.max(1, Math.round(w * scale)) + canvas.height = Math.max(1, Math.round(h * scale)) + const ctx = canvas.getContext('2d') + if (!ctx) return null + ctx.drawImage(video, 0, 0, canvas.width, canvas.height) + try { + return canvas.toDataURL('image/jpeg', CAPTURE_QUALITY) + } catch { + // Tainted canvas shouldn't be possible for a same-origin camera + // stream, but a failed capture must not take the séance down. + return null + } + } + + close(): void { + this.generation++ + this.stream?.getTracks().forEach((t) => t.stop()) + this.stream = null + } +} + +/** Strips the `data:image/jpeg;base64,` prefix — Ollama wants raw base64. */ +export function toBase64(dataUrl: string): string { + const comma = dataUrl.indexOf(',') + return comma === -1 ? dataUrl : dataUrl.slice(comma + 1) +} diff --git a/frontend/src/lib/pageMeta.ts b/frontend/src/lib/pageMeta.ts new file mode 100644 index 0000000..244093d --- /dev/null +++ b/frontend/src/lib/pageMeta.ts @@ -0,0 +1,111 @@ +// Per-route document metadata. +// +// This is a client-rendered SPA, so every route ships the same static +// <title>, description and canonical from index.html. That is fine for +// humans and bad for search: the genuinely valuable, unique content here is +// the Codex — every spirit is a distinct, LLM-written biography — and a +// crawler that renders the page still sees one shared title for all of +// them, so they compete with each other instead of ranking. +// +// Google does execute JS and picks up title/meta/canonical mutated after +// load, so patching the head per route is enough to make each entity page +// its own result. This deliberately stays a tiny imperative helper rather +// than pulling in react-helmet: three tags, no provider, no dependency. +// +// Not a substitute for SSR. If indexing ever needs to be bulletproof +// (crawlers that don't run JS, richer previews), the honest fix is +// prerendering /codex/:id server-side — this closes most of the gap for a +// fraction of the work. + +import { useEffect } from 'react' + +const SITE_NAME = 'Quantumancy' + +/** Must match backend Settings.public_base_url — absolute canonical URLs + * have to point at the public origin, not whatever host served the app. */ +export const PUBLIC_ORIGIN = 'https://spirit.thetempleofdoom.com' + +function setMeta(selector: string, attr: string, value: string): void { + let el = document.head.querySelector<HTMLMetaElement>(selector) + if (!el) { + el = document.createElement('meta') + const [, name] = selector.match(/\[(?:name|property)="([^"]+)"\]/) ?? [] + if (!name) return + el.setAttribute(selector.includes('property=') ? 'property' : 'name', name) + document.head.appendChild(el) + } + el.setAttribute(attr, value) +} + +function setCanonical(href: string): void { + let link = document.head.querySelector<HTMLLinkElement>('link[rel="canonical"]') + if (!link) { + link = document.createElement('link') + link.rel = 'canonical' + document.head.appendChild(link) + } + link.href = href +} + +export type PageMeta = { + /** Page-specific part of the title; SITE_NAME is appended. Omit on the + * landing page, which owns the bare brand title. */ + title?: string + description?: string + /** Path only, e.g. "/codex/abc". Resolved against PUBLIC_ORIGIN. */ + path?: string + /** JSON-LD to publish for this page, if any. */ + structuredData?: Record<string, unknown> +} + +const STRUCTURED_DATA_ID = 'qm-structured-data' + +function setStructuredData(data: Record<string, unknown> | undefined): void { + const existing = document.getElementById(STRUCTURED_DATA_ID) + if (!data) { + existing?.remove() + return + } + const script = + (existing as HTMLScriptElement | null) ?? document.createElement('script') + script.id = STRUCTURED_DATA_ID + script.setAttribute('type', 'application/ld+json') + script.textContent = JSON.stringify(data) + if (!existing) document.head.appendChild(script) +} + +/** + * Applies page metadata for as long as the component is mounted. + * + * Nothing is restored on unmount: the next route's own hook overwrites it, + * and restoring would briefly flash the previous page's title during + * navigation. Values are stringified defensively because entity personas + * come from an LLM and can contain anything. + */ +export function usePageMeta({ title, description, path, structuredData }: PageMeta): void { + // Callers build structuredData inline, so it is a fresh object on every + // render. Depending on it directly would re-run this effect (and rewrite + // the head) on every render; serializing gives it value semantics. + const structuredKey = structuredData ? JSON.stringify(structuredData) : '' + + useEffect(() => { + document.title = title ? `${title} · ${SITE_NAME}` : SITE_NAME + if (description) { + const clean = description.replace(/\s+/g, ' ').trim().slice(0, 300) + setMeta('meta[name="description"]', 'content', clean) + setMeta('meta[property="og:description"]', 'content', clean) + setMeta('meta[name="twitter:description"]', 'content', clean) + } + const ogTitle = title ? `${title} · ${SITE_NAME}` : SITE_NAME + setMeta('meta[property="og:title"]', 'content', ogTitle) + setMeta('meta[name="twitter:title"]', 'content', ogTitle) + if (path) { + const url = `${PUBLIC_ORIGIN}${path}` + setCanonical(url) + setMeta('meta[property="og:url"]', 'content', url) + } + setStructuredData(structuredData) + // structuredKey stands in for structuredData; see above. + // eslint-disable-next-line react-hooks/exhaustive-deps + }, [title, description, path, structuredKey]) +} diff --git a/frontend/src/pages/CodexEntityPage.tsx b/frontend/src/pages/CodexEntityPage.tsx index 38354f3..be871bd 100644 --- a/frontend/src/pages/CodexEntityPage.tsx +++ b/frontend/src/pages/CodexEntityPage.tsx @@ -4,6 +4,7 @@ import { useEffect, useState, type CSSProperties } from 'react' import { Link, useParams } from 'react-router-dom' import { useTranslation } from 'react-i18next' +import { PUBLIC_ORIGIN, usePageMeta } from '../lib/pageMeta' import type { CodexEntityDetail } from '../lib/types' import './CodexEntityPage.css' @@ -38,6 +39,30 @@ export function CodexEntityPage() { const { id } = useParams<{ id: string }>() const [state, setState] = useState<DetailState>({ status: 'loading' }) + // Every spirit is a distinct, unique page — give each one its own title, + // description and canonical URL so they rank individually instead of all + // sharing the app's single static title. The persona doubles as the + // description: it is the actual unique prose a searcher would be looking + // for. Marked up as a fictional Person so previews and rich results can + // never imply this is a record of a real dead human. + const detail = state.status === 'ready' ? state.entity : null + usePageMeta({ + title: detail ? `${detail.name} — ${detail.epithet}` : undefined, + description: detail?.persona, + path: id ? `/codex/${id}` : undefined, + structuredData: detail + ? { + '@context': 'https://schema.org', + '@type': 'Person', + name: detail.name, + alternateName: detail.epithet, + description: detail.persona, + additionalType: 'https://schema.org/FictionalCharacter', + url: `${PUBLIC_ORIGIN}/codex/${id}`, + } + : undefined, + }) + useEffect(() => { if (!id) { setState({ status: 'notfound' }) diff --git a/frontend/src/pages/CodexPage.tsx b/frontend/src/pages/CodexPage.tsx index 01075b9..40db548 100644 --- a/frontend/src/pages/CodexPage.tsx +++ b/frontend/src/pages/CodexPage.tsx @@ -4,6 +4,7 @@ import { useEffect, useState, type CSSProperties } from 'react' import { Link } from 'react-router-dom' import { useTranslation } from 'react-i18next' +import { usePageMeta } from '../lib/pageMeta' import type { CodexEntity, CodexListResponse, Rarity, SortOrder } from '../lib/types' import './CodexPage.css' @@ -25,6 +26,12 @@ function formatDate(iso: string, locale: string): string { export function CodexPage() { const { t, i18n } = useTranslation() + usePageMeta({ + title: 'The Codex of Contacted Spirits', + description: + 'Every spirit ever reached through Quantumancy — names, epithets, and the lives they left unfinished, recorded as each was contacted.', + path: '/codex', + }) const [rarity, setRarity] = useState<RarityFilter>('all') const [sort, setSort] = useState<SortOrder>('recent') const [entities, setEntities] = useState<CodexEntity[]>([]) diff --git a/frontend/src/pages/LandingPage.tsx b/frontend/src/pages/LandingPage.tsx index 7c3e874..4af7aff 100644 --- a/frontend/src/pages/LandingPage.tsx +++ b/frontend/src/pages/LandingPage.tsx @@ -8,6 +8,7 @@ import { useTranslation } from 'react-i18next' import { useAuth } from '../state/auth' import { GhostGlyph } from '../components/GhostGlyph' import type { CodexEntity, CodexListResponse, Mode, Stats } from '../lib/types' +import { PUBLIC_ORIGIN, usePageMeta } from '../lib/pageMeta' import './LandingPage.css' const STATS_REFRESH_MS = 15_000 @@ -192,6 +193,24 @@ export function LandingPage() { const { user } = useAuth() const navigate = useNavigate() + // The landing page keeps the bare brand title (no suffix) but still needs + // a canonical URL, and WebApplication markup so a search result can show + // what this actually is rather than guessing from the copy. + usePageMeta({ + path: '/', + structuredData: { + '@context': 'https://schema.org', + '@type': 'WebApplication', + name: 'Quantumancy', + url: PUBLIC_ORIGIN, + applicationCategory: 'EntertainmentApplication', + operatingSystem: 'Any modern browser', + description: + "A self-hosted séance. Talk to spirits through your microphone, an RTL-SDR dongle, your network's jitter, your phone's motion sensors, or a paired ESP32 device — a locally-run LLM gives each real anomaly a voice.", + offers: { '@type': 'Offer', price: '0', priceCurrency: 'USD' }, + }, + }) + const [stats, setStats] = useState<Stats | null>(null) const [featured, setFeatured] = useState<CodexEntity[] | null>(null) const [taglineIdx, setTaglineIdx] = useState(0)