"""robots.txt and a live sitemap. Served from the backend rather than dropped in `public/` because the valuable, indexable surface of this site is the Codex, and the Codex grows every time somebody summons. A static sitemap would be stale within an hour; this one is generated from the actual entity rows. What is deliberately NOT listed: /seance, /enter, /profile, /log, /inventory, /devices — anything per-seeker or interactive. A crawler hitting /seance would provision a guest account on arrival (the open door), which would fill the users table with wanderers that never existed as people. robots.txt disallows those paths for the same reason, and the sitemap only advertises pages that are genuinely public, stable, and worth a search result: the landing page, the shop, and every discovered spirit. """ from datetime import datetime, timezone from fastapi import APIRouter, Depends, Response from sqlalchemy import select from sqlalchemy.ext.asyncio import AsyncSession from app.config import settings from app.db import get_db from app.models.entity import Entity router = APIRouter(tags=["seo"]) # Cap the sitemap so a runaway Codex can't produce a multi-megabyte # document. 5k is far inside the 50k/50MB sitemap limit and orders of # magnitude beyond what this install will realistically hold. MAX_SITEMAP_ENTITIES = 5000 # Paths that must never be crawled: they either mutate state (provisioning a # guest) or are meaningless without a session. DISALLOWED = ( "/seance", "/enter", "/profile", "/hunters", "/log", "/inventory", "/devices", "/api/", ) def _base_url() -> str: return settings.public_base_url.rstrip("/") def _iso(dt: datetime | None) -> str: value = dt or datetime.now(timezone.utc) if value.tzinfo is None: value = value.replace(tzinfo=timezone.utc) return value.date().isoformat() def _xml_escape(text: str) -> str: return ( text.replace("&", "&") .replace("<", "<") .replace(">", ">") .replace('"', """) ) @router.get("/robots.txt", include_in_schema=False) async def robots() -> Response: lines = ["User-agent: *"] lines += [f"Disallow: {path}" for path in DISALLOWED] lines.append("Allow: /") lines.append(f"Sitemap: {_base_url()}/sitemap.xml") return Response("\n".join(lines) + "\n", media_type="text/plain") @router.get("/sitemap.xml", include_in_schema=False) async def sitemap(db: AsyncSession = Depends(get_db)) -> Response: base = _base_url() result = await db.execute( select(Entity.id, Entity.discovered_at) .order_by(Entity.discovered_at.desc()) .limit(MAX_SITEMAP_ENTITIES) ) entities = result.all() parts = [ '', '', f"{base}/daily" "1.0", f"{base}/codexhourly" "0.9", f"{base}/shopweekly" "0.7", ] for entity_id, discovered_at in entities: loc = _xml_escape(f"{base}/codex/{entity_id}") parts.append( f"{loc}{_iso(discovered_at)}" "weekly0.6" ) parts.append("") return Response("\n".join(parts), media_type="application/xml")