feat: spirit engine — seance WS, entity minting/Codex, Piper TTS voices, wire telemetry
- WS /ws/session: modes, summon, anomaly fragments, streaming direct contact,
passive wire-ghost ambient loop, per-user rate limits, event transcript
- entities: anomaly-signature fingerprinting, LLM minting + procedural
fallback, Codex matching with contact counts and sightings
- tts: 8 local Piper voices (EN/ES), per-entity voice profiles, numpy
effects chain (pitch/rate/bitcrush/echo/static)
- llm: streaming client, submit_stream in bounded queue, SpiritService
with offline fallbacks for every channel
- routes: public /api/codex, /api/codex/{id}, /api/stats; /audio static mount
- models: Entity, EntitySighting, Event, ContactSession(entity_id, language)
This commit is contained in:
@@ -1,5 +1,5 @@
|
||||
import asyncio
|
||||
from collections.abc import Awaitable, Callable
|
||||
from collections.abc import AsyncIterator, Awaitable, Callable
|
||||
from typing import TypeVar
|
||||
|
||||
T = TypeVar("T")
|
||||
@@ -44,3 +44,27 @@ class LLMQueue:
|
||||
self._semaphore.release()
|
||||
else:
|
||||
self._waiting -= 1
|
||||
|
||||
async def submit_stream(
|
||||
self, gen_factory: Callable[[], AsyncIterator[T]]
|
||||
) -> AsyncIterator[T]:
|
||||
"""Like submit(), but for async generators (token streams). The
|
||||
concurrency slot is held for the stream's whole lifetime, since the
|
||||
Ollama box stays busy until the last token."""
|
||||
async with self._lock:
|
||||
if self._waiting >= self._max_queue_depth:
|
||||
raise QueueFullError("too many seekers right now")
|
||||
self._waiting += 1
|
||||
|
||||
acquired = False
|
||||
try:
|
||||
await self._semaphore.acquire()
|
||||
acquired = True
|
||||
self._waiting -= 1
|
||||
async for item in gen_factory():
|
||||
yield item
|
||||
finally:
|
||||
if acquired:
|
||||
self._semaphore.release()
|
||||
else:
|
||||
self._waiting -= 1
|
||||
|
||||
Reference in New Issue
Block a user