feat: add Ollama client and bounded request queue

This commit is contained in:
Indiana
2026-07-20 18:01:36 +00:00
parent cd9fc49eed
commit 52afad1ad5
4 changed files with 107 additions and 0 deletions

18
backend/app/llm/client.py Normal file
View File

@@ -0,0 +1,18 @@
import httpx
from app.config import settings
class OllamaClient:
def __init__(self, base_url: str | None = None):
self._base_url = base_url or settings.ollama_base_url
async def generate(self, model: str, prompt: str, system: str | None = None) -> str:
payload = {"model": model, "prompt": prompt, "stream": False}
if system is not None:
payload["system"] = system
async with httpx.AsyncClient(base_url=self._base_url, timeout=120.0) as http_client:
response = await http_client.post("/api/generate", json=payload)
response.raise_for_status()
return response.json()["response"]