feat: add Ollama client and bounded request queue
This commit is contained in:
18
backend/app/llm/client.py
Normal file
18
backend/app/llm/client.py
Normal file
@@ -0,0 +1,18 @@
|
||||
import httpx
|
||||
|
||||
from app.config import settings
|
||||
|
||||
|
||||
class OllamaClient:
|
||||
def __init__(self, base_url: str | None = None):
|
||||
self._base_url = base_url or settings.ollama_base_url
|
||||
|
||||
async def generate(self, model: str, prompt: str, system: str | None = None) -> str:
|
||||
payload = {"model": model, "prompt": prompt, "stream": False}
|
||||
if system is not None:
|
||||
payload["system"] = system
|
||||
|
||||
async with httpx.AsyncClient(base_url=self._base_url, timeout=120.0) as http_client:
|
||||
response = await http_client.post("/api/generate", json=payload)
|
||||
response.raise_for_status()
|
||||
return response.json()["response"]
|
||||
Reference in New Issue
Block a user