47 lines
1.5 KiB
Python
47 lines
1.5 KiB
Python
"""Call Ollama's /api/chat (non-streaming) with a single retry. Callers see
|
|
success or ModelError (charter §12/§13 — one retry, then fail; the client
|
|
degrades on any non-200). Built so a streaming mode is an additive path later.
|
|
"""
|
|
|
|
import httpx
|
|
|
|
from . import config
|
|
|
|
|
|
class ModelError(Exception):
|
|
"""The model call failed after one retry (transport / timeout / non-2xx / empty)."""
|
|
|
|
|
|
def chat(
|
|
model: str,
|
|
messages: list[dict],
|
|
options: dict,
|
|
*,
|
|
think: bool | None = None,
|
|
client: httpx.Client | None = None,
|
|
) -> str:
|
|
payload: dict = {"model": model, "messages": messages, "stream": False, "options": options}
|
|
if think is not None:
|
|
payload["think"] = think
|
|
|
|
owns = client is None
|
|
http = client or httpx.Client(
|
|
base_url=config.ollama_base_url(), timeout=config.ollama_timeout_seconds()
|
|
)
|
|
try:
|
|
last = "no attempt made"
|
|
for _ in range(2): # one try + one retry (§12: never more than once)
|
|
try:
|
|
resp = http.post("/api/chat", json=payload)
|
|
resp.raise_for_status()
|
|
content = resp.json().get("message", {}).get("content", "")
|
|
if content.strip():
|
|
return content
|
|
last = "empty response from model"
|
|
except httpx.HTTPError as exc: # TimeoutException, TransportError, HTTPStatusError
|
|
last = f"{type(exc).__name__}: {exc}"
|
|
raise ModelError(last)
|
|
finally:
|
|
if owns:
|
|
http.close()
|