feat(api): Ollama /api/chat client with one retry + typed ModelError

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
2026-07-10 08:15:45 -05:00
parent 4d8a19353e
commit be133c4f3a
2 changed files with 129 additions and 0 deletions

46
api/app/ollama_client.py Normal file
View File

@@ -0,0 +1,46 @@
"""Call Ollama's /api/chat (non-streaming) with a single retry. Callers see
success or ModelError (charter §12/§13 — one retry, then fail; the client
degrades on any non-200). Built so a streaming mode is an additive path later.
"""
import httpx
from . import config
class ModelError(Exception):
"""The model call failed after one retry (transport / timeout / non-2xx / empty)."""
def chat(
model: str,
messages: list[dict],
options: dict,
*,
think: bool | None = None,
client: httpx.Client | None = None,
) -> str:
payload: dict = {"model": model, "messages": messages, "stream": False, "options": options}
if think is not None:
payload["think"] = think
owns = client is None
http = client or httpx.Client(
base_url=config.ollama_base_url(), timeout=config.ollama_timeout_seconds()
)
try:
last = "no attempt made"
for _ in range(2): # one try + one retry (§12: never more than once)
try:
resp = http.post("/api/chat", json=payload)
resp.raise_for_status()
content = resp.json().get("message", {}).get("content", "")
if content.strip():
return content
last = "empty response from model"
except httpx.HTTPError as exc: # TimeoutException, TransportError, HTTPStatusError
last = f"{type(exc).__name__}: {exc}"
raise ModelError(last)
finally:
if owns:
http.close()

View File

@@ -0,0 +1,83 @@
import json
import httpx
import pytest
from app.ollama_client import ModelError, chat
def _client(handler):
return httpx.Client(transport=httpx.MockTransport(handler), base_url="http://ollama")
def test_success_returns_content_and_sends_payload():
seen = {}
def handler(request):
seen["body"] = json.loads(request.content)
return httpx.Response(200, json={"message": {"content": "You stand on the wharf."}})
out = chat("qwen3.5:latest", [{"role": "user", "content": "hi"}], {"seed": 7},
think=False, client=_client(handler))
assert out == "You stand on the wharf."
assert seen["body"]["model"] == "qwen3.5:latest"
assert seen["body"]["stream"] is False
assert seen["body"]["options"]["seed"] == 7
assert seen["body"]["think"] is False
def test_think_omitted_when_none():
seen = {}
def handler(request):
seen["body"] = json.loads(request.content)
return httpx.Response(200, json={"message": {"content": "ok"}})
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
assert "think" not in seen["body"]
def test_retries_once_then_succeeds():
calls = {"n": 0}
def handler(request):
calls["n"] += 1
if calls["n"] == 1:
raise httpx.ConnectError("boom")
return httpx.Response(200, json={"message": {"content": "recovered"}})
out = chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
assert out == "recovered"
assert calls["n"] == 2
def test_two_transport_failures_raise_modelerror():
def handler(request):
raise httpx.ConnectError("down")
with pytest.raises(ModelError):
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
def test_timeout_exhausted_raises_modelerror():
def handler(request):
raise httpx.TimeoutException("slow")
with pytest.raises(ModelError):
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
def test_non_2xx_retried_then_modelerror():
def handler(request):
return httpx.Response(500, json={"error": "nope"})
with pytest.raises(ModelError):
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
def test_empty_content_retried_then_modelerror():
def handler(request):
return httpx.Response(200, json={"message": {"content": " "}})
with pytest.raises(ModelError):
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))