feat(api): Ollama /api/chat client with one retry + typed ModelError
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
46
api/app/ollama_client.py
Normal file
46
api/app/ollama_client.py
Normal file
@@ -0,0 +1,46 @@
|
|||||||
|
"""Call Ollama's /api/chat (non-streaming) with a single retry. Callers see
|
||||||
|
success or ModelError (charter §12/§13 — one retry, then fail; the client
|
||||||
|
degrades on any non-200). Built so a streaming mode is an additive path later.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
|
||||||
|
from . import config
|
||||||
|
|
||||||
|
|
||||||
|
class ModelError(Exception):
|
||||||
|
"""The model call failed after one retry (transport / timeout / non-2xx / empty)."""
|
||||||
|
|
||||||
|
|
||||||
|
def chat(
|
||||||
|
model: str,
|
||||||
|
messages: list[dict],
|
||||||
|
options: dict,
|
||||||
|
*,
|
||||||
|
think: bool | None = None,
|
||||||
|
client: httpx.Client | None = None,
|
||||||
|
) -> str:
|
||||||
|
payload: dict = {"model": model, "messages": messages, "stream": False, "options": options}
|
||||||
|
if think is not None:
|
||||||
|
payload["think"] = think
|
||||||
|
|
||||||
|
owns = client is None
|
||||||
|
http = client or httpx.Client(
|
||||||
|
base_url=config.ollama_base_url(), timeout=config.ollama_timeout_seconds()
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
last = "no attempt made"
|
||||||
|
for _ in range(2): # one try + one retry (§12: never more than once)
|
||||||
|
try:
|
||||||
|
resp = http.post("/api/chat", json=payload)
|
||||||
|
resp.raise_for_status()
|
||||||
|
content = resp.json().get("message", {}).get("content", "")
|
||||||
|
if content.strip():
|
||||||
|
return content
|
||||||
|
last = "empty response from model"
|
||||||
|
except httpx.HTTPError as exc: # TimeoutException, TransportError, HTTPStatusError
|
||||||
|
last = f"{type(exc).__name__}: {exc}"
|
||||||
|
raise ModelError(last)
|
||||||
|
finally:
|
||||||
|
if owns:
|
||||||
|
http.close()
|
||||||
83
api/tests/test_ollama_client.py
Normal file
83
api/tests/test_ollama_client.py
Normal file
@@ -0,0 +1,83 @@
|
|||||||
|
import json
|
||||||
|
|
||||||
|
import httpx
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.ollama_client import ModelError, chat
|
||||||
|
|
||||||
|
|
||||||
|
def _client(handler):
|
||||||
|
return httpx.Client(transport=httpx.MockTransport(handler), base_url="http://ollama")
|
||||||
|
|
||||||
|
|
||||||
|
def test_success_returns_content_and_sends_payload():
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
seen["body"] = json.loads(request.content)
|
||||||
|
return httpx.Response(200, json={"message": {"content": "You stand on the wharf."}})
|
||||||
|
|
||||||
|
out = chat("qwen3.5:latest", [{"role": "user", "content": "hi"}], {"seed": 7},
|
||||||
|
think=False, client=_client(handler))
|
||||||
|
assert out == "You stand on the wharf."
|
||||||
|
assert seen["body"]["model"] == "qwen3.5:latest"
|
||||||
|
assert seen["body"]["stream"] is False
|
||||||
|
assert seen["body"]["options"]["seed"] == 7
|
||||||
|
assert seen["body"]["think"] is False
|
||||||
|
|
||||||
|
|
||||||
|
def test_think_omitted_when_none():
|
||||||
|
seen = {}
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
seen["body"] = json.loads(request.content)
|
||||||
|
return httpx.Response(200, json={"message": {"content": "ok"}})
|
||||||
|
|
||||||
|
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
|
||||||
|
assert "think" not in seen["body"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_retries_once_then_succeeds():
|
||||||
|
calls = {"n": 0}
|
||||||
|
|
||||||
|
def handler(request):
|
||||||
|
calls["n"] += 1
|
||||||
|
if calls["n"] == 1:
|
||||||
|
raise httpx.ConnectError("boom")
|
||||||
|
return httpx.Response(200, json={"message": {"content": "recovered"}})
|
||||||
|
|
||||||
|
out = chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
|
||||||
|
assert out == "recovered"
|
||||||
|
assert calls["n"] == 2
|
||||||
|
|
||||||
|
|
||||||
|
def test_two_transport_failures_raise_modelerror():
|
||||||
|
def handler(request):
|
||||||
|
raise httpx.ConnectError("down")
|
||||||
|
|
||||||
|
with pytest.raises(ModelError):
|
||||||
|
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
|
||||||
|
|
||||||
|
|
||||||
|
def test_timeout_exhausted_raises_modelerror():
|
||||||
|
def handler(request):
|
||||||
|
raise httpx.TimeoutException("slow")
|
||||||
|
|
||||||
|
with pytest.raises(ModelError):
|
||||||
|
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
|
||||||
|
|
||||||
|
|
||||||
|
def test_non_2xx_retried_then_modelerror():
|
||||||
|
def handler(request):
|
||||||
|
return httpx.Response(500, json={"error": "nope"})
|
||||||
|
|
||||||
|
with pytest.raises(ModelError):
|
||||||
|
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
|
||||||
|
|
||||||
|
|
||||||
|
def test_empty_content_retried_then_modelerror():
|
||||||
|
def handler(request):
|
||||||
|
return httpx.Response(200, json={"message": {"content": " "}})
|
||||||
|
|
||||||
|
with pytest.raises(ModelError):
|
||||||
|
chat("m", [{"role": "user", "content": "x"}], {}, client=_client(handler))
|
||||||
Reference in New Issue
Block a user