feat(api): config + role→model routing (qwen3.5 narrator default, think off)
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
22
api/app/config.py
Normal file
22
api/app/config.py
Normal file
@@ -0,0 +1,22 @@
|
|||||||
|
"""Server config, read from the environment at call time so tests can override
|
||||||
|
without reimport. Charter §4 — all of this lives server-side; the client never
|
||||||
|
sees it.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
|
||||||
|
def ollama_base_url() -> str:
|
||||||
|
return os.environ.get("OLLAMA_BASE_URL", "http://localhost:11434")
|
||||||
|
|
||||||
|
|
||||||
|
def ollama_timeout_seconds() -> float:
|
||||||
|
return float(os.environ.get("OLLAMA_TIMEOUT_SECONDS", "30"))
|
||||||
|
|
||||||
|
|
||||||
|
def narrator_model() -> str:
|
||||||
|
return os.environ.get("OLLAMA_NARRATOR_MODEL", "qwen3.5:latest")
|
||||||
|
|
||||||
|
|
||||||
|
def call_log_path() -> str | None:
|
||||||
|
return os.environ.get("CALL_LOG_PATH") or None
|
||||||
25
api/app/routing.py
Normal file
25
api/app/routing.py
Normal file
@@ -0,0 +1,25 @@
|
|||||||
|
"""Role → model routing (charter §4/§5). Model choice is config, not law — a
|
||||||
|
deploy, not a client patch. Read at call time so an env override needs no reimport.
|
||||||
|
The provider abstraction (Ollama → Replicate) slots in here later (YAGNI now).
|
||||||
|
"""
|
||||||
|
|
||||||
|
from dataclasses import dataclass
|
||||||
|
|
||||||
|
from . import config
|
||||||
|
|
||||||
|
|
||||||
|
@dataclass(frozen=True)
|
||||||
|
class RoleConfig:
|
||||||
|
model: str
|
||||||
|
options: dict
|
||||||
|
think: bool | None = None # Ollama top-level `think`; None omits it
|
||||||
|
|
||||||
|
|
||||||
|
def for_role(role: str) -> RoleConfig:
|
||||||
|
if role == "narrator":
|
||||||
|
return RoleConfig(
|
||||||
|
model=config.narrator_model(),
|
||||||
|
options={"temperature": 0.8, "top_p": 0.9, "num_predict": 300},
|
||||||
|
think=False, # qwen3.x thinking OFF; harmless on non-thinking models
|
||||||
|
)
|
||||||
|
raise KeyError(f"no routing for role: {role}")
|
||||||
21
api/tests/test_routing.py
Normal file
21
api/tests/test_routing.py
Normal file
@@ -0,0 +1,21 @@
|
|||||||
|
import pytest
|
||||||
|
|
||||||
|
from app.routing import for_role
|
||||||
|
|
||||||
|
|
||||||
|
def test_narrator_defaults():
|
||||||
|
cfg = for_role("narrator")
|
||||||
|
assert cfg.model == "qwen3.5:latest"
|
||||||
|
assert cfg.think is False
|
||||||
|
assert cfg.options["num_predict"] == 300
|
||||||
|
assert cfg.options["temperature"] == 0.8
|
||||||
|
|
||||||
|
|
||||||
|
def test_env_overrides_model(monkeypatch):
|
||||||
|
monkeypatch.setenv("OLLAMA_NARRATOR_MODEL", "llama3.1:latest")
|
||||||
|
assert for_role("narrator").model == "llama3.1:latest"
|
||||||
|
|
||||||
|
|
||||||
|
def test_unknown_role_raises():
|
||||||
|
with pytest.raises(KeyError):
|
||||||
|
for_role("wizard")
|
||||||
Reference in New Issue
Block a user