feat(api): config + role→model routing (qwen3.5 narrator default, think off)
Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
22
api/app/config.py
Normal file
22
api/app/config.py
Normal file
@@ -0,0 +1,22 @@
|
||||
"""Server config, read from the environment at call time so tests can override
|
||||
without reimport. Charter §4 — all of this lives server-side; the client never
|
||||
sees it.
|
||||
"""
|
||||
|
||||
import os
|
||||
|
||||
|
||||
def ollama_base_url() -> str:
|
||||
return os.environ.get("OLLAMA_BASE_URL", "http://localhost:11434")
|
||||
|
||||
|
||||
def ollama_timeout_seconds() -> float:
|
||||
return float(os.environ.get("OLLAMA_TIMEOUT_SECONDS", "30"))
|
||||
|
||||
|
||||
def narrator_model() -> str:
|
||||
return os.environ.get("OLLAMA_NARRATOR_MODEL", "qwen3.5:latest")
|
||||
|
||||
|
||||
def call_log_path() -> str | None:
|
||||
return os.environ.get("CALL_LOG_PATH") or None
|
||||
25
api/app/routing.py
Normal file
25
api/app/routing.py
Normal file
@@ -0,0 +1,25 @@
|
||||
"""Role → model routing (charter §4/§5). Model choice is config, not law — a
|
||||
deploy, not a client patch. Read at call time so an env override needs no reimport.
|
||||
The provider abstraction (Ollama → Replicate) slots in here later (YAGNI now).
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
|
||||
from . import config
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RoleConfig:
|
||||
model: str
|
||||
options: dict
|
||||
think: bool | None = None # Ollama top-level `think`; None omits it
|
||||
|
||||
|
||||
def for_role(role: str) -> RoleConfig:
|
||||
if role == "narrator":
|
||||
return RoleConfig(
|
||||
model=config.narrator_model(),
|
||||
options={"temperature": 0.8, "top_p": 0.9, "num_predict": 300},
|
||||
think=False, # qwen3.x thinking OFF; harmless on non-thinking models
|
||||
)
|
||||
raise KeyError(f"no routing for role: {role}")
|
||||
21
api/tests/test_routing.py
Normal file
21
api/tests/test_routing.py
Normal file
@@ -0,0 +1,21 @@
|
||||
import pytest
|
||||
|
||||
from app.routing import for_role
|
||||
|
||||
|
||||
def test_narrator_defaults():
|
||||
cfg = for_role("narrator")
|
||||
assert cfg.model == "qwen3.5:latest"
|
||||
assert cfg.think is False
|
||||
assert cfg.options["num_predict"] == 300
|
||||
assert cfg.options["temperature"] == 0.8
|
||||
|
||||
|
||||
def test_env_overrides_model(monkeypatch):
|
||||
monkeypatch.setenv("OLLAMA_NARRATOR_MODEL", "llama3.1:latest")
|
||||
assert for_role("narrator").model == "llama3.1:latest"
|
||||
|
||||
|
||||
def test_unknown_role_raises():
|
||||
with pytest.raises(KeyError):
|
||||
for_role("wizard")
|
||||
Reference in New Issue
Block a user