feat(models): auto-mode intent remap + real coder models in preference

Settings "Auto model routing" picks which installed model fires for chat vs
coding intent when no model is pinned (auto_chat_model / auto_code_model).
_auto_select_model honors the remap; _MODEL_PREFERENCE["code"] prefers real
coder models first.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
jon
2026-07-23 16:06:46 -05:00
co-authored by Claude Opus 4.8
parent 9aea6d4228
commit 031e522704
5 changed files with 51 additions and 1 deletions
+28
View File
@@ -3,6 +3,8 @@ import { API_BASE } from "./config";
const DEFAULTS = {
model: "",
auto_chat_model: "", // Auto-mode: model for chat intent ("" = built-in preference)
auto_code_model: "", // Auto-mode: model for code intent
think: false, // Qwen3-style reasoning; off = much faster chat/memory
temperature: 0.7,
num_ctx: 0, // context window in tokens; 0 = model default
@@ -30,8 +32,11 @@ export function Settings() {
const [selectedApps, setSelectedApps] = useState([]);
const [brandQueue, setBrandQueue] = useState([]);
const [isBranding, setIsBranding] = useState(false);
const [modelList, setModelList] = useState([]);
useEffect(() => {
fetch(`${API_BASE}/models`).then(r => r.ok ? r.json() : null)
.then(d => setModelList((d && d.models) || [])).catch(() => {});
fetch(`${API_BASE}/settings`)
.then(r => r.ok ? r.json() : null)
.then(d => { if (d) setForm(f => ({ ...f, ...d })); })
@@ -165,6 +170,29 @@ export function Settings() {
<div style={{ maxWidth: "720px" }}>
<h2 style={{ margin: "0 0 1.25rem", fontSize: "1.1rem", color: "#eee" }}>Settings</h2>
{/* Auto model routing */}
<div style={sectionStyle}>
<h3 style={{ margin: "0 0 0.5rem", fontSize: "0.95rem", color: "#bbb" }}>Auto model routing</h3>
<div style={{ fontSize: "0.72rem", color: "#555", marginBottom: "0.75rem" }}>
When no model is pinned (chat picker on "auto"), which model fires for each
detected intent. "Auto" = the built-in preference for your hardware.
{form.model && <span style={{ color: "#c9a227" }}> A model is currently pinned in chat, so routing is bypassed until you set it back to auto.</span>}
</div>
<div style={{ display: "flex", gap: "1rem" }}>
{[["auto_chat_model", "Chat / general"], ["auto_code_model", "Coding questions"]].map(([key, label]) => (
<div key={key} style={{ flex: 1 }}>
<label style={labelStyle}>{label}</label>
<select value={form[key]} onChange={e => update(key, e.target.value)}
style={{ width: "100%", padding: "0.6rem", background: "#222", color: "#eee", border: "1px solid #333", borderRadius: "8px", marginTop: "0.3rem" }}>
<option value="">Auto (best installed)</option>
{modelList.map(m => <option key={m} value={m}>{m}</option>)}
{form[key] && !modelList.includes(form[key]) && <option value={form[key]}>{form[key]} (not installed)</option>}
</select>
</div>
))}
</div>
</div>
{/* Generation */}
<div style={sectionStyle}>
<h3 style={{ margin: "0 0 1rem", fontSize: "0.95rem", color: "#bbb" }}>Generation</h3>
+5
View File
@@ -104,6 +104,11 @@ async def _auto_select_model(message: str = "") -> str:
if s.get("model"):
return s["model"]
intent = _detect_intent(message) if message else "chat"
# Auto-mode remap: a configured model for this intent fires first;
# otherwise fall back to the built-in preference list.
remap = s.get(f"auto_{intent}_model")
if remap:
return remap
return await get_ollama_manager().select_best_model(intent)
except Exception:
return DEFAULT_CHAT_MODEL
+4
View File
@@ -986,6 +986,10 @@ class PersistentMemoryStore:
# -----------------------------
_SETTINGS_DEFAULTS: Dict[str, Any] = {
"model": "",
# Auto-mode routing overrides (blank = built-in preference). When no model
# is pinned, a message's detected intent picks which of these fires first.
"auto_chat_model": "",
"auto_code_model": "",
# Qwen3-style reasoning. Off by default: the hidden <think> block is pure
# latency for chat/memory. Turn on for hard multi-step problems.
"think": False,
+1 -1
View File
@@ -162,7 +162,7 @@ def _detect_gpu_backend() -> tuple[str, dict]:
# tail differs by task. Prefix-matched against installed model names.
_MODEL_PREFERENCE = {
"chat": ("qwen2.5:3b", "qwen2.5", "gemma3:1b", "gemma3", "phi3", "phi-3", "gemma2", "gemma"),
"code": ("qwen2.5:3b", "qwen2.5", "gemma3:1b", "gemma3", "phi3", "phi-3", "codellama", "deepseek-coder", "codegemma"),
"code": ("qwen2.5-coder", "qwen3-coder", "deepseek-coder", "codellama", "codegemma", "qwen2.5:3b", "qwen2.5", "gemma3", "phi3", "phi-3"),
}
+13
View File
@@ -43,6 +43,19 @@ def test_keep_alive_pins_the_model():
assert "keep_alive" not in mgr._apply_keep_alive({"model": "x"})
def test_auto_model_remap(monkeypatch):
import asyncio
from synapse import main
cfg = {"model": "", "auto_chat_model": "chatX", "auto_code_model": "coderY"}
monkeypatch.setattr(main.store, "get_settings", lambda: cfg)
# code intent ("function") routes to the code remap; chat intent to the chat remap
assert asyncio.run(main._auto_select_model("write a function to sort a list")) == "coderY"
assert asyncio.run(main._auto_select_model("how are you today")) == "chatX"
# an explicit pin beats the remap
cfg["model"] = "pinnedZ"
assert asyncio.run(main._auto_select_model("debug this code")) == "pinnedZ"
def test_hardware_fit_logic():
from synapse import hardware
assert hardware._fit(2.5, 4.0, 16.0) == "gpu" # 2.5+1 <= 4 -> fits GPU