forked from retoor/devplacepy
Add OpenCode Zen support, model health/stats dashboard, and gateway fallback fixes
AI gateway: - Add a generic, admin-selectable `client_profile` field on gateway_providers (e.g. "opencode") so a provider needing special request headers (OpenCode Zen's client-identity spoofing) is configured like any other provider, not hardcoded by name. - Track per-(provider, model) reliability/speed/latency health in memory, seeded from the existing gateway_usage_ledger at startup - purely observational, never influences routing. - New Stats tab on /admin/gateway: request volume, latency, per-model breakdowns, and reliability weight, charted with a vendored Chart.js and devplace's own theme tokens. - Record which model a failed request actually fell back to (fallback_used_route), surfaced in the Recent Failures table. - Stop excluding context_length errors from fallback, and skip a primary attempt outright when its known context window is already too small for the estimated request size, going straight to the fallback. - gateway_usage_ledger's provider/fallback_used_route columns and indexes are ensured centrally in database/schema.py's init_db(), the single point of truth for this table's schema. - Non-OpenAI upstream routing and client-model passthrough; trust only the upstream's own X-Gateway-Model header for served-model attribution. Devii agent: - Fix a real lockup: plan/verify tools could be individually disabled via the admin tool toggles while still being required by the protocol gate, permanently bricking any task that needed tools. They can no longer be disabled, and the gate now also checks the tool is actually offered. - Fix compaction being silently calibrated for a 1M-token model while running a much smaller one: context budget is now percentage-based and the summarizer's own request is sized to fit the real model. - Give a specific, actionable retry message when plan()'s own arguments get cut off by the output limit, and tighten its schema to discourage overlong plans. Other: - Backup service: offload completed backups to a remote Hetzner Storage Box. - Container manager: fix orphan blob leaks from sync races, add a two-phase plan/execute `system prune` CLI command. - Admin gateway UI: replace the JS-rendered model/provider tables with server-rendered forms and pages. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01DhmEkvutuwtzFVcLbTrhdo
This commit is contained in:
@@ -41,6 +41,10 @@ class FakeResp_openai_gateway:
|
||||
content = payload["choices"][0]["message"].get("content") or ""
|
||||
except (KeyError, IndexError, TypeError):
|
||||
content = ""
|
||||
try:
|
||||
reasoning = payload["choices"][0]["message"].get("reasoning") or ""
|
||||
except (KeyError, IndexError, TypeError):
|
||||
reasoning = ""
|
||||
|
||||
def frame(delta=None, finish=None, usage=None):
|
||||
body = {"id": chunk_id, "object": "chat.completion.chunk", "model": model}
|
||||
@@ -52,6 +56,8 @@ class FakeResp_openai_gateway:
|
||||
return f"data: {json.dumps(body)}"
|
||||
|
||||
yield frame({"role": "assistant"})
|
||||
for i in range(0, len(reasoning), 5):
|
||||
yield frame({"reasoning": reasoning[i : i + 5]})
|
||||
for i in range(0, len(content), 5):
|
||||
yield frame({"content": content[i : i + 5]})
|
||||
yield frame({}, finish="stop")
|
||||
@@ -693,6 +699,59 @@ class FakeAlwaysFailClient_openai_gateway:
|
||||
pass
|
||||
|
||||
|
||||
class FakeBadRequestClient_openai_gateway:
|
||||
def __init__(self, *a, **k):
|
||||
self.calls = []
|
||||
|
||||
def build_request(self, method, url, headers=None, json=None, content=None):
|
||||
return FakeRequest(method, url, json)
|
||||
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
return FakeResp_openai_gateway(
|
||||
status=400, payload={"error": {"message": "invalid request: bad param"}}
|
||||
)
|
||||
|
||||
async def aclose(self):
|
||||
pass
|
||||
|
||||
|
||||
def test_chat_does_not_fall_back_on_an_unrecoverable_bad_request(local_db, monkeypatch):
|
||||
from devplacepy.services.openai_gateway import routing
|
||||
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeBadRequestClient_openai_gateway)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(source_model="fb-badreq-backup", target_model="backup-target")
|
||||
)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(
|
||||
source_model="fb-badreq-primary",
|
||||
target_model="primary-target",
|
||||
fallback_model="fb-badreq-backup",
|
||||
)
|
||||
)
|
||||
try:
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_vision_enabled"] = False
|
||||
cfg["gateway_max_retries"] = 0
|
||||
rt = svc.runtime()
|
||||
response = run_async(
|
||||
rt.handle_chat(
|
||||
{"model": "fb-badreq-primary", "messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "no_fallback_on_bad_request"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert response.status_code == 400
|
||||
assert len(rt._client.calls) == 1
|
||||
finally:
|
||||
routing.model_store.remove("fb-badreq-primary")
|
||||
routing.model_store.remove("fb-badreq-backup")
|
||||
|
||||
|
||||
def test_chat_falls_back_when_the_primary_model_fails(local_db, monkeypatch):
|
||||
from devplacepy.services.openai_gateway import routing
|
||||
|
||||
@@ -735,6 +794,127 @@ def test_chat_falls_back_when_the_primary_model_fails(local_db, monkeypatch):
|
||||
routing.model_store.remove("fb-backup-route")
|
||||
|
||||
|
||||
class FakeContextLengthClient_openai_gateway:
|
||||
def __init__(self, *a, **k):
|
||||
self.calls = []
|
||||
|
||||
def build_request(self, method, url, headers=None, json=None, content=None):
|
||||
return FakeRequest(method, url, json)
|
||||
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
model = body.get("model")
|
||||
if model == "ctx-primary-target":
|
||||
return FakeResp_openai_gateway(
|
||||
status=400,
|
||||
payload={"error": {"message": "maximum context length exceeded"}},
|
||||
)
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"id": "x",
|
||||
"model": model,
|
||||
"choices": [{"message": {"content": "fallback ok"}}],
|
||||
}
|
||||
)
|
||||
|
||||
async def aclose(self):
|
||||
pass
|
||||
|
||||
|
||||
def test_chat_falls_back_on_a_context_length_error(local_db, monkeypatch):
|
||||
from devplacepy.services.openai_gateway import routing
|
||||
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeContextLengthClient_openai_gateway)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(source_model="ctx-backup-route", target_model="ctx-backup-target")
|
||||
)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(
|
||||
source_model="ctx-primary-route",
|
||||
target_model="ctx-primary-target",
|
||||
fallback_model="ctx-backup-route",
|
||||
)
|
||||
)
|
||||
try:
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_vision_enabled"] = False
|
||||
cfg["gateway_max_retries"] = 0
|
||||
rt = svc.runtime()
|
||||
response = run_async(
|
||||
rt.handle_chat(
|
||||
{"model": "ctx-primary-route", "messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "context_length_fallback_success"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert response.status_code == 200
|
||||
assert rt._client.calls[0][1]["model"] == "ctx-primary-target"
|
||||
assert rt._client.calls[-1][1]["model"] == "ctx-backup-target"
|
||||
row = get_table("gateway_usage_ledger").find_one(
|
||||
owner_id="context_length_fallback_success"
|
||||
)
|
||||
assert row["success"] == 1
|
||||
assert row["fallback_used_route"] == "ctx-backup-route"
|
||||
finally:
|
||||
routing.model_store.remove("ctx-primary-route")
|
||||
routing.model_store.remove("ctx-backup-route")
|
||||
|
||||
|
||||
def test_chat_skips_a_too_small_primary_and_goes_straight_to_fallback(local_db, monkeypatch):
|
||||
from devplacepy.services.openai_gateway import routing
|
||||
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeContextLengthClient_openai_gateway)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(
|
||||
source_model="ctx-precheck-backup",
|
||||
target_model="ctx-backup-target",
|
||||
context_window=1_000_000,
|
||||
)
|
||||
)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(
|
||||
source_model="ctx-precheck-primary",
|
||||
target_model="ctx-primary-target",
|
||||
context_window=50,
|
||||
fallback_model="ctx-precheck-backup",
|
||||
)
|
||||
)
|
||||
try:
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_vision_enabled"] = False
|
||||
cfg["gateway_max_retries"] = 0
|
||||
rt = svc.runtime()
|
||||
long_message = "word " * 2000
|
||||
response = run_async(
|
||||
rt.handle_chat(
|
||||
{
|
||||
"model": "ctx-precheck-primary",
|
||||
"messages": [{"role": "user", "content": long_message}],
|
||||
},
|
||||
cfg,
|
||||
("guest", "context_precheck_skips_primary"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert response.status_code == 200
|
||||
# Only the fallback was ever called - the primary was known too small.
|
||||
assert len(rt._client.calls) == 1
|
||||
assert rt._client.calls[0][1]["model"] == "ctx-backup-target"
|
||||
row = get_table("gateway_usage_ledger").find_one(
|
||||
owner_id="context_precheck_skips_primary"
|
||||
)
|
||||
assert row["fallback_used_route"] == "ctx-precheck-backup"
|
||||
finally:
|
||||
routing.model_store.remove("ctx-precheck-primary")
|
||||
routing.model_store.remove("ctx-precheck-backup")
|
||||
|
||||
|
||||
def test_chat_falls_back_via_molodetz_when_the_client_sends_an_unrouted_model_name(
|
||||
local_db, monkeypatch
|
||||
):
|
||||
@@ -1271,6 +1451,50 @@ def test_stream_records_ttft_and_inter_token_in_ledger(local_db, monkeypatch):
|
||||
assert row["inter_token_ms"] is not None and row["inter_token_ms"] >= 0.0
|
||||
|
||||
|
||||
class FakeClientReasoning_openai_gateway(FakeClient_openai_gateway):
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"id": "x",
|
||||
"model": body.get("model"),
|
||||
"choices": [{"message": {"content": "x", "reasoning": "aaaaa"}}],
|
||||
"usage": {"prompt_tokens": 7, "completion_tokens": 3, "total_tokens": 10},
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def test_ollama_reasoning_delta_is_counted_as_a_content_chunk(local_db, monkeypatch):
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeClientReasoning_openai_gateway)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_chat(
|
||||
{
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"stream": True,
|
||||
"stream_options": {"include_usage": True},
|
||||
},
|
||||
cfg,
|
||||
("guest", "ollama_reasoning_probe"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
|
||||
async def drain():
|
||||
async for _ in resp.body_iterator:
|
||||
pass
|
||||
|
||||
run_async(drain())
|
||||
|
||||
row = get_table("gateway_usage_ledger").find_one(owner_id="ollama_reasoning_probe")
|
||||
assert row is not None
|
||||
assert row["inter_token_ms"] is not None and row["inter_token_ms"] >= 0.0
|
||||
|
||||
|
||||
def test_stream_client_disconnect_records_failure_and_closes_upstream(local_db, monkeypatch):
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeClient_openai_gateway)
|
||||
svc = GatewayService()
|
||||
@@ -1552,6 +1776,153 @@ def test_no_upstream_model_header_keeps_our_own_model(local_db, monkeypatch):
|
||||
assert resp.headers["X-Gateway-Model"] == "deepseek-chat"
|
||||
|
||||
|
||||
class FakeEmbeddedErrorClient_openai_gateway:
|
||||
"""Mimics OpenRouter's documented behavior of answering 200 OK with the
|
||||
failure embedded in the JSON body (openrouter.ai/docs/api-reference/errors)
|
||||
instead of a non-2xx status."""
|
||||
|
||||
def __init__(self, *a, **k):
|
||||
self.calls = []
|
||||
|
||||
def build_request(self, method, url, headers=None, json=None, content=None):
|
||||
return FakeRequest(method, url, json)
|
||||
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
model = body.get("model")
|
||||
if model == "primary-target":
|
||||
return FakeResp_openai_gateway(
|
||||
status=200,
|
||||
payload={"error": {"message": "no provider available", "code": 502}},
|
||||
)
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"id": "x",
|
||||
"model": model,
|
||||
"choices": [{"message": {"content": "fallback ok"}}],
|
||||
}
|
||||
)
|
||||
|
||||
async def aclose(self):
|
||||
pass
|
||||
|
||||
|
||||
def test_chat_falls_back_when_upstream_returns_200_with_an_embedded_error(
|
||||
local_db, monkeypatch
|
||||
):
|
||||
from devplacepy.services.openai_gateway import routing
|
||||
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeEmbeddedErrorClient_openai_gateway
|
||||
)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(source_model="eb-backup-route", target_model="backup-target")
|
||||
)
|
||||
routing.model_store.set(
|
||||
routing.ModelRouteIn(
|
||||
source_model="eb-primary-route",
|
||||
target_model="primary-target",
|
||||
fallback_model="eb-backup-route",
|
||||
)
|
||||
)
|
||||
try:
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_vision_enabled"] = False
|
||||
cfg["gateway_max_retries"] = 0
|
||||
rt = svc.runtime()
|
||||
response = run_async(
|
||||
rt.handle_chat(
|
||||
{"model": "eb-primary-route", "messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "embedded_error_chat"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert response.status_code == 200
|
||||
assert rt._client.calls[0][1]["model"] == "primary-target"
|
||||
assert rt._client.calls[-1][1]["model"] == "backup-target"
|
||||
row = get_table("gateway_usage_ledger").find_one(owner_id="embedded_error_chat")
|
||||
assert row is not None
|
||||
assert row["requested_model"] == "eb-primary-route"
|
||||
assert row["model"] == "backup-target"
|
||||
assert row["success"] == 1
|
||||
finally:
|
||||
routing.model_store.remove("eb-primary-route")
|
||||
routing.model_store.remove("eb-backup-route")
|
||||
|
||||
|
||||
class FakeAlwaysEmbeddedErrorClient_openai_gateway:
|
||||
def __init__(self, *a, **k):
|
||||
self.calls = []
|
||||
|
||||
def build_request(self, method, url, headers=None, json=None, content=None):
|
||||
return FakeRequest(method, url, json)
|
||||
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
return FakeResp_openai_gateway(
|
||||
status=200,
|
||||
payload={"error": {"message": "no provider available", "code": 502}},
|
||||
)
|
||||
|
||||
async def aclose(self):
|
||||
pass
|
||||
|
||||
|
||||
def test_chat_reports_502_when_upstream_returns_200_with_an_embedded_error_and_no_fallback(
|
||||
local_db, monkeypatch
|
||||
):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeAlwaysEmbeddedErrorClient_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_vision_enabled"] = False
|
||||
cfg["gateway_max_retries"] = 0
|
||||
rt = svc.runtime()
|
||||
response = run_async(
|
||||
rt.handle_chat(
|
||||
{"messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "embedded_error_no_fallback"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert response.status_code == 502
|
||||
row = get_table("gateway_usage_ledger").find_one(owner_id="embedded_error_no_fallback")
|
||||
assert row is not None
|
||||
assert row["success"] == 0
|
||||
assert row["error_category"] == "upstream_error"
|
||||
|
||||
|
||||
def test_ollama_dialect_sends_reasoning_effort_alongside_think(local_db, monkeypatch):
|
||||
# Ollama's native /api/chat honors a boolean `think`, but its
|
||||
# OpenAI-compatible /v1/chat/completions layer (what this gateway
|
||||
# actually calls) does not - it maps reasoning_effort/reasoning
|
||||
# instead (github.com/ollama/ollama issues #15288, #15293, #14820).
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeClient_openai_gateway)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_upstream_url"] = "http://127.0.0.1:11434/v1/chat/completions"
|
||||
rt = svc.runtime()
|
||||
run_async(
|
||||
rt.handle_chat(
|
||||
{"messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "ollama_reasoning_effort"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
body = rt._client.calls[-1][1]
|
||||
assert body["think"] is False
|
||||
assert body["reasoning_effort"] == "none"
|
||||
|
||||
|
||||
def test_hostile_upstream_model_header_is_ignored_end_to_end(local_db, monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeClientHostileModelHeader_openai_gateway
|
||||
|
||||
Reference in New Issue
Block a user