Add OpenCode Zen support, model health/stats dashboard, and gateway fallback fixes

AI gateway:
- Add a generic, admin-selectable `client_profile` field on gateway_providers
  (e.g. "opencode") so a provider needing special request headers (OpenCode
  Zen's client-identity spoofing) is configured like any other provider, not
  hardcoded by name.
- Track per-(provider, model) reliability/speed/latency health in memory,
  seeded from the existing gateway_usage_ledger at startup - purely
  observational, never influences routing.
- New Stats tab on /admin/gateway: request volume, latency, per-model
  breakdowns, and reliability weight, charted with a vendored Chart.js and
  devplace's own theme tokens.
- Record which model a failed request actually fell back to
  (fallback_used_route), surfaced in the Recent Failures table.
- Stop excluding context_length errors from fallback, and skip a primary
  attempt outright when its known context window is already too small for
  the estimated request size, going straight to the fallback.
- gateway_usage_ledger's provider/fallback_used_route columns and indexes
  are ensured centrally in database/schema.py's init_db(), the single point
  of truth for this table's schema.
- Non-OpenAI upstream routing and client-model passthrough; trust only the
  upstream's own X-Gateway-Model header for served-model attribution.

Devii agent:
- Fix a real lockup: plan/verify tools could be individually disabled via
  the admin tool toggles while still being required by the protocol gate,
  permanently bricking any task that needed tools. They can no longer be
  disabled, and the gate now also checks the tool is actually offered.
- Fix compaction being silently calibrated for a 1M-token model while
  running a much smaller one: context budget is now percentage-based and
  the summarizer's own request is sized to fit the real model.
- Give a specific, actionable retry message when plan()'s own arguments get
  cut off by the output limit, and tighten its schema to discourage
  overlong plans.

Other:
- Backup service: offload completed backups to a remote Hetzner Storage Box.
- Container manager: fix orphan blob leaks from sync races, add a two-phase
  plan/execute `system prune` CLI command.
- Admin gateway UI: replace the JS-rendered model/provider tables with
  server-rendered forms and pages.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01DhmEkvutuwtzFVcLbTrhdo
This commit is contained in:
2026-09-09 07:38:24 +02:00
co-authored by Claude Sonnet 5
parent b475e7d6ed
commit 569f1dcc64
67 changed files with 6151 additions and 814 deletions
@@ -5,6 +5,45 @@ from pydantic import ValidationError
from devplacepy.services.openai_gateway import routing as r
from devplacepy.services.openai_gateway.usage import pricing_from_cfg
from tests.conftest import run_async
class _FakeModelsResponse:
def __init__(self, status_code, payload=None):
self.status_code = status_code
self._payload = payload
def json(self):
return self._payload
class _FakeModelsClient:
def __init__(self, status_code, payload=None, error=None):
self.status_code = status_code
self.payload = payload
self.error = error
self.requested_url = None
self.headers = None
async def __aenter__(self):
return self
async def __aexit__(self, *exc_info):
return False
async def get(self, url):
self.requested_url = url
if self.error is not None:
raise self.error
return _FakeModelsResponse(self.status_code, self.payload)
def _patch_stealth_client(monkeypatch, fake_client):
def _factory(**kwargs):
fake_client.headers = kwargs.get("headers")
return fake_client
monkeypatch.setattr("devplacepy.stealth.stealth_async_client", _factory)
def _cleanup(providers, models):
@@ -262,6 +301,46 @@ def test_blank_provider_uses_default_upstream(local_db):
_cleanup([], ["ut-default"])
def test_provider_with_a_client_profile_gets_extra_headers(local_db):
r.provider_store.set(
r.ProviderIn(
name="zen",
base_url="https://opencode.ai/zen/v1/chat/completions",
api_key="public",
client_profile="opencode",
)
)
r.model_store.set(
r.ModelRouteIn(source_model="ut-zen", provider="zen", target_model="kimi-k3")
)
try:
overlay = r.chat_overlay("ut-zen", {})
headers = overlay["gateway_extra_request_headers"]
assert headers["x-opencode-client"]
assert "User-Agent" in headers
finally:
_cleanup(["zen"], ["ut-zen"])
def test_provider_without_a_client_profile_gets_no_extra_headers(local_db):
r.provider_store.set(
r.ProviderIn(name="plainprov", base_url="https://up.example/v1/chat/completions")
)
r.model_store.set(
r.ModelRouteIn(source_model="ut-plain", provider="plainprov", target_model="vendor/x")
)
try:
overlay = r.chat_overlay("ut-plain", {})
assert "gateway_extra_request_headers" not in overlay
finally:
_cleanup(["plainprov"], ["ut-plain"])
def test_client_profile_rejects_unknown_value():
with pytest.raises(ValidationError):
r.ProviderIn(name="badprofile", client_profile="not-a-real-profile")
def test_tier2_and_off_peak_fields_propagate_through_overlay(local_db):
r.model_store.set(
r.ModelRouteIn(
@@ -381,3 +460,112 @@ def test_seed_publishes_molodetz_aliases(local_db):
finally:
r.model_store.remove("molodetz")
r.model_store.remove("molodetz-pro")
def test_models_url_from_base_swaps_chat_completions():
assert (
r._models_url_from_base("https://x.example/v1/chat/completions")
== "https://x.example/v1/models"
)
def test_models_url_from_base_appends_when_no_chat_completions_suffix():
assert r._models_url_from_base("https://x.example/v1") == "https://x.example/v1/models"
assert r._models_url_from_base("https://x.example/v1/") == "https://x.example/v1/models"
def test_models_url_from_base_blank_is_blank():
assert r._models_url_from_base("") == ""
assert r._models_url_from_base(" ") == ""
def test_fetch_provider_models_returns_ids_on_success(local_db, monkeypatch):
r.provider_store.set(
r.ProviderIn(
name="probeprov",
base_url="https://x.example/v1/chat/completions",
api_key="sk-probe",
)
)
fake_client = _FakeModelsClient(200, {"data": [{"id": "vendor/a"}, {"id": "vendor/b"}]})
_patch_stealth_client(monkeypatch, fake_client)
try:
models = run_async(r.fetch_provider_models("probeprov"))
assert models == ["vendor/a", "vendor/b"]
assert fake_client.requested_url == "https://x.example/v1/models"
assert fake_client.headers == {"authorization": "Bearer sk-probe"}
finally:
r.provider_store.remove("probeprov")
def test_fetch_provider_models_uses_default_provider_when_blank(local_db, monkeypatch):
monkeypatch.setattr(
r,
"_default_provider_credentials",
lambda: ("https://default.example/v1/chat/completions", "sk-default"),
)
fake_client = _FakeModelsClient(200, {"data": [{"id": "default/model"}]})
_patch_stealth_client(monkeypatch, fake_client)
models = run_async(r.fetch_provider_models(""))
assert models == ["default/model"]
assert fake_client.requested_url == "https://default.example/v1/models"
def test_fetch_provider_models_none_for_unknown_provider(local_db):
assert run_async(r.fetch_provider_models("no-such-provider")) is None
def test_fetch_provider_models_none_on_non_200(local_db, monkeypatch):
r.provider_store.set(
r.ProviderIn(name="probeprov404", base_url="https://x.example/v1/chat/completions")
)
fake_client = _FakeModelsClient(404, {})
_patch_stealth_client(monkeypatch, fake_client)
try:
assert run_async(r.fetch_provider_models("probeprov404")) is None
finally:
r.provider_store.remove("probeprov404")
def test_fetch_provider_models_none_on_malformed_payload(local_db, monkeypatch):
r.provider_store.set(
r.ProviderIn(name="probeprovbad", base_url="https://x.example/v1/chat/completions")
)
fake_client = _FakeModelsClient(200, {"not_data": []})
_patch_stealth_client(monkeypatch, fake_client)
try:
assert run_async(r.fetch_provider_models("probeprovbad")) is None
finally:
r.provider_store.remove("probeprovbad")
def test_fetch_provider_models_none_on_empty_list(local_db, monkeypatch):
r.provider_store.set(
r.ProviderIn(name="probeprovempty", base_url="https://x.example/v1/chat/completions")
)
fake_client = _FakeModelsClient(200, {"data": []})
_patch_stealth_client(monkeypatch, fake_client)
try:
assert run_async(r.fetch_provider_models("probeprovempty")) is None
finally:
r.provider_store.remove("probeprovempty")
def test_fetch_provider_models_none_on_network_error(local_db, monkeypatch):
r.provider_store.set(
r.ProviderIn(name="probeprovdown", base_url="https://x.example/v1/chat/completions")
)
fake_client = _FakeModelsClient(200, error=RuntimeError("connection refused"))
_patch_stealth_client(monkeypatch, fake_client)
try:
assert run_async(r.fetch_provider_models("probeprovdown")) is None
finally:
r.provider_store.remove("probeprovdown")
def test_fetch_provider_models_none_when_provider_has_no_base_url(local_db):
r.provider_store.set(r.ProviderIn(name="probeprovnourl", base_url=""))
try:
assert run_async(r.fetch_provider_models("probeprovnourl")) is None
finally:
r.provider_store.remove("probeprovnourl")