forked from retoor/devplacepy
Add OpenCode Zen support, model health/stats dashboard, and gateway fallback fixes
AI gateway: - Add a generic, admin-selectable `client_profile` field on gateway_providers (e.g. "opencode") so a provider needing special request headers (OpenCode Zen's client-identity spoofing) is configured like any other provider, not hardcoded by name. - Track per-(provider, model) reliability/speed/latency health in memory, seeded from the existing gateway_usage_ledger at startup - purely observational, never influences routing. - New Stats tab on /admin/gateway: request volume, latency, per-model breakdowns, and reliability weight, charted with a vendored Chart.js and devplace's own theme tokens. - Record which model a failed request actually fell back to (fallback_used_route), surfaced in the Recent Failures table. - Stop excluding context_length errors from fallback, and skip a primary attempt outright when its known context window is already too small for the estimated request size, going straight to the fallback. - gateway_usage_ledger's provider/fallback_used_route columns and indexes are ensured centrally in database/schema.py's init_db(), the single point of truth for this table's schema. - Non-OpenAI upstream routing and client-model passthrough; trust only the upstream's own X-Gateway-Model header for served-model attribution. Devii agent: - Fix a real lockup: plan/verify tools could be individually disabled via the admin tool toggles while still being required by the protocol gate, permanently bricking any task that needed tools. They can no longer be disabled, and the gate now also checks the tool is actually offered. - Fix compaction being silently calibrated for a 1M-token model while running a much smaller one: context budget is now percentage-based and the summarizer's own request is sized to fit the real model. - Give a specific, actionable retry message when plan()'s own arguments get cut off by the output limit, and tighten its schema to discourage overlong plans. Other: - Backup service: offload completed backups to a remote Hetzner Storage Box. - Container manager: fix orphan blob leaks from sync races, add a two-phase plan/execute `system prune` CLI command. - Admin gateway UI: replace the JS-rendered model/provider tables with server-rendered forms and pages. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01DhmEkvutuwtzFVcLbTrhdo
This commit is contained in:
@@ -5,6 +5,45 @@ from pydantic import ValidationError
|
||||
|
||||
from devplacepy.services.openai_gateway import routing as r
|
||||
from devplacepy.services.openai_gateway.usage import pricing_from_cfg
|
||||
from tests.conftest import run_async
|
||||
|
||||
|
||||
class _FakeModelsResponse:
|
||||
def __init__(self, status_code, payload=None):
|
||||
self.status_code = status_code
|
||||
self._payload = payload
|
||||
|
||||
def json(self):
|
||||
return self._payload
|
||||
|
||||
|
||||
class _FakeModelsClient:
|
||||
def __init__(self, status_code, payload=None, error=None):
|
||||
self.status_code = status_code
|
||||
self.payload = payload
|
||||
self.error = error
|
||||
self.requested_url = None
|
||||
self.headers = None
|
||||
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *exc_info):
|
||||
return False
|
||||
|
||||
async def get(self, url):
|
||||
self.requested_url = url
|
||||
if self.error is not None:
|
||||
raise self.error
|
||||
return _FakeModelsResponse(self.status_code, self.payload)
|
||||
|
||||
|
||||
def _patch_stealth_client(monkeypatch, fake_client):
|
||||
def _factory(**kwargs):
|
||||
fake_client.headers = kwargs.get("headers")
|
||||
return fake_client
|
||||
|
||||
monkeypatch.setattr("devplacepy.stealth.stealth_async_client", _factory)
|
||||
|
||||
|
||||
def _cleanup(providers, models):
|
||||
@@ -262,6 +301,46 @@ def test_blank_provider_uses_default_upstream(local_db):
|
||||
_cleanup([], ["ut-default"])
|
||||
|
||||
|
||||
def test_provider_with_a_client_profile_gets_extra_headers(local_db):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(
|
||||
name="zen",
|
||||
base_url="https://opencode.ai/zen/v1/chat/completions",
|
||||
api_key="public",
|
||||
client_profile="opencode",
|
||||
)
|
||||
)
|
||||
r.model_store.set(
|
||||
r.ModelRouteIn(source_model="ut-zen", provider="zen", target_model="kimi-k3")
|
||||
)
|
||||
try:
|
||||
overlay = r.chat_overlay("ut-zen", {})
|
||||
headers = overlay["gateway_extra_request_headers"]
|
||||
assert headers["x-opencode-client"]
|
||||
assert "User-Agent" in headers
|
||||
finally:
|
||||
_cleanup(["zen"], ["ut-zen"])
|
||||
|
||||
|
||||
def test_provider_without_a_client_profile_gets_no_extra_headers(local_db):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(name="plainprov", base_url="https://up.example/v1/chat/completions")
|
||||
)
|
||||
r.model_store.set(
|
||||
r.ModelRouteIn(source_model="ut-plain", provider="plainprov", target_model="vendor/x")
|
||||
)
|
||||
try:
|
||||
overlay = r.chat_overlay("ut-plain", {})
|
||||
assert "gateway_extra_request_headers" not in overlay
|
||||
finally:
|
||||
_cleanup(["plainprov"], ["ut-plain"])
|
||||
|
||||
|
||||
def test_client_profile_rejects_unknown_value():
|
||||
with pytest.raises(ValidationError):
|
||||
r.ProviderIn(name="badprofile", client_profile="not-a-real-profile")
|
||||
|
||||
|
||||
def test_tier2_and_off_peak_fields_propagate_through_overlay(local_db):
|
||||
r.model_store.set(
|
||||
r.ModelRouteIn(
|
||||
@@ -381,3 +460,112 @@ def test_seed_publishes_molodetz_aliases(local_db):
|
||||
finally:
|
||||
r.model_store.remove("molodetz")
|
||||
r.model_store.remove("molodetz-pro")
|
||||
|
||||
|
||||
def test_models_url_from_base_swaps_chat_completions():
|
||||
assert (
|
||||
r._models_url_from_base("https://x.example/v1/chat/completions")
|
||||
== "https://x.example/v1/models"
|
||||
)
|
||||
|
||||
|
||||
def test_models_url_from_base_appends_when_no_chat_completions_suffix():
|
||||
assert r._models_url_from_base("https://x.example/v1") == "https://x.example/v1/models"
|
||||
assert r._models_url_from_base("https://x.example/v1/") == "https://x.example/v1/models"
|
||||
|
||||
|
||||
def test_models_url_from_base_blank_is_blank():
|
||||
assert r._models_url_from_base("") == ""
|
||||
assert r._models_url_from_base(" ") == ""
|
||||
|
||||
|
||||
def test_fetch_provider_models_returns_ids_on_success(local_db, monkeypatch):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(
|
||||
name="probeprov",
|
||||
base_url="https://x.example/v1/chat/completions",
|
||||
api_key="sk-probe",
|
||||
)
|
||||
)
|
||||
fake_client = _FakeModelsClient(200, {"data": [{"id": "vendor/a"}, {"id": "vendor/b"}]})
|
||||
_patch_stealth_client(monkeypatch, fake_client)
|
||||
try:
|
||||
models = run_async(r.fetch_provider_models("probeprov"))
|
||||
assert models == ["vendor/a", "vendor/b"]
|
||||
assert fake_client.requested_url == "https://x.example/v1/models"
|
||||
assert fake_client.headers == {"authorization": "Bearer sk-probe"}
|
||||
finally:
|
||||
r.provider_store.remove("probeprov")
|
||||
|
||||
|
||||
def test_fetch_provider_models_uses_default_provider_when_blank(local_db, monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
r,
|
||||
"_default_provider_credentials",
|
||||
lambda: ("https://default.example/v1/chat/completions", "sk-default"),
|
||||
)
|
||||
fake_client = _FakeModelsClient(200, {"data": [{"id": "default/model"}]})
|
||||
_patch_stealth_client(monkeypatch, fake_client)
|
||||
models = run_async(r.fetch_provider_models(""))
|
||||
assert models == ["default/model"]
|
||||
assert fake_client.requested_url == "https://default.example/v1/models"
|
||||
|
||||
|
||||
def test_fetch_provider_models_none_for_unknown_provider(local_db):
|
||||
assert run_async(r.fetch_provider_models("no-such-provider")) is None
|
||||
|
||||
|
||||
def test_fetch_provider_models_none_on_non_200(local_db, monkeypatch):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(name="probeprov404", base_url="https://x.example/v1/chat/completions")
|
||||
)
|
||||
fake_client = _FakeModelsClient(404, {})
|
||||
_patch_stealth_client(monkeypatch, fake_client)
|
||||
try:
|
||||
assert run_async(r.fetch_provider_models("probeprov404")) is None
|
||||
finally:
|
||||
r.provider_store.remove("probeprov404")
|
||||
|
||||
|
||||
def test_fetch_provider_models_none_on_malformed_payload(local_db, monkeypatch):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(name="probeprovbad", base_url="https://x.example/v1/chat/completions")
|
||||
)
|
||||
fake_client = _FakeModelsClient(200, {"not_data": []})
|
||||
_patch_stealth_client(monkeypatch, fake_client)
|
||||
try:
|
||||
assert run_async(r.fetch_provider_models("probeprovbad")) is None
|
||||
finally:
|
||||
r.provider_store.remove("probeprovbad")
|
||||
|
||||
|
||||
def test_fetch_provider_models_none_on_empty_list(local_db, monkeypatch):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(name="probeprovempty", base_url="https://x.example/v1/chat/completions")
|
||||
)
|
||||
fake_client = _FakeModelsClient(200, {"data": []})
|
||||
_patch_stealth_client(monkeypatch, fake_client)
|
||||
try:
|
||||
assert run_async(r.fetch_provider_models("probeprovempty")) is None
|
||||
finally:
|
||||
r.provider_store.remove("probeprovempty")
|
||||
|
||||
|
||||
def test_fetch_provider_models_none_on_network_error(local_db, monkeypatch):
|
||||
r.provider_store.set(
|
||||
r.ProviderIn(name="probeprovdown", base_url="https://x.example/v1/chat/completions")
|
||||
)
|
||||
fake_client = _FakeModelsClient(200, error=RuntimeError("connection refused"))
|
||||
_patch_stealth_client(monkeypatch, fake_client)
|
||||
try:
|
||||
assert run_async(r.fetch_provider_models("probeprovdown")) is None
|
||||
finally:
|
||||
r.provider_store.remove("probeprovdown")
|
||||
|
||||
|
||||
def test_fetch_provider_models_none_when_provider_has_no_base_url(local_db):
|
||||
r.provider_store.set(r.ProviderIn(name="probeprovnourl", base_url=""))
|
||||
try:
|
||||
assert run_async(r.fetch_provider_models("probeprovnourl")) is None
|
||||
finally:
|
||||
r.provider_store.remove("probeprovnourl")
|
||||
|
||||
Reference in New Issue
Block a user