Add OpenCode Zen support, model health/stats dashboard, and gateway fallback fixes

AI gateway:
- Add a generic, admin-selectable `client_profile` field on gateway_providers
  (e.g. "opencode") so a provider needing special request headers (OpenCode
  Zen's client-identity spoofing) is configured like any other provider, not
  hardcoded by name.
- Track per-(provider, model) reliability/speed/latency health in memory,
  seeded from the existing gateway_usage_ledger at startup - purely
  observational, never influences routing.
- New Stats tab on /admin/gateway: request volume, latency, per-model
  breakdowns, and reliability weight, charted with a vendored Chart.js and
  devplace's own theme tokens.
- Record which model a failed request actually fell back to
  (fallback_used_route), surfaced in the Recent Failures table.
- Stop excluding context_length errors from fallback, and skip a primary
  attempt outright when its known context window is already too small for
  the estimated request size, going straight to the fallback.
- gateway_usage_ledger's provider/fallback_used_route columns and indexes
  are ensured centrally in database/schema.py's init_db(), the single point
  of truth for this table's schema.
- Non-OpenAI upstream routing and client-model passthrough; trust only the
  upstream's own X-Gateway-Model header for served-model attribution.

Devii agent:
- Fix a real lockup: plan/verify tools could be individually disabled via
  the admin tool toggles while still being required by the protocol gate,
  permanently bricking any task that needed tools. They can no longer be
  disabled, and the gate now also checks the tool is actually offered.
- Fix compaction being silently calibrated for a 1M-token model while
  running a much smaller one: context budget is now percentage-based and
  the summarizer's own request is sized to fit the real model.
- Give a specific, actionable retry message when plan()'s own arguments get
  cut off by the output limit, and tighten its schema to discourage
  overlong plans.

Other:
- Backup service: offload completed backups to a remote Hetzner Storage Box.
- Container manager: fix orphan blob leaks from sync races, add a two-phase
  plan/execute `system prune` CLI command.
- Admin gateway UI: replace the JS-rendered model/provider tables with
  server-rendered forms and pages.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01DhmEkvutuwtzFVcLbTrhdo
This commit is contained in:
2026-09-09 07:38:24 +02:00
co-authored by Claude Sonnet 5
parent b475e7d6ed
commit 569f1dcc64
67 changed files with 6151 additions and 814 deletions
+107
View File
@@ -224,6 +224,113 @@ def test_model_route_fallback_rejects_self_reference(seeded_db):
assert response.json()["ok"] is False
def test_model_form_pages_require_admin(seeded_db):
assert (
requests.get(
f"{BASE_URL}/admin/gateway/models/new",
headers=JSON_gateway,
allow_redirects=False,
).status_code
== 401
)
key = member_key()
assert (
requests.get(
f"{BASE_URL}/admin/gateway/models/new",
headers={**JSON_gateway, "X-API-KEY": key},
allow_redirects=False,
).status_code
== 403
)
def test_model_add_edit_delete_via_page(seeded_db):
admin = admin_session(seeded_db)
source = _unique_gateway("modelpage")
new_page = admin.get(f"{BASE_URL}/admin/gateway/models/new")
assert new_page.status_code == 200
assert new_page.json()["is_edit"] is False
created = admin.post(
f"{BASE_URL}/admin/gateway/models/new",
data={
"source_model": source,
"target_model": "vendor/page",
"kind": "chat",
"price_output_per_m": "1.5",
"off_peak_start": "22:30",
"off_peak_end": "06:00",
"off_peak_discount_pct": "10",
"is_active": "1",
},
allow_redirects=False,
)
assert created.status_code == 302
assert created.headers["location"] == "/admin/gateway?tab=models"
edit_page = admin.get(f"{BASE_URL}/admin/gateway/models/{source}/edit")
assert edit_page.status_code == 200
form = edit_page.json()["form"]
assert form["target_model"] == "vendor/page"
assert form["off_peak_start"] == "22:30"
assert form["off_peak_end"] == "06:00"
edit_html = admin.get(f"{BASE_URL}/admin/gateway/models/{source}/edit", headers={"Accept": "text/html"})
assert f'value="{source}"' in edit_html.text
assert "readonly" in edit_html.text
updated = admin.post(
f"{BASE_URL}/admin/gateway/models/{source}/edit",
data={"target_model": "vendor/page2", "kind": "chat", "is_active": "1"},
allow_redirects=False,
)
assert updated.status_code == 302
relisted = admin.get(f"{BASE_URL}/admin/gateway/models").json()
row = next(m for m in relisted["models"] if m["source_model"] == source)
assert row["target_model"] == "vendor/page2"
deleted = admin.post(
f"{BASE_URL}/admin/gateway/models/{source}/delete", allow_redirects=False
)
assert deleted.status_code == 302
assert admin.get(f"{BASE_URL}/admin/gateway/models/{source}/edit").status_code == 404
def test_model_page_fallback_must_be_the_same_kind(seeded_db):
admin = admin_session(seeded_db)
embed_route = _unique_gateway("fbpage-embed")
chat_route = _unique_gateway("fbpage-chat")
admin.post(
f"{BASE_URL}/admin/gateway/models",
json={"source_model": embed_route, "target_model": "vendor/e", "kind": "embed"},
)
response = admin.post(
f"{BASE_URL}/admin/gateway/models/new",
data={
"source_model": chat_route,
"target_model": "vendor/c",
"kind": "chat",
"fallback_model": embed_route,
},
)
assert response.status_code == 400
assert "message" in response.json()["error"]
admin.delete(f"{BASE_URL}/admin/gateway/models/{embed_route}")
def test_model_edit_page_404_for_missing_model(seeded_db):
admin = admin_session(seeded_db)
missing = _unique_gateway("ghostmodel")
assert admin.get(f"{BASE_URL}/admin/gateway/models/{missing}/edit").status_code == 404
assert (
admin.post(f"{BASE_URL}/admin/gateway/models/{missing}/delete").status_code == 404
)
def test_model_route_validation(seeded_db):
admin = admin_session(seeded_db)
missing_target = admin.post(