forked from retoor/devplacepy
Add OpenCode Zen support, model health/stats dashboard, and gateway fallback fixes
AI gateway: - Add a generic, admin-selectable `client_profile` field on gateway_providers (e.g. "opencode") so a provider needing special request headers (OpenCode Zen's client-identity spoofing) is configured like any other provider, not hardcoded by name. - Track per-(provider, model) reliability/speed/latency health in memory, seeded from the existing gateway_usage_ledger at startup - purely observational, never influences routing. - New Stats tab on /admin/gateway: request volume, latency, per-model breakdowns, and reliability weight, charted with a vendored Chart.js and devplace's own theme tokens. - Record which model a failed request actually fell back to (fallback_used_route), surfaced in the Recent Failures table. - Stop excluding context_length errors from fallback, and skip a primary attempt outright when its known context window is already too small for the estimated request size, going straight to the fallback. - gateway_usage_ledger's provider/fallback_used_route columns and indexes are ensured centrally in database/schema.py's init_db(), the single point of truth for this table's schema. - Non-OpenAI upstream routing and client-model passthrough; trust only the upstream's own X-Gateway-Model header for served-model attribution. Devii agent: - Fix a real lockup: plan/verify tools could be individually disabled via the admin tool toggles while still being required by the protocol gate, permanently bricking any task that needed tools. They can no longer be disabled, and the gate now also checks the tool is actually offered. - Fix compaction being silently calibrated for a 1M-token model while running a much smaller one: context budget is now percentage-based and the summarizer's own request is sized to fit the real model. - Give a specific, actionable retry message when plan()'s own arguments get cut off by the output limit, and tighten its schema to discourage overlong plans. Other: - Backup service: offload completed backups to a remote Hetzner Storage Box. - Container manager: fix orphan blob leaks from sync races, add a two-phase plan/execute `system prune` CLI command. - Admin gateway UI: replace the JS-rendered model/provider tables with server-rendered forms and pages. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01DhmEkvutuwtzFVcLbTrhdo
This commit is contained in:
@@ -101,3 +101,36 @@ def test_primary_admin_tool_denied_for_regular_admin_is_audited(monkeypatch):
|
||||
assert event_key == "security.authz.denied"
|
||||
assert kwargs["metadata"]["tool"] == "db_list_tables"
|
||||
assert kwargs["actor_role"] == "admin"
|
||||
|
||||
|
||||
def test_admin_disabled_tool_is_refused_even_for_admin(monkeypatch):
|
||||
from devplacepy.services.devii import tool_prefs
|
||||
|
||||
recorder = _patch_recorder(monkeypatch)
|
||||
monkeypatch.setattr(
|
||||
tool_prefs, "disabled_tool_names", lambda: frozenset({"create_post"})
|
||||
)
|
||||
dispatcher = _bare_dispatcher("user", "admin-uid-9", is_admin=True, is_primary_admin=True)
|
||||
|
||||
result = run_async(dispatcher.dispatch("create_post", {"content": "hello"}))
|
||||
|
||||
payload = json.loads(result)
|
||||
assert payload["error"] == "tool_disabled"
|
||||
assert len(recorder.calls) == 1
|
||||
event_key, kwargs = recorder.calls[0]
|
||||
assert event_key == "security.authz.denied"
|
||||
assert kwargs["metadata"]["tool"] == "create_post"
|
||||
assert kwargs["metadata"]["reason"] == "disabled by administrator"
|
||||
|
||||
|
||||
def test_non_disabled_tool_unaffected_by_disabled_set(monkeypatch):
|
||||
from devplacepy.services.devii import tool_prefs
|
||||
|
||||
monkeypatch.setattr(
|
||||
tool_prefs, "disabled_tool_names", lambda: frozenset({"create_post"})
|
||||
)
|
||||
dispatcher = _bare_dispatcher("user", "admin-uid-9", is_admin=True, is_primary_admin=True)
|
||||
|
||||
result = run_async(dispatcher.dispatch("admin_list_users", {}))
|
||||
|
||||
assert json.loads(result).get("error") != "tool_disabled"
|
||||
|
||||
@@ -0,0 +1,154 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
import json
|
||||
|
||||
from devplacepy.services.devii.agentic.compaction import (
|
||||
compact_messages,
|
||||
find_compaction_split,
|
||||
is_context_length_error,
|
||||
)
|
||||
from devplacepy.services.devii.errors import LLMError
|
||||
from tests.conftest import run_async
|
||||
|
||||
|
||||
def _error(status, body):
|
||||
return LLMError("Model endpoint returned error", status=status, body=body)
|
||||
|
||||
|
||||
def test_openrouter_style_message_detected():
|
||||
body = json.dumps(
|
||||
{
|
||||
"error": {
|
||||
"message": (
|
||||
"This endpoint's maximum context length is 131072 tokens. "
|
||||
"However, you requested about 403355 tokens (349784 of text "
|
||||
"input, 53571 of tool input). Please reduce the length of "
|
||||
"either one, or use the context-compression plugin."
|
||||
),
|
||||
"code": 400,
|
||||
"metadata": {"provider_name": None},
|
||||
}
|
||||
}
|
||||
)
|
||||
assert is_context_length_error(_error(400, body)) is True
|
||||
|
||||
|
||||
def test_openai_style_code_detected():
|
||||
body = json.dumps(
|
||||
{
|
||||
"error": {
|
||||
"message": "This model's maximum context length is 128000 tokens.",
|
||||
"type": "invalid_request_error",
|
||||
"param": None,
|
||||
"code": "context_length_exceeded",
|
||||
}
|
||||
}
|
||||
)
|
||||
assert is_context_length_error(_error(400, body)) is True
|
||||
|
||||
|
||||
def test_unrelated_400_not_detected():
|
||||
body = json.dumps({"error": {"message": "Invalid API key.", "code": 400}})
|
||||
assert is_context_length_error(_error(400, body)) is False
|
||||
|
||||
|
||||
def test_non_400_status_not_detected_even_with_matching_text():
|
||||
body = json.dumps(
|
||||
{"error": {"message": "maximum context length is 131072 tokens"}}
|
||||
)
|
||||
assert is_context_length_error(_error(429, body)) is False
|
||||
|
||||
|
||||
def test_malformed_body_falls_back_to_phrase_match():
|
||||
truncated = "maximum context length is 131072 tokens, please reduce the length"
|
||||
assert is_context_length_error(_error(400, truncated)) is True
|
||||
|
||||
|
||||
def test_malformed_body_with_no_match_is_false():
|
||||
assert is_context_length_error(_error(400, "not valid json at all")) is False
|
||||
|
||||
|
||||
class _StubLlm:
|
||||
def __init__(self, summary="a concise summary of the earlier turns"):
|
||||
self.summary = summary
|
||||
self.calls = 0
|
||||
|
||||
async def summarize(self, prompt):
|
||||
self.calls += 1
|
||||
return self.summary
|
||||
|
||||
|
||||
def _tool_call_message(name="run_tool"):
|
||||
return {
|
||||
"role": "assistant",
|
||||
"content": "",
|
||||
"tool_calls": [
|
||||
{"id": "c1", "function": {"name": name, "arguments": "{}"}}
|
||||
],
|
||||
}
|
||||
|
||||
|
||||
def _tool_result_message(content="result"):
|
||||
return {"role": "tool", "tool_call_id": "c1", "name": "run_tool", "content": content}
|
||||
|
||||
|
||||
def _long_tool_heavy_conversation(rounds=20):
|
||||
messages = [
|
||||
{"role": "system", "content": "sys"},
|
||||
{"role": "user", "content": "start the long task"},
|
||||
]
|
||||
for i in range(rounds):
|
||||
messages.append(_tool_call_message())
|
||||
messages.append(_tool_result_message(f"result {i}" * 200))
|
||||
return messages
|
||||
|
||||
|
||||
def test_find_compaction_split_prefers_a_user_message():
|
||||
messages = [
|
||||
{"role": "system", "content": "sys"},
|
||||
{"role": "user", "content": "first"},
|
||||
{"role": "assistant", "content": "reply"},
|
||||
{"role": "user", "content": "second"},
|
||||
{"role": "assistant", "content": "reply2"},
|
||||
{"role": "user", "content": "third"},
|
||||
{"role": "assistant", "content": "reply3"},
|
||||
]
|
||||
split = find_compaction_split(messages, keep_tail=2)
|
||||
assert messages[split]["role"] == "user"
|
||||
|
||||
|
||||
def test_find_compaction_split_falls_back_to_a_non_tool_boundary_without_a_recent_user_message():
|
||||
messages = _long_tool_heavy_conversation(rounds=20)
|
||||
split = find_compaction_split(messages, keep_tail=4)
|
||||
assert split > 1
|
||||
assert messages[split].get("role") != "tool"
|
||||
|
||||
|
||||
def test_find_compaction_split_never_lands_inside_a_tool_result_run():
|
||||
messages = _long_tool_heavy_conversation(rounds=30)
|
||||
for keep_tail in (2, 3, 4, 5, 8, 10, 15):
|
||||
split = find_compaction_split(messages, keep_tail)
|
||||
assert messages[split].get("role") != "tool", (
|
||||
f"keep_tail={keep_tail} split at a tool message, orphaning its tool_calls"
|
||||
)
|
||||
|
||||
|
||||
def test_compact_messages_shrinks_a_tool_heavy_conversation_with_no_recent_user_message():
|
||||
messages = _long_tool_heavy_conversation(rounds=20)
|
||||
original_len = len(messages)
|
||||
llm = _StubLlm()
|
||||
compacted = run_async(compact_messages(llm, messages, keep_tail=4))
|
||||
assert llm.calls == 1
|
||||
assert len(compacted) < original_len
|
||||
assert compacted[0]["role"] == "system"
|
||||
assert "[compacted earlier turns]" in compacted[1]["content"]
|
||||
assert compacted[-1] == messages[-1]
|
||||
|
||||
|
||||
def test_compact_messages_tail_never_starts_with_a_dangling_tool_result():
|
||||
messages = _long_tool_heavy_conversation(rounds=25)
|
||||
llm = _StubLlm()
|
||||
compacted = run_async(compact_messages(llm, messages, keep_tail=6))
|
||||
tail = compacted[2:]
|
||||
assert tail
|
||||
assert tail[0].get("role") != "tool"
|
||||
@@ -1,7 +1,17 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
import dataclasses
|
||||
import json
|
||||
from devplacepy.services.devii.agentic.loop import _run_tool_call
|
||||
|
||||
from devplacepy.services.devii.agentic.compaction import context_size
|
||||
from devplacepy.services.devii.agentic.loop import (
|
||||
MAX_CONTEXT_OVERFLOW_RETRIES,
|
||||
_run_tool_call,
|
||||
react_loop,
|
||||
)
|
||||
from devplacepy.services.devii.agentic.state import AgentState
|
||||
from devplacepy.services.devii.config import load_settings
|
||||
from devplacepy.services.devii.errors import LLMError
|
||||
from tests.conftest import run_async
|
||||
class _FakeDispatcher:
|
||||
def __init__(self):
|
||||
@@ -54,3 +64,173 @@ def test_missing_arguments_defaults_to_empty_object():
|
||||
out = _run({"function": {"name": "auth_status"}}, dispatcher)
|
||||
assert out["status"] == "ok"
|
||||
assert dispatcher.calls == [("auth_status", {})]
|
||||
|
||||
|
||||
_CONTEXT_LENGTH_BODY = json.dumps(
|
||||
{
|
||||
"error": {
|
||||
"message": (
|
||||
"This endpoint's maximum context length is 131072 tokens. "
|
||||
"However, you requested about 403355 tokens. Please reduce "
|
||||
"the length."
|
||||
),
|
||||
"code": 400,
|
||||
}
|
||||
}
|
||||
)
|
||||
|
||||
|
||||
def _context_length_error():
|
||||
return LLMError(
|
||||
"Model endpoint returned 400: over limit", status=400, body=_CONTEXT_LENGTH_BODY
|
||||
)
|
||||
|
||||
|
||||
def _settings_for_test(keep_tail=4, threshold=10**9):
|
||||
return dataclasses.replace(
|
||||
load_settings(),
|
||||
context_compact_threshold=threshold,
|
||||
context_keep_tail=keep_tail,
|
||||
)
|
||||
|
||||
|
||||
def _long_message_history(count=10):
|
||||
messages = [{"role": "system", "content": "system prompt"}]
|
||||
for i in range(count):
|
||||
role = "user" if i % 2 == 0 else "assistant"
|
||||
messages.append({"role": role, "content": f"turn {i}"})
|
||||
return messages
|
||||
|
||||
|
||||
class _FakeLLM:
|
||||
def __init__(self, complete_results):
|
||||
self._complete_results = list(complete_results)
|
||||
self.complete_calls = 0
|
||||
self.summarize_calls = 0
|
||||
|
||||
async def complete(self, messages, tools):
|
||||
self.complete_calls += 1
|
||||
result = self._complete_results[
|
||||
min(self.complete_calls, len(self._complete_results)) - 1
|
||||
]
|
||||
if isinstance(result, Exception):
|
||||
raise result
|
||||
return result
|
||||
|
||||
async def summarize(self, text):
|
||||
self.summarize_calls += 1
|
||||
return "compacted summary"
|
||||
|
||||
|
||||
def test_context_overflow_triggers_compaction_and_retries():
|
||||
llm = _FakeLLM(
|
||||
[_context_length_error(), {"role": "assistant", "content": "Recovered answer"}]
|
||||
)
|
||||
messages = _long_message_history()
|
||||
result = run_async(
|
||||
react_loop(
|
||||
llm,
|
||||
_FakeDispatcher(),
|
||||
messages,
|
||||
tools=[],
|
||||
state=AgentState(),
|
||||
settings=_settings_for_test(),
|
||||
max_iterations=5,
|
||||
plan_required=False,
|
||||
verify_required=False,
|
||||
)
|
||||
)
|
||||
assert result == "Recovered answer"
|
||||
assert llm.complete_calls == 2
|
||||
assert llm.summarize_calls == 1
|
||||
assert not result.startswith("[model error]")
|
||||
|
||||
|
||||
def test_context_overflow_gives_up_after_max_retries_with_clear_message():
|
||||
llm = _FakeLLM([_context_length_error()])
|
||||
messages = _long_message_history()
|
||||
result = run_async(
|
||||
react_loop(
|
||||
llm,
|
||||
_FakeDispatcher(),
|
||||
messages,
|
||||
tools=[],
|
||||
state=AgentState(),
|
||||
settings=_settings_for_test(),
|
||||
max_iterations=10,
|
||||
plan_required=False,
|
||||
verify_required=False,
|
||||
)
|
||||
)
|
||||
assert result.startswith("[model error]")
|
||||
assert "context length" in result.lower() or "400" in result
|
||||
assert llm.complete_calls == MAX_CONTEXT_OVERFLOW_RETRIES + 1
|
||||
assert 0 < llm.summarize_calls <= MAX_CONTEXT_OVERFLOW_RETRIES
|
||||
|
||||
|
||||
def test_non_context_length_error_never_triggers_compaction():
|
||||
llm = _FakeLLM([LLMError("Model endpoint returned 500: boom", status=500, body="{}")])
|
||||
messages = _long_message_history()
|
||||
result = run_async(
|
||||
react_loop(
|
||||
llm,
|
||||
_FakeDispatcher(),
|
||||
messages,
|
||||
tools=[],
|
||||
state=AgentState(),
|
||||
settings=_settings_for_test(),
|
||||
max_iterations=5,
|
||||
plan_required=False,
|
||||
verify_required=False,
|
||||
)
|
||||
)
|
||||
assert result == "[model error] Model endpoint returned 500: boom"
|
||||
assert llm.complete_calls == 1
|
||||
assert llm.summarize_calls == 0
|
||||
|
||||
|
||||
class _FakeLLMWithRealLimit:
|
||||
def __init__(self, simulated_limit_chars):
|
||||
self.simulated_limit_chars = simulated_limit_chars
|
||||
self.complete_calls = 0
|
||||
self.summarize_calls = 0
|
||||
|
||||
async def complete(self, messages, tools):
|
||||
self.complete_calls += 1
|
||||
if context_size(messages) > self.simulated_limit_chars:
|
||||
raise _context_length_error()
|
||||
return {"role": "assistant", "content": "Recovered answer"}
|
||||
|
||||
async def summarize(self, text):
|
||||
self.summarize_calls += 1
|
||||
return "short summary"
|
||||
|
||||
|
||||
def test_one_oversized_tail_message_alone_still_recovers():
|
||||
giant = "X" * 300_000
|
||||
messages = _long_message_history(16)
|
||||
messages.append(
|
||||
{"role": "tool", "tool_call_id": "1", "name": "big_tool", "content": giant}
|
||||
)
|
||||
messages.append({"role": "user", "content": "please continue"})
|
||||
|
||||
llm = _FakeLLMWithRealLimit(simulated_limit_chars=30_000)
|
||||
result = run_async(
|
||||
react_loop(
|
||||
llm,
|
||||
_FakeDispatcher(),
|
||||
messages,
|
||||
tools=[],
|
||||
state=AgentState(),
|
||||
settings=_settings_for_test(keep_tail=4),
|
||||
max_iterations=10,
|
||||
plan_required=False,
|
||||
verify_required=False,
|
||||
)
|
||||
)
|
||||
|
||||
assert result == "Recovered answer"
|
||||
assert llm.complete_calls > MAX_CONTEXT_OVERFLOW_RETRIES - 1
|
||||
giant_message = next(m for m in messages if m.get("name") == "big_tool")
|
||||
assert len(giant_message["content"]) < len(giant)
|
||||
assert "truncated" in giant_message["content"]
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
# retoor <retoor@molodetz.nl>
|
||||
|
||||
from devplacepy.services.devii import tool_prefs
|
||||
|
||||
|
||||
def test_disabled_tool_names_round_trips(local_db):
|
||||
try:
|
||||
tool_prefs.set_disabled_tool_names({"create_post", "delete_post"})
|
||||
assert tool_prefs.disabled_tool_names() == frozenset({"create_post", "delete_post"})
|
||||
finally:
|
||||
tool_prefs.set_disabled_tool_names(set())
|
||||
|
||||
|
||||
def test_set_disabled_tool_names_drops_unknown_names(local_db):
|
||||
try:
|
||||
tool_prefs.set_disabled_tool_names({"create_post", "not_a_real_tool_xyz"})
|
||||
assert tool_prefs.disabled_tool_names() == frozenset({"create_post"})
|
||||
finally:
|
||||
tool_prefs.set_disabled_tool_names(set())
|
||||
|
||||
|
||||
def test_disabled_tool_names_empty_by_default(local_db):
|
||||
tool_prefs.set_disabled_tool_names(set())
|
||||
assert tool_prefs.disabled_tool_names() == frozenset()
|
||||
|
||||
|
||||
def test_filter_disabled_removes_matching_schemas():
|
||||
schemas = [
|
||||
{"function": {"name": "create_post"}},
|
||||
{"function": {"name": "list_posts"}},
|
||||
]
|
||||
result = tool_prefs.filter_disabled(schemas, disabled=frozenset({"create_post"}))
|
||||
assert [s["function"]["name"] for s in result] == ["list_posts"]
|
||||
|
||||
|
||||
def test_filter_disabled_is_a_no_op_when_nothing_disabled():
|
||||
schemas = [{"function": {"name": "create_post"}}]
|
||||
assert tool_prefs.filter_disabled(schemas, disabled=frozenset()) == schemas
|
||||
|
||||
|
||||
def test_group_overview_covers_every_group_and_marks_disabled(local_db):
|
||||
try:
|
||||
tool_prefs.set_disabled_tool_names({"create_post"})
|
||||
overview = tool_prefs.group_overview()
|
||||
assert len(overview) == len(tool_prefs.GROUPS)
|
||||
posts_group = next(g for g in overview if g["key"] == "posts")
|
||||
create_post_tool = next(t for t in posts_group["tools"] if t["name"] == "create_post")
|
||||
assert create_post_tool["disabled"] is True
|
||||
assert posts_group["enabled_count"] == posts_group["total_count"] - 1
|
||||
finally:
|
||||
tool_prefs.set_disabled_tool_names(set())
|
||||
|
||||
|
||||
def test_groups_by_tool_name_covers_every_action_in_every_group():
|
||||
total_actions = sum(len(actions) for actions in tool_prefs.GROUPS.values())
|
||||
assert len(tool_prefs.GROUPS_BY_TOOL_NAME) == total_actions
|
||||
Reference in New Issue
Block a user