forked from retoor/devplacepy
AI gateway: - Add a generic, admin-selectable `client_profile` field on gateway_providers (e.g. "opencode") so a provider needing special request headers (OpenCode Zen's client-identity spoofing) is configured like any other provider, not hardcoded by name. - Track per-(provider, model) reliability/speed/latency health in memory, seeded from the existing gateway_usage_ledger at startup - purely observational, never influences routing. - New Stats tab on /admin/gateway: request volume, latency, per-model breakdowns, and reliability weight, charted with a vendored Chart.js and devplace's own theme tokens. - Record which model a failed request actually fell back to (fallback_used_route), surfaced in the Recent Failures table. - Stop excluding context_length errors from fallback, and skip a primary attempt outright when its known context window is already too small for the estimated request size, going straight to the fallback. - gateway_usage_ledger's provider/fallback_used_route columns and indexes are ensured centrally in database/schema.py's init_db(), the single point of truth for this table's schema. - Non-OpenAI upstream routing and client-model passthrough; trust only the upstream's own X-Gateway-Model header for served-model attribution. Devii agent: - Fix a real lockup: plan/verify tools could be individually disabled via the admin tool toggles while still being required by the protocol gate, permanently bricking any task that needed tools. They can no longer be disabled, and the gate now also checks the tool is actually offered. - Fix compaction being silently calibrated for a 1M-token model while running a much smaller one: context budget is now percentage-based and the summarizer's own request is sized to fit the real model. - Give a specific, actionable retry message when plan()'s own arguments get cut off by the output limit, and tighten its schema to discourage overlong plans. Other: - Backup service: offload completed backups to a remote Hetzner Storage Box. - Container manager: fix orphan blob leaks from sync races, add a two-phase plan/execute `system prune` CLI command. - Admin gateway UI: replace the JS-rendered model/provider tables with server-rendered forms and pages. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01DhmEkvutuwtzFVcLbTrhdo
805 lines
29 KiB
Python
805 lines
29 KiB
Python
# retoor <retoor@molodetz.nl>
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import logging
|
|
from dataclasses import dataclass
|
|
from datetime import datetime, timedelta, timezone
|
|
from typing import Any, Optional
|
|
|
|
import httpx
|
|
|
|
from devplacepy.database import db, get_table
|
|
from devplacepy.services.openai_gateway import config
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
GATEWAY_LEDGER = "gateway_usage_ledger"
|
|
GATEWAY_CONCURRENCY = "gateway_concurrency_samples"
|
|
PER_MILLION = 1_000_000
|
|
|
|
|
|
def _now() -> datetime:
|
|
return datetime.now(timezone.utc)
|
|
|
|
|
|
def _iso(moment: datetime) -> str:
|
|
return moment.isoformat(timespec="microseconds")
|
|
|
|
|
|
@dataclass(frozen=True)
|
|
class Pricing:
|
|
chat_cache_hit_per_m: float
|
|
chat_cache_miss_per_m: float
|
|
chat_output_per_m: float
|
|
vision_input_per_m: float
|
|
vision_output_per_m: float
|
|
embed_input_per_m: float
|
|
image_per_call: float = 0.0
|
|
chat_cache_hit_per_m_tier2: Optional[float] = None
|
|
chat_cache_miss_per_m_tier2: Optional[float] = None
|
|
chat_output_per_m_tier2: Optional[float] = None
|
|
vision_input_per_m_tier2: Optional[float] = None
|
|
vision_output_per_m_tier2: Optional[float] = None
|
|
embed_input_per_m_tier2: Optional[float] = None
|
|
context_tier_threshold_tokens: int = 0
|
|
off_peak_start_minute: Optional[int] = None
|
|
off_peak_end_minute: Optional[int] = None
|
|
off_peak_discount_pct: float = 0.0
|
|
|
|
|
|
def _cfg_float(cfg: dict, key: str) -> Optional[float]:
|
|
value = cfg.get(key)
|
|
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
return float(value)
|
|
return None
|
|
|
|
|
|
def _cfg_int(cfg: dict, key: str) -> Optional[int]:
|
|
value = cfg.get(key)
|
|
if isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
return int(value)
|
|
return None
|
|
|
|
|
|
def pricing_from_cfg(cfg: dict) -> Pricing:
|
|
return Pricing(
|
|
chat_cache_hit_per_m=float(
|
|
cfg.get(
|
|
"gateway_price_cache_hit_per_m", config.PRICE_CACHE_HIT_PER_M_DEFAULT
|
|
)
|
|
),
|
|
chat_cache_miss_per_m=float(
|
|
cfg.get(
|
|
"gateway_price_cache_miss_per_m", config.PRICE_CACHE_MISS_PER_M_DEFAULT
|
|
)
|
|
),
|
|
chat_output_per_m=float(
|
|
cfg.get("gateway_price_output_per_m", config.PRICE_OUTPUT_PER_M_DEFAULT)
|
|
),
|
|
vision_input_per_m=float(
|
|
cfg.get(
|
|
"gateway_vision_price_input_per_m",
|
|
config.VISION_PRICE_INPUT_PER_M_DEFAULT,
|
|
)
|
|
),
|
|
vision_output_per_m=float(
|
|
cfg.get(
|
|
"gateway_vision_price_output_per_m",
|
|
config.VISION_PRICE_OUTPUT_PER_M_DEFAULT,
|
|
)
|
|
),
|
|
embed_input_per_m=float(
|
|
cfg.get(
|
|
"gateway_embed_price_input_per_m",
|
|
config.EMBED_PRICE_INPUT_PER_M_DEFAULT,
|
|
)
|
|
),
|
|
image_per_call=float(
|
|
cfg.get(
|
|
"gateway_image_price_per_call",
|
|
config.IMAGE_PRICE_PER_CALL_DEFAULT,
|
|
)
|
|
),
|
|
chat_cache_hit_per_m_tier2=_cfg_float(
|
|
cfg, "gateway_price_cache_hit_per_m_tier2"
|
|
),
|
|
chat_cache_miss_per_m_tier2=_cfg_float(
|
|
cfg, "gateway_price_cache_miss_per_m_tier2"
|
|
),
|
|
chat_output_per_m_tier2=_cfg_float(cfg, "gateway_price_output_per_m_tier2"),
|
|
vision_input_per_m_tier2=_cfg_float(
|
|
cfg, "gateway_vision_price_input_per_m_tier2"
|
|
),
|
|
vision_output_per_m_tier2=_cfg_float(
|
|
cfg, "gateway_vision_price_output_per_m_tier2"
|
|
),
|
|
embed_input_per_m_tier2=_cfg_float(
|
|
cfg, "gateway_embed_price_input_per_m_tier2"
|
|
),
|
|
context_tier_threshold_tokens=int(
|
|
cfg.get("gateway_context_tier_threshold_tokens", 0) or 0
|
|
),
|
|
off_peak_start_minute=_cfg_int(cfg, "gateway_off_peak_start_minute"),
|
|
off_peak_end_minute=_cfg_int(cfg, "gateway_off_peak_end_minute"),
|
|
off_peak_discount_pct=float(cfg.get("gateway_off_peak_discount_pct", 0.0) or 0.0),
|
|
)
|
|
|
|
|
|
def parse_context_map(raw: Any) -> dict[str, int]:
|
|
if isinstance(raw, dict):
|
|
return {str(k): int(v) for k, v in raw.items()}
|
|
if not raw:
|
|
return dict(config.MODEL_CONTEXT_MAP_DEFAULT)
|
|
try:
|
|
loaded = json.loads(raw)
|
|
if isinstance(loaded, dict):
|
|
return {str(k): int(v) for k, v in loaded.items()}
|
|
except (ValueError, TypeError):
|
|
logger.warning("Invalid gateway_model_context_map, using defaults")
|
|
return dict(config.MODEL_CONTEXT_MAP_DEFAULT)
|
|
|
|
|
|
def extract_image_usage(data: Optional[dict]) -> dict:
|
|
data = data or {}
|
|
usage = data.get("usage") if isinstance(data.get("usage"), dict) else {}
|
|
cost = usage.get("cost")
|
|
if cost is None:
|
|
cost = data.get("cost")
|
|
result: dict = {}
|
|
if isinstance(cost, (int, float)) and not isinstance(cost, bool):
|
|
result["cost"] = float(cost)
|
|
return result
|
|
|
|
|
|
def normalize_usage(usage: Optional[dict]) -> dict:
|
|
usage = usage or {}
|
|
prompt = int(usage.get("prompt_tokens", 0) or 0)
|
|
completion = int(usage.get("completion_tokens", 0) or 0)
|
|
total = int(usage.get("total_tokens", prompt + completion) or 0)
|
|
|
|
hit = usage.get("prompt_cache_hit_tokens")
|
|
if hit is None:
|
|
details = usage.get("prompt_tokens_details") or {}
|
|
hit = details.get("cached_tokens", 0)
|
|
hit = int(hit or 0)
|
|
|
|
miss = usage.get("prompt_cache_miss_tokens")
|
|
if miss is None:
|
|
miss = max(prompt - hit, 0)
|
|
miss = int(miss or 0)
|
|
|
|
completion_details = usage.get("completion_tokens_details") or {}
|
|
reasoning = int(completion_details.get("reasoning_tokens", 0) or 0)
|
|
return {
|
|
"prompt": prompt,
|
|
"completion": completion,
|
|
"total": total,
|
|
"cache_hit": hit,
|
|
"cache_miss": miss,
|
|
"reasoning": reasoning,
|
|
}
|
|
|
|
|
|
def _off_peak_active(pricing: Pricing, now: Optional[datetime] = None) -> bool:
|
|
if pricing.off_peak_start_minute is None or pricing.off_peak_end_minute is None:
|
|
return False
|
|
moment = now or _now()
|
|
minute_of_day = moment.hour * 60 + moment.minute
|
|
start, end = pricing.off_peak_start_minute, pricing.off_peak_end_minute
|
|
if start == end:
|
|
return True
|
|
if start < end:
|
|
return start <= minute_of_day < end
|
|
return minute_of_day >= start or minute_of_day < end
|
|
|
|
|
|
def _tiered_rate(base: float, tier2: Optional[float], use_tier2: bool) -> float:
|
|
if use_tier2 and tier2 is not None:
|
|
return tier2
|
|
return base
|
|
|
|
|
|
def _effective_rate(
|
|
base: float,
|
|
tier2: Optional[float],
|
|
use_tier2: bool,
|
|
pricing: Pricing,
|
|
off_peak: bool,
|
|
) -> float:
|
|
rate = _tiered_rate(base, tier2, use_tier2)
|
|
if off_peak and pricing.off_peak_discount_pct > 0:
|
|
rate = rate * (1 - min(pricing.off_peak_discount_pct, 100.0) / 100.0)
|
|
return rate
|
|
|
|
|
|
def compute_cost(
|
|
usage: dict,
|
|
norm: dict,
|
|
pricing: Pricing,
|
|
backend: str,
|
|
now: Optional[datetime] = None,
|
|
) -> tuple[float, float, float, bool]:
|
|
threshold = pricing.context_tier_threshold_tokens
|
|
use_tier2 = bool(threshold > 0 and norm["prompt"] > threshold)
|
|
off_peak = _off_peak_active(pricing, now)
|
|
if backend == "vision":
|
|
input_rate = _effective_rate(
|
|
pricing.vision_input_per_m,
|
|
pricing.vision_input_per_m_tier2,
|
|
use_tier2,
|
|
pricing,
|
|
off_peak,
|
|
)
|
|
output_rate = _effective_rate(
|
|
pricing.vision_output_per_m,
|
|
pricing.vision_output_per_m_tier2,
|
|
use_tier2,
|
|
pricing,
|
|
off_peak,
|
|
)
|
|
input_cost = norm["prompt"] / PER_MILLION * input_rate
|
|
output_cost = norm["completion"] / PER_MILLION * output_rate
|
|
elif backend == "embed":
|
|
input_rate = _effective_rate(
|
|
pricing.embed_input_per_m,
|
|
pricing.embed_input_per_m_tier2,
|
|
use_tier2,
|
|
pricing,
|
|
off_peak,
|
|
)
|
|
input_cost = norm["prompt"] / PER_MILLION * input_rate
|
|
output_cost = 0.0
|
|
elif backend == "image":
|
|
rate = pricing.image_per_call
|
|
if off_peak and pricing.off_peak_discount_pct > 0:
|
|
rate = rate * (1 - min(pricing.off_peak_discount_pct, 100.0) / 100.0)
|
|
input_cost = rate
|
|
output_cost = 0.0
|
|
else:
|
|
hit_rate = _effective_rate(
|
|
pricing.chat_cache_hit_per_m,
|
|
pricing.chat_cache_hit_per_m_tier2,
|
|
use_tier2,
|
|
pricing,
|
|
off_peak,
|
|
)
|
|
miss_rate = _effective_rate(
|
|
pricing.chat_cache_miss_per_m,
|
|
pricing.chat_cache_miss_per_m_tier2,
|
|
use_tier2,
|
|
pricing,
|
|
off_peak,
|
|
)
|
|
output_rate = _effective_rate(
|
|
pricing.chat_output_per_m,
|
|
pricing.chat_output_per_m_tier2,
|
|
use_tier2,
|
|
pricing,
|
|
off_peak,
|
|
)
|
|
input_cost = (
|
|
norm["cache_hit"] / PER_MILLION * hit_rate
|
|
+ norm["cache_miss"] / PER_MILLION * miss_rate
|
|
)
|
|
output_cost = norm["completion"] / PER_MILLION * output_rate
|
|
native = usage.get("cost") if isinstance(usage, dict) else None
|
|
if isinstance(native, (int, float)) and not isinstance(native, bool):
|
|
total = max(0.0, float(native))
|
|
modeled = input_cost + output_cost
|
|
if modeled > 0:
|
|
input_share = input_cost / modeled
|
|
elif norm["prompt"] + norm["completion"] > 0:
|
|
input_share = norm["prompt"] / (norm["prompt"] + norm["completion"])
|
|
else:
|
|
input_share = 0.0
|
|
native_input = total * input_share
|
|
return total, native_input, total - native_input, True
|
|
return input_cost + output_cost, input_cost, output_cost, False
|
|
|
|
|
|
def context_utilization(
|
|
total_tokens: int, model: str, context_map: dict
|
|
) -> tuple[Optional[int], Optional[float]]:
|
|
window = context_map.get(model)
|
|
if not window or window <= 0:
|
|
return None, None
|
|
return int(window), round(total_tokens / window, 4)
|
|
|
|
|
|
def extract_params(body: Any) -> dict:
|
|
if not isinstance(body, dict):
|
|
return {
|
|
"requested_model": "",
|
|
"stream_requested": False,
|
|
"temperature": None,
|
|
"top_p": None,
|
|
"max_tokens": None,
|
|
"has_tools": False,
|
|
}
|
|
temperature = body.get("temperature")
|
|
top_p = body.get("top_p")
|
|
max_tokens = body.get("max_tokens")
|
|
if max_tokens is None:
|
|
max_tokens = body.get("max_completion_tokens")
|
|
return {
|
|
"requested_model": body.get("model") or "",
|
|
"stream_requested": bool(body.get("stream")),
|
|
"temperature": float(temperature)
|
|
if isinstance(temperature, (int, float)) and not isinstance(temperature, bool)
|
|
else None,
|
|
"top_p": float(top_p)
|
|
if isinstance(top_p, (int, float)) and not isinstance(top_p, bool)
|
|
else None,
|
|
"max_tokens": int(max_tokens)
|
|
if isinstance(max_tokens, (int, float)) and not isinstance(max_tokens, bool)
|
|
else None,
|
|
"has_tools": bool(body.get("tools") or body.get("functions")),
|
|
}
|
|
|
|
|
|
def classify_error(
|
|
status_code: int, exc: Optional[Exception] = None, message: str = ""
|
|
) -> str:
|
|
if exc is not None:
|
|
if isinstance(exc, httpx.TimeoutException):
|
|
return "timeout"
|
|
return "gateway"
|
|
if status_code == 429:
|
|
return "rate_limit"
|
|
if status_code in (401, 403):
|
|
return "auth"
|
|
if status_code == 402:
|
|
return "insufficient_balance"
|
|
if status_code == 404:
|
|
return "model_not_found"
|
|
if status_code in (400, 422):
|
|
lowered = (message or "").lower()
|
|
if "context" in lowered or "maximum" in lowered or "too long" in lowered:
|
|
return "context_length"
|
|
return "bad_request"
|
|
if status_code and status_code >= 500:
|
|
return "upstream_error"
|
|
return "gateway"
|
|
|
|
|
|
def embedded_error_message(data: Any) -> Optional[str]:
|
|
"""Detect a provider-side failure reported with HTTP 200.
|
|
|
|
OpenRouter (and providers it fronts) can answer 200 OK with the failure
|
|
embedded in the JSON body instead of a non-2xx status - documented at
|
|
openrouter.ai/docs/api-reference/errors: "the HTTP response status is
|
|
matched with the error.code for validation/credits issues, otherwise
|
|
returns 200 OK with the error embedded in the response body." A caller
|
|
that only checks status_code would silently record this as a success
|
|
with no usage and never trigger a retry or model fallback.
|
|
"""
|
|
if not isinstance(data, dict):
|
|
return None
|
|
error = data.get("error")
|
|
if not error:
|
|
return None
|
|
if "choices" in data or "data" in data:
|
|
# A well-formed choices/data payload alongside a per-item `error`
|
|
# field (e.g. a single failed choice) is not a whole-response failure.
|
|
return None
|
|
if isinstance(error, dict):
|
|
message = error.get("message") or error.get("code") or "unknown error"
|
|
return str(message)[:500]
|
|
return str(error)[:500]
|
|
|
|
|
|
def audit_actor_for(owner_kind: str, owner_id: str) -> tuple[str, Optional[str], str]:
|
|
actor_kind = (
|
|
"guest"
|
|
if owner_kind == "guest"
|
|
else ("user" if owner_kind in ("user", "admin") else "system")
|
|
)
|
|
actor_uid = owner_id if actor_kind == "user" else None
|
|
actor_role = (
|
|
"admin"
|
|
if owner_kind == "admin"
|
|
else (actor_kind if actor_kind != "user" else "member")
|
|
)
|
|
return actor_kind, actor_uid, actor_role
|
|
|
|
|
|
def usage_response_headers(row: Optional[dict]) -> dict:
|
|
if not row:
|
|
return {}
|
|
headers = {
|
|
"X-Gateway-Model": str(row.get("model") or ""),
|
|
"X-Gateway-Backend": str(row.get("backend") or ""),
|
|
"X-Gateway-Prompt-Tokens": str(int(row.get("prompt_tokens") or 0)),
|
|
"X-Gateway-Completion-Tokens": str(int(row.get("completion_tokens") or 0)),
|
|
"X-Gateway-Total-Tokens": str(int(row.get("total_tokens") or 0)),
|
|
"X-Gateway-Cache-Hit-Tokens": str(int(row.get("cache_hit_tokens") or 0)),
|
|
"X-Gateway-Cache-Miss-Tokens": str(int(row.get("cache_miss_tokens") or 0)),
|
|
"X-Gateway-Reasoning-Tokens": str(int(row.get("reasoning_tokens") or 0)),
|
|
"X-Gateway-Cost-USD": f"{float(row.get('cost_usd') or 0.0):.8f}",
|
|
"X-Gateway-Input-Cost-USD": f"{float(row.get('input_cost_usd') or 0.0):.8f}",
|
|
"X-Gateway-Output-Cost-USD": f"{float(row.get('output_cost_usd') or 0.0):.8f}",
|
|
"X-Gateway-Cost-Native": "1" if row.get("native_cost") else "0",
|
|
"X-Gateway-Tokens-Per-Second": str(row.get("tokens_per_second") or 0),
|
|
"X-Gateway-Upstream-Latency-Ms": str(row.get("upstream_latency_ms") or 0),
|
|
"X-Gateway-Total-Latency-Ms": str(row.get("total_latency_ms") or 0),
|
|
"X-Gateway-Gateway-Overhead-Ms": str(row.get("gateway_overhead_ms") or 0),
|
|
"X-Gateway-Queue-Wait-Ms": str(row.get("queue_wait_ms") or 0),
|
|
"X-Gateway-Connect-Ms": str(row.get("connect_ms") or 0),
|
|
}
|
|
if row.get("context_window"):
|
|
headers["X-Gateway-Context-Window"] = str(int(row["context_window"]))
|
|
if row.get("context_utilization") is not None:
|
|
headers["X-Gateway-Context-Utilization"] = str(row["context_utilization"])
|
|
headers["X-App-Reference"] = str(row.get("app_reference") or "default")
|
|
return headers
|
|
|
|
|
|
UPSTREAM_MODEL_HEADER_MAX_LEN = 200
|
|
|
|
|
|
def upstream_reported_model(headers: Any) -> Optional[str]:
|
|
"""Extract a served-model name an upstream (which may itself be another
|
|
DevPlace-style gateway) reported on its own X-Gateway-Model response
|
|
header, if present and well-formed.
|
|
|
|
This is the ONE upstream response header this gateway ever trusts and
|
|
relays: model name is the single field an upstream can know better than
|
|
we do (it may have resolved an alias/fallback on its own side). Every
|
|
other X-Gateway-* figure (cost, tokens, latency, context) stays our own
|
|
measurement regardless of what an upstream claims - see usage_response_headers.
|
|
|
|
An upstream response is untrusted input, so this never raises: a missing,
|
|
malformed, oversized, or control-character-laced value is treated as
|
|
absent rather than propagated or allowed to break our own response.
|
|
"""
|
|
if not headers:
|
|
return None
|
|
value = None
|
|
try:
|
|
value = headers.get("X-Gateway-Model")
|
|
except Exception:
|
|
value = None
|
|
if value is None:
|
|
try:
|
|
for key, candidate in headers.items():
|
|
if isinstance(key, str) and key.lower() == "x-gateway-model":
|
|
value = candidate
|
|
break
|
|
except Exception:
|
|
value = None
|
|
if not isinstance(value, str):
|
|
return None
|
|
value = value.strip()
|
|
if not value or len(value) > UPSTREAM_MODEL_HEADER_MAX_LEN:
|
|
return None
|
|
if any(ch in value for ch in ("\r", "\n", "\x00")):
|
|
return None
|
|
return value
|
|
|
|
|
|
def parse_usage_headers(headers) -> Optional[dict]:
|
|
if not headers or "X-Gateway-Cost-USD" not in headers:
|
|
return None
|
|
|
|
def _int(name: str) -> int:
|
|
try:
|
|
return int(headers.get(name) or 0)
|
|
except (TypeError, ValueError):
|
|
return 0
|
|
|
|
def _float(name: str) -> float:
|
|
try:
|
|
return float(headers.get(name) or 0.0)
|
|
except (TypeError, ValueError):
|
|
return 0.0
|
|
|
|
return {
|
|
"calls": 1,
|
|
"model": headers.get("X-Gateway-Model") or "",
|
|
"prompt_tokens": _int("X-Gateway-Prompt-Tokens"),
|
|
"completion_tokens": _int("X-Gateway-Completion-Tokens"),
|
|
"total_tokens": _int("X-Gateway-Total-Tokens"),
|
|
"cost_usd": _float("X-Gateway-Cost-USD"),
|
|
"native_cost": headers.get("X-Gateway-Cost-Native") == "1",
|
|
"upstream_latency_ms": _float("X-Gateway-Upstream-Latency-Ms"),
|
|
"total_latency_ms": _float("X-Gateway-Total-Latency-Ms"),
|
|
}
|
|
|
|
|
|
USAGE_FIELDS = (
|
|
"calls",
|
|
"prompt_tokens",
|
|
"completion_tokens",
|
|
"total_tokens",
|
|
"cost_usd",
|
|
"upstream_latency_ms",
|
|
"total_latency_ms",
|
|
)
|
|
|
|
|
|
def new_usage_totals() -> dict:
|
|
return {field: 0 for field in USAGE_FIELDS}
|
|
|
|
|
|
def accumulate_usage(totals: Optional[dict], response) -> None:
|
|
if totals is None:
|
|
return
|
|
parsed = parse_usage_headers(getattr(response, "headers", None))
|
|
if not parsed:
|
|
return
|
|
for field in USAGE_FIELDS:
|
|
totals[field] += parsed[field]
|
|
|
|
|
|
def usage_metric_cards(usage: dict) -> list[dict]:
|
|
return [
|
|
{"label": "AI calls", "value": usage["calls"]},
|
|
{"label": "Total tokens", "value": usage["total_tokens"]},
|
|
{"label": "Prompt tokens", "value": usage["prompt_tokens"]},
|
|
{"label": "Completion tokens", "value": usage["completion_tokens"]},
|
|
{"label": "Total cost", "value": f"${usage['cost_usd']:.4f}"},
|
|
{"label": "Avg tokens/call", "value": usage["avg_tokens"]},
|
|
{"label": "Avg cost/call", "value": f"${usage['avg_cost_usd']:.6f}"},
|
|
{"label": "Avg latency", "value": f"{usage['avg_upstream_latency_ms']:.0f}ms"},
|
|
{"label": "Avg tokens/sec", "value": usage["avg_tokens_per_second"]},
|
|
]
|
|
|
|
|
|
class GatewayUsageLedger:
|
|
def record(self, raw: dict, pricing: Pricing, context_map: dict) -> Optional[dict]:
|
|
try:
|
|
usage = raw.get("usage") or {}
|
|
norm = normalize_usage(usage)
|
|
cost_usd, input_cost, output_cost, native = compute_cost(
|
|
usage, norm, pricing, raw["backend"]
|
|
)
|
|
window, util = context_utilization(
|
|
norm["total"], raw.get("model") or "", context_map
|
|
)
|
|
upstream_ms = float(raw.get("upstream_latency_ms") or 0)
|
|
completion = norm["completion"]
|
|
tps = (
|
|
completion / (upstream_ms / 1000.0)
|
|
if upstream_ms > 0 and completion
|
|
else 0.0
|
|
)
|
|
row = {
|
|
"created_at": _iso(_now()),
|
|
"owner_kind": raw.get("owner_kind") or "unknown",
|
|
"owner_id": raw.get("owner_id") or "unknown",
|
|
"backend": raw["backend"],
|
|
"endpoint": raw.get("endpoint") or "",
|
|
"requested_model": raw.get("requested_model") or "",
|
|
"model": raw.get("model") or "",
|
|
"provider": (raw.get("provider") or "default")[:60],
|
|
"fallback_used_route": (
|
|
str(raw["fallback_used_route"])[:128]
|
|
if raw.get("fallback_used_route")
|
|
else None
|
|
),
|
|
"status_code": int(raw.get("status_code") or 0),
|
|
"success": 1 if raw.get("success") else 0,
|
|
"error_category": raw.get("error_category"),
|
|
"upstream_latency_ms": upstream_ms,
|
|
"gateway_overhead_ms": float(raw.get("gateway_overhead_ms") or 0),
|
|
"queue_wait_ms": float(raw.get("queue_wait_ms") or 0),
|
|
"connect_ms": float(raw.get("connect_ms") or 0),
|
|
"total_latency_ms": float(raw.get("total_latency_ms") or 0),
|
|
"prompt_tokens": norm["prompt"],
|
|
"completion_tokens": norm["completion"],
|
|
"cache_hit_tokens": norm["cache_hit"],
|
|
"cache_miss_tokens": norm["cache_miss"],
|
|
"reasoning_tokens": norm["reasoning"],
|
|
"total_tokens": norm["total"],
|
|
"tokens_per_second": round(tps, 3),
|
|
"context_window": window,
|
|
"context_utilization": util,
|
|
"cost_usd": round(cost_usd, 8),
|
|
"input_cost_usd": round(input_cost, 8),
|
|
"output_cost_usd": round(output_cost, 8),
|
|
"native_cost": 1 if native else 0,
|
|
"stream_requested": 1 if raw.get("stream_requested") else 0,
|
|
"temperature": raw.get("temperature"),
|
|
"top_p": raw.get("top_p"),
|
|
"max_tokens": raw.get("max_tokens"),
|
|
"has_tools": 1 if raw.get("has_tools") else 0,
|
|
"retries_attempted": int(raw.get("retries_attempted") or 0),
|
|
"retry_succeeded": 1 if raw.get("retry_succeeded") else 0,
|
|
"circuit_open": 1 if raw.get("circuit_open") else 0,
|
|
"user_agent": (raw.get("user_agent") or "")[:300],
|
|
"app_reference": (raw.get("app_reference") or "default")[:30],
|
|
"ttft_ms": raw.get("ttft_ms"),
|
|
"inter_token_ms": raw.get("inter_token_ms"),
|
|
}
|
|
get_table(GATEWAY_LEDGER).insert(row)
|
|
self._audit(raw, norm, cost_usd)
|
|
self._track_health(row)
|
|
return row
|
|
except Exception as exc:
|
|
logger.warning("gateway usage record failed: %s", exc)
|
|
return None
|
|
|
|
def _track_health(self, row: dict) -> None:
|
|
# Observational only - never gates or influences dispatch. Isolated in
|
|
# its own try/except so a health-tracking bug can never turn an
|
|
# already-committed ledger write into a failed record() call.
|
|
try:
|
|
from devplacepy.services.openai_gateway import model_health
|
|
|
|
model_health.record_outcome(
|
|
row["provider"],
|
|
row["model"],
|
|
bool(row["success"]),
|
|
latency_ms=row.get("upstream_latency_ms") or None,
|
|
tokens_per_second=row.get("tokens_per_second") or None,
|
|
)
|
|
except Exception as exc: # noqa: BLE001
|
|
logger.debug("model health tracking failed: %s", exc)
|
|
|
|
def record_external(
|
|
self,
|
|
*,
|
|
owner_kind: str,
|
|
owner_id: str,
|
|
backend: str,
|
|
endpoint: str,
|
|
model: str,
|
|
cost_usd: float,
|
|
success: bool,
|
|
status_code: int,
|
|
latency_ms: float = 0.0,
|
|
app_reference: str = "default",
|
|
) -> Optional[dict]:
|
|
try:
|
|
row = {
|
|
"created_at": _iso(_now()),
|
|
"owner_kind": owner_kind or "unknown",
|
|
"owner_id": owner_id or "unknown",
|
|
"backend": backend,
|
|
"endpoint": endpoint or "",
|
|
"requested_model": model or "",
|
|
"model": model or "",
|
|
"status_code": int(status_code or 0),
|
|
"success": 1 if success else 0,
|
|
"error_category": None,
|
|
"upstream_latency_ms": float(latency_ms or 0),
|
|
"gateway_overhead_ms": 0.0,
|
|
"queue_wait_ms": 0.0,
|
|
"connect_ms": 0.0,
|
|
"total_latency_ms": float(latency_ms or 0),
|
|
"prompt_tokens": 0,
|
|
"completion_tokens": 0,
|
|
"cache_hit_tokens": 0,
|
|
"cache_miss_tokens": 0,
|
|
"reasoning_tokens": 0,
|
|
"total_tokens": 0,
|
|
"tokens_per_second": 0.0,
|
|
"context_window": None,
|
|
"context_utilization": None,
|
|
"cost_usd": round(float(cost_usd or 0), 8),
|
|
"input_cost_usd": 0.0,
|
|
"output_cost_usd": 0.0,
|
|
"native_cost": 0,
|
|
"stream_requested": 0,
|
|
"temperature": None,
|
|
"top_p": None,
|
|
"max_tokens": None,
|
|
"has_tools": 0,
|
|
"retries_attempted": 0,
|
|
"retry_succeeded": 0,
|
|
"circuit_open": 0,
|
|
"user_agent": "",
|
|
"app_reference": app_reference or "default",
|
|
"ttft_ms": None,
|
|
"inter_token_ms": None,
|
|
}
|
|
get_table(GATEWAY_LEDGER).insert(row)
|
|
self._audit_external(row)
|
|
return row
|
|
except Exception as exc:
|
|
logger.warning("gateway external usage record failed: %s", exc)
|
|
return None
|
|
|
|
def _audit_external(self, row: dict) -> None:
|
|
from devplacepy.services.audit import record as audit
|
|
|
|
owner_kind = row.get("owner_kind") or "unknown"
|
|
owner_id = row.get("owner_id") or "unknown"
|
|
actor_kind, actor_uid, actor_role = audit_actor_for(owner_kind, owner_id)
|
|
audit.record_system(
|
|
"ai.gateway.call",
|
|
actor_kind=actor_kind,
|
|
actor_uid=actor_uid,
|
|
actor_role=actor_role,
|
|
origin="api",
|
|
result="success" if row.get("success") else "failure",
|
|
summary=f"external AI call by {owner_kind}/{owner_id} ({row.get('backend')})",
|
|
metadata={
|
|
"backend": row.get("backend"),
|
|
"endpoint": row.get("endpoint"),
|
|
"cost_usd": row.get("cost_usd"),
|
|
"status_code": row.get("status_code"),
|
|
"owner_kind": owner_kind,
|
|
"owner_id": owner_id,
|
|
},
|
|
)
|
|
|
|
def _audit(self, raw: dict, norm: dict, cost_usd: float) -> None:
|
|
from devplacepy.services.audit import record as audit
|
|
|
|
owner_kind = raw.get("owner_kind") or "unknown"
|
|
owner_id = raw.get("owner_id") or "unknown"
|
|
actor_kind, actor_uid, actor_role = audit_actor_for(owner_kind, owner_id)
|
|
audit.record_system(
|
|
"ai.gateway.call",
|
|
actor_kind=actor_kind,
|
|
actor_uid=actor_uid,
|
|
actor_role=actor_role,
|
|
origin="api",
|
|
result="success" if raw.get("success") else "failure",
|
|
summary=f"LLM call by {owner_kind}/{owner_id} (model {raw.get('model') or ''})",
|
|
metadata={
|
|
"model": raw.get("model"),
|
|
"backend": raw.get("backend"),
|
|
"prompt_tokens": norm["prompt"],
|
|
"completion_tokens": norm["completion"],
|
|
"total_tokens": norm["total"],
|
|
"cost_usd": round(cost_usd, 8),
|
|
"status_code": raw.get("status_code"),
|
|
"error_category": raw.get("error_category"),
|
|
"owner_kind": owner_kind,
|
|
"owner_id": owner_id,
|
|
},
|
|
)
|
|
|
|
def sample_concurrency(self, in_flight: int) -> None:
|
|
try:
|
|
get_table(GATEWAY_CONCURRENCY).insert(
|
|
{
|
|
"created_at": _iso(_now()),
|
|
"in_flight": int(in_flight),
|
|
}
|
|
)
|
|
except Exception as exc:
|
|
logger.warning("gateway concurrency sample failed: %s", exc)
|
|
|
|
def prune(self, older_than_hours: int) -> tuple[int, int]:
|
|
cutoff = _iso(_now() - timedelta(hours=max(1, older_than_hours)))
|
|
ledger_removed = 0
|
|
samples_removed = 0
|
|
if GATEWAY_LEDGER in db.tables:
|
|
ledger_removed = int(
|
|
get_table(GATEWAY_LEDGER).delete(created_at={"<": cutoff})
|
|
)
|
|
if GATEWAY_CONCURRENCY in db.tables:
|
|
samples_removed = int(
|
|
get_table(GATEWAY_CONCURRENCY).delete(created_at={"<": cutoff})
|
|
)
|
|
return ledger_removed, samples_removed
|
|
|
|
|
|
def record_rsearch_call(
|
|
owner_kind: str, owner_id: str, endpoint: str, success: bool, status_code: int
|
|
) -> None:
|
|
try:
|
|
from devplacepy.services.manager import service_manager
|
|
|
|
service = service_manager.get_service("openai")
|
|
cfg = service.get_config() if service is not None else {}
|
|
cost = float(cfg.get("gateway_rsearch_cost_per_call", 0.0) or 0.0)
|
|
GatewayUsageLedger().record_external(
|
|
owner_kind=owner_kind or "system",
|
|
owner_id=owner_id or "",
|
|
backend="rsearch",
|
|
endpoint=endpoint or "/search",
|
|
model="rsearch",
|
|
cost_usd=cost,
|
|
success=success,
|
|
status_code=status_code,
|
|
app_reference="devplace-devii-rsearch-v-1-0-0",
|
|
)
|
|
except Exception as exc:
|
|
logger.warning("rsearch usage ledger failed: %s", exc)
|