forked from retoor/devplacepy
Trust only the upstream X-Gateway-Model header in the AI gateway
An upstream the gateway forwards to may itself emit X-Gateway-* headers (e.g. another DevPlace-style gateway), which can collide with the ones about to be built for the response. Only X-Gateway-Model is ever trusted from upstream and relayed as-is - it is the one field an upstream can legitimately know better than we do (it may have resolved an alias or served a different pinned version). Every other header (cost, tokens, latency, context, app-reference) is always our own measurement and is never overwritten, since blending in an upstream's own accounting would corrupt the usage ledger's per-model rollups and the quota math built on top of it. usage.upstream_reported_model() extracts that one header defensively (case-insensitive lookup, rejects anything oversized or containing a control character) and gateway._apply_served_model() applies it, display- only, at the tail of every response-header build across chat, streaming, embeddings, images, and passthrough. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01BWJy6PrMMt5hwWxQwia2rd
This commit is contained in:
@@ -8,11 +8,18 @@ from devplacepy.utils import generate_uid
|
||||
import devplacepy.services.openai_gateway.gateway as gwmod
|
||||
from devplacepy.services.openai_gateway import GatewayService
|
||||
class FakeResp_openai_gateway:
|
||||
def __init__(self, status=200, payload=None, ctype="application/json", content=b""):
|
||||
def __init__(
|
||||
self,
|
||||
status=200,
|
||||
payload=None,
|
||||
ctype="application/json",
|
||||
content=b"",
|
||||
extra_headers=None,
|
||||
):
|
||||
self.status_code = status
|
||||
self._payload = payload
|
||||
self.text = json.dumps(payload) if payload is not None else ""
|
||||
self.headers = {"content-type": ctype}
|
||||
self.headers = {"content-type": ctype, **(extra_headers or {})}
|
||||
self.content = content
|
||||
|
||||
def json(self):
|
||||
@@ -1309,3 +1316,258 @@ def test_models_endpoint_publishes_molodetz(local_db):
|
||||
finally:
|
||||
routing.model_store.remove("molodetz")
|
||||
routing.model_store.remove("molodetz-pro")
|
||||
|
||||
|
||||
# Upstream header collision policy: only X-Gateway-Model is ever trusted and
|
||||
# forwarded from an upstream that happens to speak our own X-Gateway-* header
|
||||
# convention (e.g. another DevPlace-style gateway). Every other figure
|
||||
# (cost/tokens/latency/context/app-reference) must remain ours regardless of
|
||||
# what an upstream claims. See usage.upstream_reported_model / CLAUDE.md.
|
||||
|
||||
|
||||
class FakeClientUpstreamModelHeader_openai_gateway(FakeClient_openai_gateway):
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"id": "x",
|
||||
"model": body.get("model"),
|
||||
"choices": [{"message": {"content": "hi there"}}],
|
||||
},
|
||||
extra_headers={"X-Gateway-Model": "upstream/served-model-x"},
|
||||
)
|
||||
|
||||
|
||||
class FakeClientHostileModelHeader_openai_gateway(FakeClient_openai_gateway):
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"id": "x",
|
||||
"model": body.get("model"),
|
||||
"choices": [{"message": {"content": "hi there"}}],
|
||||
},
|
||||
extra_headers={"X-Gateway-Model": "evil\r\nX-Injected: true"},
|
||||
)
|
||||
|
||||
|
||||
class FakeEmbedClientUpstreamModelHeader_openai_gateway(FakeEmbedClient_openai_gateway):
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"object": "list",
|
||||
"model": body.get("model"),
|
||||
"data": [{"object": "embedding", "index": 0, "embedding": [0.1, 0.2]}],
|
||||
"usage": {"prompt_tokens": 5, "total_tokens": 5},
|
||||
},
|
||||
extra_headers={"X-Gateway-Model": "upstream/embed-served"},
|
||||
)
|
||||
|
||||
|
||||
class FakeImageClientUpstreamModelHeader_openai_gateway(FakeImageClient_openai_gateway):
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
body = request.json_body or {}
|
||||
return FakeResp_openai_gateway(
|
||||
payload={
|
||||
"created": 1,
|
||||
"model": body.get("model"),
|
||||
"data": [{"b64_json": "aGVsbG8="}],
|
||||
"usage": {"cost": 0.05},
|
||||
},
|
||||
extra_headers={"X-Gateway-Model": "upstream/image-served"},
|
||||
)
|
||||
|
||||
|
||||
class FakePassthroughClientUpstreamModelHeader_openai_gateway:
|
||||
def __init__(self, *a, **k):
|
||||
self.calls = []
|
||||
|
||||
def build_request(self, method, url, headers=None, json=None, content=None):
|
||||
return FakeRequest(method, url, json)
|
||||
|
||||
async def send(self, request, stream=False):
|
||||
self.calls.append((request.url, request.json_body))
|
||||
return FakeResp_openai_gateway(
|
||||
payload={"ok": True},
|
||||
extra_headers={"X-Gateway-Model": "upstream/passthrough-served"},
|
||||
)
|
||||
|
||||
async def aclose(self):
|
||||
pass
|
||||
|
||||
|
||||
def test_upstream_x_gateway_model_header_is_forwarded_for_chat(local_db, monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeClientUpstreamModelHeader_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_force_model"] = True
|
||||
cfg["gateway_model"] = "deepseek-chat"
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_chat(
|
||||
{"messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "upstream_model_chat"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "upstream/served-model-x"
|
||||
row = get_table("gateway_usage_ledger").find_one(owner_id="upstream_model_chat")
|
||||
assert row is not None
|
||||
assert row["model"] == "deepseek-chat"
|
||||
|
||||
|
||||
def test_upstream_x_gateway_model_header_is_forwarded_for_streaming_chat(
|
||||
local_db, monkeypatch
|
||||
):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeClientUpstreamModelHeader_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_chat(
|
||||
{"messages": [{"role": "user", "content": "hi"}], "stream": True},
|
||||
cfg,
|
||||
("guest", "upstream_model_stream"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "upstream/served-model-x"
|
||||
assert resp.headers["X-Gateway-Backend"] == "chat"
|
||||
|
||||
async def drain():
|
||||
async for _ in resp.body_iterator:
|
||||
pass
|
||||
|
||||
run_async(drain())
|
||||
|
||||
|
||||
def test_upstream_x_gateway_model_header_is_forwarded_for_embeddings(
|
||||
local_db, monkeypatch
|
||||
):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeEmbedClientUpstreamModelHeader_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_embed_enabled"] = True
|
||||
cfg["gateway_embed_model"] = "qwen/qwen3-embedding-8b"
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_embeddings(
|
||||
{"input": "hello"},
|
||||
cfg,
|
||||
("guest", "upstream_model_embed"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "upstream/embed-served"
|
||||
row = get_table("gateway_usage_ledger").find_one(owner_id="upstream_model_embed")
|
||||
assert row is not None
|
||||
assert row["model"] == "qwen/qwen3-embedding-8b"
|
||||
|
||||
|
||||
def test_upstream_x_gateway_model_header_is_forwarded_for_images(local_db, monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeImageClientUpstreamModelHeader_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_image_enabled"] = True
|
||||
cfg["gateway_image_model"] = "black-forest-labs/flux.2-pro"
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_images(
|
||||
{"prompt": "badge"},
|
||||
cfg,
|
||||
("guest", "upstream_model_img"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "upstream/image-served"
|
||||
row = get_table("gateway_usage_ledger").find_one(owner_id="upstream_model_img")
|
||||
assert row is not None
|
||||
assert row["model"] == "black-forest-labs/flux.2-pro"
|
||||
|
||||
|
||||
def test_upstream_x_gateway_model_header_is_forwarded_for_passthrough(
|
||||
local_db, monkeypatch
|
||||
):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakePassthroughClientUpstreamModelHeader_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_model"] = "deepseek-chat"
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_passthrough(
|
||||
"POST",
|
||||
"responses",
|
||||
"application/json",
|
||||
b"{}",
|
||||
cfg,
|
||||
("guest", "upstream_model_passthrough"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "upstream/passthrough-served"
|
||||
row = get_table("gateway_usage_ledger").find_one(
|
||||
owner_id="upstream_model_passthrough"
|
||||
)
|
||||
assert row is not None
|
||||
assert row["model"] == "deepseek-chat"
|
||||
|
||||
|
||||
def test_no_upstream_model_header_keeps_our_own_model(local_db, monkeypatch):
|
||||
monkeypatch.setattr(gwmod.httpx, "AsyncClient", FakeClient_openai_gateway)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_force_model"] = True
|
||||
cfg["gateway_model"] = "deepseek-chat"
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_chat(
|
||||
{"messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "no_upstream_header"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "deepseek-chat"
|
||||
|
||||
|
||||
def test_hostile_upstream_model_header_is_ignored_end_to_end(local_db, monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
gwmod.httpx, "AsyncClient", FakeClientHostileModelHeader_openai_gateway
|
||||
)
|
||||
svc = GatewayService()
|
||||
cfg = svc.effective_config()
|
||||
cfg["gateway_force_model"] = True
|
||||
cfg["gateway_model"] = "deepseek-chat"
|
||||
rt = svc.runtime()
|
||||
resp = run_async(
|
||||
rt.handle_chat(
|
||||
{"messages": [{"role": "user", "content": "hi"}]},
|
||||
cfg,
|
||||
("guest", "hostile_header"),
|
||||
"test",
|
||||
"default",
|
||||
)
|
||||
)
|
||||
assert resp.headers["X-Gateway-Model"] == "deepseek-chat"
|
||||
|
||||
Reference in New Issue
Block a user