# retoor from __future__ import annotations import logging import time import httpx from devplacepy import stealth from devplacepy.config import INTERNAL_GATEWAY_URL, INTERNAL_MODEL logger = logging.getLogger(__name__) CHAT_TIMEOUT_SECONDS = 120.0 DEFAULT_MAX_TOKENS = 1200 _gateway_client: httpx.AsyncClient | None = None def gateway_client() -> httpx.AsyncClient: global _gateway_client if _gateway_client is None or _gateway_client.is_closed: _gateway_client = stealth.stealth_async_client(timeout=CHAT_TIMEOUT_SECONDS) return _gateway_client async def request_completion( messages: list[dict], api_key: str, *, gateway_url: str = INTERNAL_GATEWAY_URL, model: str = INTERNAL_MODEL, max_tokens: int = DEFAULT_MAX_TOKENS, temperature: float = 0.2, timeout: float = CHAT_TIMEOUT_SECONDS, ) -> tuple[str, dict, int]: payload = { "model": model, "messages": messages, "max_tokens": max_tokens, "temperature": temperature, } headers = { "Authorization": f"Bearer {api_key}", "Content-Type": "application/json", "X-App-Reference": "devplace-deepsearch-v-1-0-0", } start: float = time.monotonic() response = await gateway_client().post( gateway_url, json=payload, headers=headers, timeout=timeout ) elapsed_ms: int = int((time.monotonic() - start) * 1000) if response.status_code >= 400: raise RuntimeError(f"chat gateway returned {response.status_code}") data: dict = response.json() return data, data.get("usage") or {}, elapsed_ms async def complete_chat( messages: list[dict], api_key: str, *, gateway_url: str = INTERNAL_GATEWAY_URL, model: str = INTERNAL_MODEL, max_tokens: int = DEFAULT_MAX_TOKENS, temperature: float = 0.2, ) -> str: data, _usage, _elapsed_ms = await request_completion( messages, api_key, gateway_url=gateway_url, model=model, max_tokens=max_tokens, temperature=temperature, ) choices = data.get("choices") or [] if not choices: raise RuntimeError("chat gateway returned no choices") return (choices[0].get("message", {}).get("content") or "").strip()