# retoor from dataclasses import replace from datetime import datetime, timezone from devplacepy.services.openai_gateway.usage import ( USAGE_FIELDS, Pricing, accumulate_usage, compute_cost, new_usage_totals, normalize_usage, parse_usage_headers, usage_metric_cards, ) BASE_PRICING = Pricing( chat_cache_hit_per_m=0.0028, chat_cache_miss_per_m=0.14, chat_output_per_m=0.28, vision_input_per_m=0.0, vision_output_per_m=0.0, embed_input_per_m=0.01, ) GATEWAY_HEADERS = { "X-Gateway-Cost-USD": "0.00010000", "X-Gateway-Model": "molodetz", "X-Gateway-Prompt-Tokens": "500", "X-Gateway-Completion-Tokens": "100", "X-Gateway-Total-Tokens": "600", "X-Gateway-Upstream-Latency-Ms": "400", "X-Gateway-Total-Latency-Ms": "450", } class FakeResp: def __init__(self, headers=None): self.headers = headers if headers is not None else {} def test_new_usage_totals_is_zeroed(): totals = new_usage_totals() assert set(totals) == set(USAGE_FIELDS) assert all(value == 0 for value in totals.values()) def test_accumulate_usage_sums_headers(): totals = new_usage_totals() accumulate_usage(totals, FakeResp(GATEWAY_HEADERS)) accumulate_usage(totals, FakeResp(GATEWAY_HEADERS)) assert totals["calls"] == 2 assert totals["prompt_tokens"] == 1000 assert totals["completion_tokens"] == 200 assert totals["total_tokens"] == 1200 assert abs(totals["cost_usd"] - 0.0002) < 1e-9 assert totals["upstream_latency_ms"] == 800.0 assert totals["total_latency_ms"] == 900.0 def test_accumulate_usage_is_noop_without_headers_or_totals(): totals = new_usage_totals() accumulate_usage(None, FakeResp(GATEWAY_HEADERS)) accumulate_usage(totals, FakeResp({})) accumulate_usage(totals, object()) assert totals["calls"] == 0 def test_parse_usage_headers_requires_cost_header(): assert parse_usage_headers({}) is None assert parse_usage_headers(None) is None parsed = parse_usage_headers(GATEWAY_HEADERS) assert parsed["calls"] == 1 assert parsed["total_tokens"] == 600 def test_usage_metric_cards_labels_and_formatting(): usage = { "calls": 4, "total_tokens": 9200, "prompt_tokens": 6000, "completion_tokens": 3200, "cost_usd": 0.0492, "avg_tokens": 2300.0, "avg_cost_usd": 0.0123, "avg_upstream_latency_ms": 4200.0, "avg_tokens_per_second": 190.5, } cards = usage_metric_cards(usage) labels = [card["label"] for card in cards] assert labels == [ "AI calls", "Total tokens", "Prompt tokens", "Completion tokens", "Total cost", "Avg tokens/call", "Avg cost/call", "Avg latency", "Avg tokens/sec", ] by_label = {card["label"]: card["value"] for card in cards} assert by_label["AI calls"] == 4 assert by_label["Total cost"] == "$0.0492" assert by_label["Avg cost/call"] == "$0.012300" assert by_label["Avg latency"] == "4200ms" def test_compute_cost_matches_deepseek_flat_rates_when_no_tier_configured(): norm = normalize_usage( { "prompt_tokens": 2000, "completion_tokens": 100, "prompt_cache_hit_tokens": 500, "prompt_cache_miss_tokens": 1500, } ) total, input_cost, output_cost, native = compute_cost({}, norm, BASE_PRICING, "chat") assert native is False assert abs(input_cost - (500 / 1e6 * 0.0028 + 1500 / 1e6 * 0.14)) < 1e-12 assert abs(output_cost - (100 / 1e6 * 0.28)) < 1e-12 assert abs(total - (input_cost + output_cost)) < 1e-12 def test_compute_cost_switches_to_tier2_above_threshold(): tiered = replace( BASE_PRICING, chat_cache_hit_per_m_tier2=0.005, chat_cache_miss_per_m_tier2=0.25, chat_output_per_m_tier2=0.5, context_tier_threshold_tokens=1000, ) below = normalize_usage({"prompt_tokens": 900, "completion_tokens": 50}) above = normalize_usage({"prompt_tokens": 1001, "completion_tokens": 50}) _, below_input, below_output, _ = compute_cost({}, below, tiered, "chat") _, above_input, above_output, _ = compute_cost({}, above, tiered, "chat") assert abs(below_output - (50 / 1e6 * 0.28)) < 1e-12 assert abs(above_output - (50 / 1e6 * 0.5)) < 1e-12 assert below_input != above_input def test_compute_cost_tier2_leaves_unset_component_at_tier1(): tiered = replace( BASE_PRICING, chat_output_per_m_tier2=0.5, context_tier_threshold_tokens=100, ) norm = normalize_usage({"prompt_tokens": 200, "completion_tokens": 10}) _, input_cost, output_cost, _ = compute_cost({}, norm, tiered, "chat") assert abs(output_cost - (10 / 1e6 * 0.5)) < 1e-12 assert abs(input_cost - (200 / 1e6 * 0.14)) < 1e-12 def test_compute_cost_applies_off_peak_discount(): discounted = replace( BASE_PRICING, off_peak_start_minute=60, off_peak_end_minute=120, off_peak_discount_pct=50.0, ) norm = normalize_usage({"prompt_tokens": 1000, "completion_tokens": 100}) in_window = datetime(2026, 1, 1, 1, 30, tzinfo=timezone.utc) outside_window = datetime(2026, 1, 1, 10, 0, tzinfo=timezone.utc) total_in, _, _, _ = compute_cost({}, norm, discounted, "chat", now=in_window) total_out, _, _, _ = compute_cost({}, norm, discounted, "chat", now=outside_window) total_flat, _, _, _ = compute_cost({}, norm, BASE_PRICING, "chat") assert abs(total_out - total_flat) < 1e-12 assert abs(total_in - total_flat / 2) < 1e-12 def test_compute_cost_off_peak_wraps_past_midnight(): wrapped = replace( BASE_PRICING, off_peak_start_minute=23 * 60, off_peak_end_minute=60, off_peak_discount_pct=100.0, ) norm = normalize_usage({"prompt_tokens": 1000, "completion_tokens": 100}) just_after_midnight = datetime(2026, 1, 1, 0, 30, tzinfo=timezone.utc) midday = datetime(2026, 1, 1, 12, 0, tzinfo=timezone.utc) total_wrapped, _, _, _ = compute_cost( {}, norm, wrapped, "chat", now=just_after_midnight ) total_midday, _, _, _ = compute_cost({}, norm, wrapped, "chat", now=midday) assert total_wrapped == 0.0 assert total_midday > 0.0 def test_compute_cost_image_branch_flat_per_call(): image_pricing = replace(BASE_PRICING, image_per_call=0.04) norm = normalize_usage({}) total, input_cost, output_cost, native = compute_cost( {}, norm, image_pricing, "image" ) assert native is False assert output_cost == 0.0 assert abs(total - 0.04) < 1e-9 assert abs(input_cost - 0.04) < 1e-9 def test_compute_cost_image_native_cost_preferred(): from devplacepy.services.openai_gateway.usage import extract_image_usage image_pricing = replace(BASE_PRICING, image_per_call=0.04) norm = normalize_usage({}) usage = extract_image_usage({"usage": {"cost": 0.055}}) total, _, _, native = compute_cost(usage, norm, image_pricing, "image") assert native is True assert total == 0.055 def test_compute_cost_native_upstream_cost_unaffected_by_tiering(): tiered = replace( BASE_PRICING, chat_output_per_m_tier2=100.0, context_tier_threshold_tokens=1, ) norm = normalize_usage({"prompt_tokens": 1000, "completion_tokens": 100}) total, _, _, native = compute_cost({"cost": 0.05}, norm, tiered, "chat") assert native is True assert total == 0.05