This commit is contained in:
2026-07-09 02:52:54 +02:00
parent 818568c609
commit 48bb6c2ec2
95 changed files with 6115 additions and 267 deletions
+130
View File
@@ -1,13 +1,28 @@
# retoor <retoor@molodetz.nl>
from dataclasses import replace
from datetime import datetime, timezone
from devplacepy.services.openai_gateway.usage import (
USAGE_FIELDS,
Pricing,
accumulate_usage,
compute_cost,
new_usage_totals,
normalize_usage,
parse_usage_headers,
usage_metric_cards,
)
BASE_PRICING = Pricing(
chat_cache_hit_per_m=0.0028,
chat_cache_miss_per_m=0.14,
chat_output_per_m=0.28,
vision_input_per_m=0.0,
vision_output_per_m=0.0,
embed_input_per_m=0.01,
)
GATEWAY_HEADERS = {
"X-Gateway-Cost-USD": "0.00010000",
"X-Gateway-Model": "molodetz",
@@ -89,3 +104,118 @@ def test_usage_metric_cards_labels_and_formatting():
assert by_label["Total cost"] == "$0.0492"
assert by_label["Avg cost/call"] == "$0.012300"
assert by_label["Avg latency"] == "4200ms"
def test_compute_cost_matches_deepseek_flat_rates_when_no_tier_configured():
norm = normalize_usage(
{
"prompt_tokens": 2000,
"completion_tokens": 100,
"prompt_cache_hit_tokens": 500,
"prompt_cache_miss_tokens": 1500,
}
)
total, input_cost, output_cost, native = compute_cost({}, norm, BASE_PRICING, "chat")
assert native is False
assert abs(input_cost - (500 / 1e6 * 0.0028 + 1500 / 1e6 * 0.14)) < 1e-12
assert abs(output_cost - (100 / 1e6 * 0.28)) < 1e-12
assert abs(total - (input_cost + output_cost)) < 1e-12
def test_compute_cost_switches_to_tier2_above_threshold():
tiered = replace(
BASE_PRICING,
chat_cache_hit_per_m_tier2=0.005,
chat_cache_miss_per_m_tier2=0.25,
chat_output_per_m_tier2=0.5,
context_tier_threshold_tokens=1000,
)
below = normalize_usage({"prompt_tokens": 900, "completion_tokens": 50})
above = normalize_usage({"prompt_tokens": 1001, "completion_tokens": 50})
_, below_input, below_output, _ = compute_cost({}, below, tiered, "chat")
_, above_input, above_output, _ = compute_cost({}, above, tiered, "chat")
assert abs(below_output - (50 / 1e6 * 0.28)) < 1e-12
assert abs(above_output - (50 / 1e6 * 0.5)) < 1e-12
assert below_input != above_input
def test_compute_cost_tier2_leaves_unset_component_at_tier1():
tiered = replace(
BASE_PRICING,
chat_output_per_m_tier2=0.5,
context_tier_threshold_tokens=100,
)
norm = normalize_usage({"prompt_tokens": 200, "completion_tokens": 10})
_, input_cost, output_cost, _ = compute_cost({}, norm, tiered, "chat")
assert abs(output_cost - (10 / 1e6 * 0.5)) < 1e-12
assert abs(input_cost - (200 / 1e6 * 0.14)) < 1e-12
def test_compute_cost_applies_off_peak_discount():
discounted = replace(
BASE_PRICING,
off_peak_start_minute=60,
off_peak_end_minute=120,
off_peak_discount_pct=50.0,
)
norm = normalize_usage({"prompt_tokens": 1000, "completion_tokens": 100})
in_window = datetime(2026, 1, 1, 1, 30, tzinfo=timezone.utc)
outside_window = datetime(2026, 1, 1, 10, 0, tzinfo=timezone.utc)
total_in, _, _, _ = compute_cost({}, norm, discounted, "chat", now=in_window)
total_out, _, _, _ = compute_cost({}, norm, discounted, "chat", now=outside_window)
total_flat, _, _, _ = compute_cost({}, norm, BASE_PRICING, "chat")
assert abs(total_out - total_flat) < 1e-12
assert abs(total_in - total_flat / 2) < 1e-12
def test_compute_cost_off_peak_wraps_past_midnight():
wrapped = replace(
BASE_PRICING,
off_peak_start_minute=23 * 60,
off_peak_end_minute=60,
off_peak_discount_pct=100.0,
)
norm = normalize_usage({"prompt_tokens": 1000, "completion_tokens": 100})
just_after_midnight = datetime(2026, 1, 1, 0, 30, tzinfo=timezone.utc)
midday = datetime(2026, 1, 1, 12, 0, tzinfo=timezone.utc)
total_wrapped, _, _, _ = compute_cost(
{}, norm, wrapped, "chat", now=just_after_midnight
)
total_midday, _, _, _ = compute_cost({}, norm, wrapped, "chat", now=midday)
assert total_wrapped == 0.0
assert total_midday > 0.0
def test_compute_cost_image_branch_flat_per_call():
image_pricing = replace(BASE_PRICING, image_per_call=0.04)
norm = normalize_usage({})
total, input_cost, output_cost, native = compute_cost(
{}, norm, image_pricing, "image"
)
assert native is False
assert output_cost == 0.0
assert abs(total - 0.04) < 1e-9
assert abs(input_cost - 0.04) < 1e-9
def test_compute_cost_image_native_cost_preferred():
from devplacepy.services.openai_gateway.usage import extract_image_usage
image_pricing = replace(BASE_PRICING, image_per_call=0.04)
norm = normalize_usage({})
usage = extract_image_usage({"usage": {"cost": 0.055}})
total, _, _, native = compute_cost(usage, norm, image_pricing, "image")
assert native is True
assert total == 0.055
def test_compute_cost_native_upstream_cost_unaffected_by_tiering():
tiered = replace(
BASE_PRICING,
chat_output_per_m_tier2=100.0,
context_tier_threshold_tokens=1,
)
norm = normalize_usage({"prompt_tokens": 1000, "completion_tokens": 100})
total, _, _, native = compute_cost({"cost": 0.05}, norm, tiered, "chat")
assert native is True
assert total == 0.05