Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/management_endpoints/cost_tracking_settings.py: 59%
229 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2COST TRACKING SETTINGS MANAGEMENT
4Endpoints for managing cost discount and margin configuration
6GET /config/cost_discount_config - Get current cost discount configuration
7PATCH /config/cost_discount_config - Update cost discount configuration
8GET /config/cost_margin_config - Get current cost margin configuration
9PATCH /config/cost_margin_config - Update cost margin configuration
10POST /cost/estimate - Estimate cost for a given model and token counts
11"""
13from collections.abc import Mapping
14from dataclasses import dataclass
15from typing import Final
17from fastapi import APIRouter, Depends, HTTPException
18from pydantic import BaseModel
20import litellm
21from litellm._internal_context import current_billing_time, pinned_billing_time
22from litellm._logging import verbose_proxy_logger
23from litellm.cost_calculator import completion_cost
24from litellm.proxy._types import (
25 CommonProxyErrors,
26 CostEstimateRequest,
27 CostEstimateResponse,
28 UserAPIKeyAuth,
29)
30from litellm.proxy.auth.user_api_key_auth import user_api_key_auth
31from litellm.proxy.management_endpoints.prompt_cache_prediction import router as prompt_cache_prediction_router
32from litellm.types.utils import (
33 CostBreakdown,
34 CostPerToken,
35 LlmProvidersSet,
36 ModelInfo,
37 ModelResponse,
38 PromptTokensDetailsWrapper,
39 Usage,
40)
42router: Final = APIRouter()
43router.include_router(prompt_cache_prediction_router)
46@dataclass(frozen=True, slots=True)
47class ResolvedCostModel:
48 model: str
49 provider: str | None
50 custom_cost_per_token: CostPerToken | None
53def _configured_price(key: str, sources: tuple[Mapping[str, object], ...]) -> float | None:
54 values: Final = (source.get(key) for source in sources)
55 numeric: Final = (float(value) for value in values if isinstance(value, (int, float)))
56 return next(numeric, None)
59def _extract_custom_pricing(
60 litellm_params: Mapping[str, object], model_info: Mapping[str, object], builtin: ModelInfo | None
61) -> CostPerToken | None:
62 """
63 Pull per-token pricing configured on a deployment so on-prem / self-hosted
64 models (absent from the public cost map) still estimate a real cost.
65 Pricing may live on ``litellm_params`` or ``model_info``; ``litellm_params``
66 wins, matching the router's cost-map registration precedence. Cache rates the
67 deployment leaves unset come from the backend model's built-in entry, then its
68 own input rate, again matching what the router registers for live billing.
69 """
70 sources: Final = (litellm_params, model_info)
71 input_price: Final = _configured_price("input_cost_per_token", sources)
72 output_price: Final = _configured_price("output_cost_per_token", sources)
74 if input_price is None and output_price is None:
75 return None
77 input_rate: Final = input_price or 0.0
78 cache_sources: Final = sources if builtin is None else (*sources, builtin)
79 cache_read_price: Final = _configured_price("cache_read_input_token_cost", cache_sources)
80 cache_creation_price: Final = _configured_price("cache_creation_input_token_cost", cache_sources)
81 return CostPerToken(
82 input_cost_per_token=input_rate,
83 output_cost_per_token=output_price or 0.0,
84 cache_read_input_token_cost=input_rate if cache_read_price is None else cache_read_price,
85 cache_creation_input_token_cost=input_rate if cache_creation_price is None else cache_creation_price,
86 )
89def _lookup_model_info(model: str, custom_llm_provider: str | None = None) -> ModelInfo | None:
90 try:
91 return litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider)
92 except Exception:
93 return None
96def _resolve_model_for_cost_lookup(model: str) -> ResolvedCostModel:
97 """
98 Resolve a model name (which may be a router alias/model_group) to the
99 underlying litellm model name, provider, and any deployment-configured
100 pricing used for cost lookup.
102 Args:
103 model: The model name from the request (could be a router alias like 'e-model-router'
104 or an actual model name like 'azure_ai/gpt-4')
105 """
106 from litellm.proxy.proxy_server import llm_router
108 # Try to resolve from router if available
109 if llm_router is not None: 109 ↛ 133line 109 didn't jump to line 133 because the condition on line 109 was always true
110 try:
111 # Get deployments for this model name (handles aliases, wildcards, etc.)
112 deployments: Final = llm_router.get_model_list(model_name=model)
114 if deployments and len(deployments) > 0: 114 ↛ 115line 114 didn't jump to line 115 because the condition on line 114 was never true
115 first_deployment: Final = deployments[0]
116 litellm_params: Final = first_deployment.get("litellm_params", {})
117 model_info: Final = first_deployment.get("model_info", {})
118 custom_llm_provider: Final = litellm_params.get("custom_llm_provider")
119 provider: Final = str(custom_llm_provider) if custom_llm_provider is not None else None
120 # base_model wins (needed for Azure custom deployment names)
121 base_model: Final = model_info.get("base_model") or litellm_params.get("base_model")
122 resolved_model: Final = base_model or litellm_params.get("model")
123 if resolved_model:
124 verbose_proxy_logger.debug("Resolved model '%s' to '%s' from router", model, resolved_model)
125 custom_cost_per_token: Final = _extract_custom_pricing(
126 litellm_params, model_info, _lookup_model_info(str(resolved_model), provider)
127 )
128 return ResolvedCostModel(str(resolved_model), provider, custom_cost_per_token)
129 except Exception as e:
130 verbose_proxy_logger.debug("Could not resolve model '%s' from router: %s", model, e)
132 # Return original model if not resolved
133 return ResolvedCostModel(model, None, None)
136@dataclass(frozen=True, slots=True)
137class CostLines:
138 """Cost of one request split the way the spend logs split it: the cache lines are
139 shares of input_cost and the reasoning line is a share of output_cost."""
141 total_cost: float
142 input_cost: float
143 output_cost: float
144 margin_cost: float
145 cache_read_cost: float
146 cache_creation_cost: float
147 reasoning_cost: float
149 def times(self, num_requests: int | None) -> "CostLines | None":
150 if not num_requests:
151 return None
152 return CostLines(
153 total_cost=self.total_cost * num_requests,
154 input_cost=self.input_cost * num_requests,
155 output_cost=self.output_cost * num_requests,
156 margin_cost=self.margin_cost * num_requests,
157 cache_read_cost=self.cache_read_cost * num_requests,
158 cache_creation_cost=self.cache_creation_cost * num_requests,
159 reasoning_cost=self.reasoning_cost * num_requests,
160 )
163def _cost_lines(cost_per_request: float, cost_breakdown: CostBreakdown | None) -> CostLines:
164 breakdown: Final = cost_breakdown if cost_breakdown is not None else CostBreakdown()
165 return CostLines(
166 total_cost=cost_per_request,
167 input_cost=breakdown.get("input_cost", 0.0),
168 output_cost=breakdown.get("output_cost", 0.0),
169 margin_cost=breakdown.get("margin_total_amount", 0.0),
170 cache_read_cost=breakdown.get("cache_read_cost", 0.0),
171 cache_creation_cost=breakdown.get("cache_creation_cost", 0.0),
172 reasoning_cost=breakdown.get("reasoning_cost", 0.0),
173 )
176def _usage_for_estimate(request: CostEstimateRequest) -> Usage:
177 cache_tokens: Final = request.cache_read_input_tokens + request.cache_creation_input_tokens
178 return Usage(
179 prompt_tokens=request.input_tokens,
180 completion_tokens=request.output_tokens,
181 total_tokens=request.input_tokens + request.output_tokens,
182 reasoning_tokens=request.reasoning_tokens,
183 prompt_tokens_details=PromptTokensDetailsWrapper(
184 cached_tokens=request.cache_read_input_tokens,
185 cache_creation_tokens=request.cache_creation_input_tokens,
186 )
187 if cache_tokens
188 else None,
189 )
192@router.get(
193 "/config/cost_discount_config",
194 tags=["Cost Tracking"],
195 dependencies=[Depends(user_api_key_auth)],
196)
197async def get_cost_discount_config(
198 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
199):
200 """
201 Get current cost discount configuration.
203 Returns the cost_discount_config from litellm_settings.
204 """
205 from litellm.proxy.proxy_server import prisma_client, proxy_config
207 if prisma_client is None: 207 ↛ 208line 207 didn't jump to line 208 because the condition on line 207 was never true
208 raise HTTPException(
209 status_code=500,
210 detail={"error": CommonProxyErrors.db_not_connected_error.value},
211 )
213 try:
214 # Load config from DB
215 config: Final = await proxy_config.get_config()
217 # Get cost_discount_config from litellm_settings
218 litellm_settings: Final = config.get("litellm_settings", {})
219 cost_discount_config: Final = litellm_settings.get("cost_discount_config", {})
221 return {"values": cost_discount_config}
222 except Exception as e:
223 verbose_proxy_logger.error("Error fetching cost discount config: %s", e)
224 return {"values": {}}
227@router.patch(
228 "/config/cost_discount_config",
229 tags=["Cost Tracking"],
230 dependencies=[Depends(user_api_key_auth)],
231)
232async def update_cost_discount_config(
233 cost_discount_config: dict[str, float],
234 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
235):
236 """
237 Update cost discount configuration.
239 Updates the cost_discount_config in litellm_settings.
240 Discounts should be between 0 and 1 (e.g., 0.05 = 5% discount).
242 Example:
243 ```json
244 {
245 "vertex_ai": 0.05,
246 "gemini": 0.05,
247 "openai": 0.01
248 }
249 ```
250 """
251 from litellm.proxy.proxy_server import (
252 prisma_client,
253 proxy_config,
254 store_model_in_db,
255 )
257 if prisma_client is None: 257 ↛ 258line 257 didn't jump to line 258 because the condition on line 257 was never true
258 raise HTTPException(
259 status_code=500,
260 detail={"error": CommonProxyErrors.db_not_connected_error.value},
261 )
263 if store_model_in_db is not True: 263 ↛ 264line 263 didn't jump to line 264 because the condition on line 263 was never true
264 raise HTTPException(
265 status_code=500,
266 detail={"error": "Set `'STORE_MODEL_IN_DB='True'` in your env to enable this feature."},
267 )
269 # Validate that all providers are valid LiteLLM providers
270 invalid_providers: Final = []
271 for provider in cost_discount_config:
272 if provider not in LlmProvidersSet: 272 ↛ 271line 272 didn't jump to line 271 because the condition on line 272 was always true
273 invalid_providers.append(provider)
275 if invalid_providers:
276 raise HTTPException(
277 status_code=400,
278 detail={
279 "error": f"Invalid provider(s): {', '.join(invalid_providers)}. Must be valid LiteLLM providers. See https://docs.litellm.ai/docs/providers for the full list."
280 },
281 )
283 # Validate discount values are between 0 and 1
284 for provider, discount in cost_discount_config.items(): 284 ↛ 285line 284 didn't jump to line 285 because the loop on line 284 never started
285 if not isinstance(discount, (int, float)):
286 raise HTTPException(status_code=400, detail=f"Discount for {provider} must be a number")
287 if not (0 <= discount <= 1):
288 raise HTTPException(
289 status_code=400,
290 detail=f"Discount for {provider} must be between 0 and 1 (0% to 100%)",
291 )
293 try:
294 # Load existing config
295 config: Final = await proxy_config.get_config()
297 # Ensure litellm_settings exists
298 if "litellm_settings" not in config: 298 ↛ 299line 298 didn't jump to line 299 because the condition on line 298 was never true
299 config["litellm_settings"] = {}
301 # Update cost_discount_config
302 config["litellm_settings"]["cost_discount_config"] = cost_discount_config
304 # Save the updated config to DB
305 await proxy_config.save_config(new_config=config)
307 # Update in-memory litellm.cost_discount_config
308 litellm.cost_discount_config = cost_discount_config
310 verbose_proxy_logger.info("Updated cost_discount_config: %s", cost_discount_config)
312 return {
313 "message": "Cost discount configuration updated successfully",
314 "status": "success",
315 "values": cost_discount_config,
316 }
317 except Exception as e:
318 verbose_proxy_logger.error("Error updating cost discount config: %s", e)
319 raise HTTPException(
320 status_code=500,
321 detail={"error": f"Failed to update cost discount config: {e}"},
322 )
325@router.get(
326 "/config/cost_margin_config",
327 tags=["Cost Tracking"],
328 dependencies=[Depends(user_api_key_auth)],
329)
330async def get_cost_margin_config(
331 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
332):
333 """
334 Get current cost margin configuration.
336 Returns the cost_margin_config from litellm_settings.
337 """
338 from litellm.proxy.proxy_server import prisma_client, proxy_config
340 if prisma_client is None: 340 ↛ 341line 340 didn't jump to line 341 because the condition on line 340 was never true
341 raise HTTPException(
342 status_code=500,
343 detail={"error": CommonProxyErrors.db_not_connected_error.value},
344 )
346 try:
347 # Load config from DB
348 config: Final = await proxy_config.get_config()
350 # Get cost_margin_config from litellm_settings
351 litellm_settings: Final = config.get("litellm_settings", {})
352 cost_margin_config: Final = litellm_settings.get("cost_margin_config", {})
354 return {"values": cost_margin_config}
355 except Exception as e:
356 verbose_proxy_logger.error("Error fetching cost margin config: %s", e)
357 return {"values": {}}
360@router.patch(
361 "/config/cost_margin_config",
362 tags=["Cost Tracking"],
363 dependencies=[Depends(user_api_key_auth)],
364)
365async def update_cost_margin_config(
366 cost_margin_config: dict[str, float | dict[str, float]],
367 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
368):
369 """
370 Update cost margin configuration.
372 Updates the cost_margin_config in litellm_settings.
373 Margins can be:
374 - Percentage: {"openai": 0.10} = 10% margin
375 - Fixed amount: {"openai": {"fixed_amount": 0.001}} = $0.001 per request
376 - Combined: {"vertex_ai": {"percentage": 0.08, "fixed_amount": 0.0005}}
377 - Global: {"global": 0.05} = 5% global margin on all providers
379 Example:
380 ```json
381 {
382 "global": 0.05,
383 "openai": 0.10,
384 "anthropic": {"fixed_amount": 0.001},
385 "vertex_ai": {"percentage": 0.08, "fixed_amount": 0.0005}
386 }
387 ```
388 """
389 from litellm.proxy.proxy_server import (
390 prisma_client,
391 proxy_config,
392 store_model_in_db,
393 )
395 if prisma_client is None: 395 ↛ 396line 395 didn't jump to line 396 because the condition on line 395 was never true
396 raise HTTPException(
397 status_code=500,
398 detail={"error": CommonProxyErrors.db_not_connected_error.value},
399 )
401 if store_model_in_db is not True: 401 ↛ 402line 401 didn't jump to line 402 because the condition on line 401 was never true
402 raise HTTPException(
403 status_code=500,
404 detail={"error": "Set `'STORE_MODEL_IN_DB='True'` in your env to enable this feature."},
405 )
407 # Validate that all providers are valid LiteLLM providers (except "global")
408 invalid_providers: Final = []
409 for provider in cost_margin_config:
410 if provider != "global" and provider not in LlmProvidersSet: 410 ↛ 409line 410 didn't jump to line 409 because the condition on line 410 was always true
411 invalid_providers.append(provider)
413 if invalid_providers:
414 raise HTTPException(
415 status_code=400,
416 detail={
417 "error": f"Invalid provider(s): {', '.join(invalid_providers)}. Must be valid LiteLLM providers or 'global'. See https://docs.litellm.ai/docs/providers for the full list."
418 },
419 )
421 # Validate margin values
422 for provider, margin_value in cost_margin_config.items(): 422 ↛ 423line 422 didn't jump to line 423 because the loop on line 422 never started
423 if isinstance(margin_value, (int, float)):
424 # Simple percentage format: {"openai": 0.10}
425 if not (0 <= margin_value <= 10): # Allow up to 1000% margin
426 raise HTTPException(
427 status_code=400,
428 detail=f"Margin percentage for {provider} must be between 0 and 10 (0% to 1000%)",
429 )
430 elif isinstance(margin_value, dict):
431 # Complex format: {"percentage": 0.08, "fixed_amount": 0.0005}
432 if "percentage" in margin_value:
433 percentage = margin_value["percentage"]
434 if not isinstance(percentage, (int, float)):
435 raise HTTPException(
436 status_code=400,
437 detail=f"Margin percentage for {provider} must be a number",
438 )
439 if not (0 <= percentage <= 10):
440 raise HTTPException(
441 status_code=400,
442 detail=f"Margin percentage for {provider} must be between 0 and 10 (0% to 1000%)",
443 )
444 if "fixed_amount" in margin_value:
445 fixed_amount = margin_value["fixed_amount"]
446 if not isinstance(fixed_amount, (int, float)):
447 raise HTTPException(
448 status_code=400,
449 detail=f"Fixed margin amount for {provider} must be a number",
450 )
451 if fixed_amount < 0:
452 raise HTTPException(
453 status_code=400,
454 detail=f"Fixed margin amount for {provider} must be non-negative",
455 )
456 if not margin_value: # Empty dict
457 raise HTTPException(
458 status_code=400,
459 detail=f"Margin config for {provider} cannot be empty. Must include 'percentage' and/or 'fixed_amount'",
460 )
461 else:
462 raise HTTPException(
463 status_code=400,
464 detail=f"Margin for {provider} must be a number (percentage) or dict with 'percentage' and/or 'fixed_amount'",
465 )
467 try:
468 # Load existing config
469 config: Final = await proxy_config.get_config()
471 # Ensure litellm_settings exists
472 if "litellm_settings" not in config: 472 ↛ 473line 472 didn't jump to line 473 because the condition on line 472 was never true
473 config["litellm_settings"] = {}
475 # Update cost_margin_config
476 config["litellm_settings"]["cost_margin_config"] = cost_margin_config
478 # Save the updated config to DB
479 await proxy_config.save_config(new_config=config)
481 # Update in-memory litellm.cost_margin_config
482 litellm.cost_margin_config = cost_margin_config
484 verbose_proxy_logger.info("Updated cost_margin_config: %s", cost_margin_config)
486 return {
487 "message": "Cost margin configuration updated successfully",
488 "status": "success",
489 "values": cost_margin_config,
490 }
491 except Exception as e:
492 verbose_proxy_logger.error("Error updating cost margin config: %s", e)
493 raise HTTPException(
494 status_code=500,
495 detail={"error": f"Failed to update cost margin config: {e}"},
496 )
499class BlockUnpricedModelsRequest(BaseModel):
500 enabled: bool
503class BlockUnpricedModelsResponse(BaseModel):
504 enabled: bool
507@router.get(
508 "/config/block_requests_for_models_without_pricing",
509 tags=("Cost Tracking",),
510 dependencies=(Depends(user_api_key_auth),),
511 response_model=BlockUnpricedModelsResponse,
512)
513async def get_block_requests_for_models_without_pricing() -> BlockUnpricedModelsResponse:
514 return BlockUnpricedModelsResponse(enabled=bool(litellm.block_requests_for_models_without_pricing))
517@router.patch(
518 "/config/block_requests_for_models_without_pricing",
519 tags=("Cost Tracking",),
520 dependencies=(Depends(user_api_key_auth),),
521 response_model=BlockUnpricedModelsResponse,
522)
523async def update_block_requests_for_models_without_pricing(
524 request: BlockUnpricedModelsRequest,
525) -> BlockUnpricedModelsResponse:
526 from litellm.proxy.proxy_server import (
527 prisma_client,
528 proxy_config,
529 store_model_in_db,
530 )
532 if prisma_client is None: 532 ↛ 533line 532 didn't jump to line 533 because the condition on line 532 was never true
533 raise HTTPException(
534 status_code=500,
535 detail={ # mutable-ok: HTTPException detail must be a plain mapping
536 "error": CommonProxyErrors.db_not_connected_error.value
537 },
538 )
540 if store_model_in_db is not True: 540 ↛ 541line 540 didn't jump to line 541 because the condition on line 540 was never true
541 raise HTTPException(
542 status_code=500,
543 detail={ # mutable-ok: HTTPException detail must be a plain mapping
544 "error": "Set `'STORE_MODEL_IN_DB='True'` in your env to enable this feature."
545 },
546 )
548 try:
549 config = await proxy_config.get_config()
550 if "litellm_settings" not in config: 550 ↛ 551line 550 didn't jump to line 551 because the condition on line 550 was never true
551 config["litellm_settings"] = {} # mutable-ok: config is a plain-dict payload for save_config
552 config["litellm_settings"]["block_requests_for_models_without_pricing"] = request.enabled
553 await proxy_config.save_config(new_config=config)
555 litellm.block_requests_for_models_without_pricing = request.enabled
556 verbose_proxy_logger.info("Updated block_requests_for_models_without_pricing: %s", request.enabled)
558 return BlockUnpricedModelsResponse(enabled=request.enabled)
559 except Exception as e: # noqa: BLE001 # any config persistence failure must surface as a 500 response, not a crash
560 verbose_proxy_logger.error("Error updating block_requests_for_models_without_pricing: %s", e)
561 raise HTTPException(
562 status_code=500,
563 detail={ # mutable-ok: HTTPException detail must be a plain mapping
564 "error": f"Failed to update setting: {e!s}"
565 },
566 )
569@router.post(
570 "/cost/estimate",
571 tags=["Cost Tracking"],
572 dependencies=[Depends(user_api_key_auth)],
573 response_model=CostEstimateResponse,
574)
575async def estimate_cost(
576 request: CostEstimateRequest,
577 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth),
578) -> CostEstimateResponse:
579 """
580 Estimate cost for a given model and token counts.
582 This endpoint uses the same cost calculation logic as actual requests,
583 including any configured margins and discounts.
585 Parameters:
586 - model: Model name (e.g., "gpt-4", "claude-3-opus")
587 - input_tokens: Expected input tokens per request
588 - output_tokens: Expected output tokens per request
589 - cache_read_input_tokens: Cache-read tokens per request, counted within input_tokens (optional)
590 - cache_creation_input_tokens: Cache-write tokens per request, counted within input_tokens (optional)
591 - reasoning_tokens: Reasoning tokens per request, counted within output_tokens (optional)
592 - num_requests_per_day: Number of requests per day (optional)
593 - num_requests_per_month: Number of requests per month (optional)
595 Returns cost breakdown including:
596 - Per-request costs (input, output, margin, plus the cache-read, cache-write and reasoning shares)
597 - Daily costs (if num_requests_per_day provided)
598 - Monthly costs (if num_requests_per_month provided)
600 Example:
601 ```json
602 {
603 "model": "gpt-4",
604 "input_tokens": 1000,
605 "cache_read_input_tokens": 800,
606 "output_tokens": 500,
607 "reasoning_tokens": 200,
608 "num_requests_per_day": 100,
609 "num_requests_per_month": 3000
610 }
611 ```
612 """
613 from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj
615 # Resolve model name (handles router aliases like 'e-model-router' -> 'azure_ai/gpt-4')
616 resolved: Final = _resolve_model_for_cost_lookup(request.model)
617 resolved_model: Final = resolved.model
618 resolved_provider: Final = resolved.provider
620 verbose_proxy_logger.debug("Cost estimate: request.model='%s' resolved to '%s'", request.model, resolved_model)
622 usage: Final = _usage_for_estimate(request)
623 mock_response: Final = ModelResponse(model=resolved_model, usage=usage)
625 # Create a logging object to capture cost breakdown
626 litellm_logging_obj: Final = LiteLLMLoggingObj(
627 model=resolved_model,
628 messages=[],
629 stream=False,
630 call_type="completion",
631 start_time=None,
632 litellm_call_id="cost-estimate",
633 function_id="cost-estimate",
634 )
636 # Pinning one moment keeps an off-peak window that opens mid-quote from pricing the totals on
637 # one side of it and the reported rates on the other.
638 with pinned_billing_time(current_billing_time()):
639 # Use completion_cost which handles all the logic including margins/discounts
640 try:
641 cost_per_request: Final = completion_cost(
642 completion_response=mock_response,
643 model=resolved_model,
644 custom_llm_provider=resolved_provider,
645 custom_cost_per_token=resolved.custom_cost_per_token,
646 litellm_logging_obj=litellm_logging_obj,
647 )
648 except Exception as e: # noqa: BLE001 # completion_cost raises a bare Exception for an unpriceable model
649 raise HTTPException(
650 status_code=404,
651 detail={
652 "error": f"Could not calculate cost for model '{request.model}' (resolved to '{resolved_model}'): {e}"
653 },
654 )
656 # The rates come back from the pricing call itself rather than a second lookup, so they are the
657 # ones the cost lines above billed at even when completion_cost infers a provider this endpoint
658 # never resolved (an unrouted "xai/grok-4" prices on xai's inclusive tier thresholds; a lookup
659 # here without that provider would report the sub-200k rate for a line billed above it).
660 rates: Final = litellm_logging_obj.billed_token_rates
661 per_request: Final = _cost_lines(cost_per_request, litellm_logging_obj.cost_breakdown)
662 daily: Final = per_request.times(request.num_requests_per_day)
663 monthly: Final = per_request.times(request.num_requests_per_month)
665 model_info: Final = _lookup_model_info(resolved_model, resolved_provider)
666 mapped_provider: Final = model_info.get("litellm_provider") if model_info is not None else None
667 custom_llm_provider: Final = mapped_provider if mapped_provider is not None else resolved_provider
669 return CostEstimateResponse(
670 model=request.model,
671 input_tokens=request.input_tokens,
672 output_tokens=request.output_tokens,
673 cache_read_input_tokens=request.cache_read_input_tokens,
674 cache_creation_input_tokens=request.cache_creation_input_tokens,
675 reasoning_tokens=request.reasoning_tokens,
676 num_requests_per_day=request.num_requests_per_day,
677 num_requests_per_month=request.num_requests_per_month,
678 cost_per_request=per_request.total_cost,
679 input_cost_per_request=per_request.input_cost,
680 output_cost_per_request=per_request.output_cost,
681 margin_cost_per_request=per_request.margin_cost,
682 cache_read_cost_per_request=per_request.cache_read_cost,
683 cache_creation_cost_per_request=per_request.cache_creation_cost,
684 reasoning_cost_per_request=per_request.reasoning_cost,
685 daily_cost=daily.total_cost if daily is not None else None,
686 daily_input_cost=daily.input_cost if daily is not None else None,
687 daily_output_cost=daily.output_cost if daily is not None else None,
688 daily_margin_cost=daily.margin_cost if daily is not None else None,
689 daily_cache_read_cost=daily.cache_read_cost if daily is not None else None,
690 daily_cache_creation_cost=daily.cache_creation_cost if daily is not None else None,
691 daily_reasoning_cost=daily.reasoning_cost if daily is not None else None,
692 monthly_cost=monthly.total_cost if monthly is not None else None,
693 monthly_input_cost=monthly.input_cost if monthly is not None else None,
694 monthly_output_cost=monthly.output_cost if monthly is not None else None,
695 monthly_margin_cost=monthly.margin_cost if monthly is not None else None,
696 monthly_cache_read_cost=monthly.cache_read_cost if monthly is not None else None,
697 monthly_cache_creation_cost=monthly.cache_creation_cost if monthly is not None else None,
698 monthly_reasoning_cost=monthly.reasoning_cost if monthly is not None else None,
699 input_cost_per_token=rates.input_cost_per_token if rates is not None else None,
700 output_cost_per_token=rates.output_cost_per_token if rates is not None else None,
701 cache_read_input_token_cost=rates.cache_read_input_token_cost if rates is not None else None,
702 cache_creation_input_token_cost=rates.cache_creation_input_token_cost if rates is not None else None,
703 output_cost_per_reasoning_token=rates.output_cost_per_reasoning_token if rates is not None else None,
704 provider=custom_llm_provider,
705 )