Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/common_utils/prompt_cache_pricing.py: 29%
37 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1from collections.abc import Mapping
2from math import isfinite
3from typing import Final
5from pydantic import TypeAdapter
7import litellm
8from litellm.cost_calculator import (
9 _select_model_name_for_cost_calc, # pyright: ignore[reportPrivateUsage] # shares completion_cost's deployment tariff selection
10 completion_cost, # pyright: ignore[reportUnknownVariableType] # legacy optional parameters are untyped
11)
12from litellm.litellm_core_utils.litellm_logging import Logging
13from litellm.types.management_endpoints.prompt_cache_prediction import CacheTokenBuckets
14from litellm.types.utils import CacheCreationTokenDetails, ModelResponse, PromptTokensDetailsWrapper, Usage
16_PRICE_ENTRY: Final = TypeAdapter(Mapping[str, object])
19def _valid_price(value: object) -> bool:
20 return isinstance(value, (int, float)) and not isinstance(value, bool) and isfinite(value) and value >= 0
23def _has_required_prices(prices: Mapping[str, object], tokens: CacheTokenBuckets) -> bool:
24 required: Final = (
25 ("input_cost_per_token", True),
26 ("cache_read_input_token_cost", tokens.cache_read_input_tokens > 0),
27 ("cache_creation_input_token_cost", tokens.cache_creation_5m_input_tokens > 0),
28 ("cache_creation_input_token_cost_above_1hr", tokens.cache_creation_1h_input_tokens > 0),
29 )
30 if any(needed and not _valid_price(prices.get(key)) for key, needed in required):
31 return False
32 return all(
33 _valid_price(value)
34 for key, value in prices.items()
35 if value is not None and any(needed and key.startswith(f"{base}_above_") for base, needed in required)
36 )
39def price_cache_tokens(model: str, deployment_id: str, tokens: CacheTokenBuckets) -> float | None:
40 try:
41 selected_model: Final = _select_model_name_for_cost_calc(
42 model=model,
43 completion_response=None,
44 custom_pricing=True,
45 custom_llm_provider="anthropic",
46 router_model_id=deployment_id,
47 )
48 if selected_model is None:
49 return None
50 model_info: Final = litellm.get_model_info(model=selected_model, custom_llm_provider="anthropic")
51 registry: Final = _PRICE_ENTRY.validate_python(litellm.model_cost) # pyright: ignore[reportUnknownMemberType] # legacy registry is validated at this boundary
52 price_entry: Final = registry.get(model_info["key"])
53 if price_entry is None:
54 return None
55 prices: Final = _PRICE_ENTRY.validate_python(price_entry)
56 if not _has_required_prices(prices, tokens):
57 return None
58 usage: Final = Usage(
59 prompt_tokens=tokens.total_tokens,
60 completion_tokens=0,
61 total_tokens=tokens.total_tokens,
62 prompt_tokens_details=PromptTokensDetailsWrapper(
63 cached_tokens=tokens.cache_read_input_tokens,
64 cache_creation_tokens=tokens.cache_creation_5m_input_tokens + tokens.cache_creation_1h_input_tokens,
65 cache_creation_token_details=CacheCreationTokenDetails(
66 ephemeral_5m_input_tokens=tokens.cache_creation_5m_input_tokens,
67 ephemeral_1h_input_tokens=tokens.cache_creation_1h_input_tokens,
68 ),
69 ),
70 )
71 logging_obj: Final = Logging(
72 model=model,
73 messages=[], # mutable-ok: Logging requires a list
74 stream=False,
75 call_type="completion",
76 start_time=None,
77 litellm_call_id="prompt-cache-prediction",
78 function_id="prompt-cache-prediction",
79 )
80 completion_cost(
81 completion_response=ModelResponse(model=model, usage=usage),
82 model=model,
83 custom_llm_provider="anthropic",
84 custom_pricing=True,
85 router_model_id=deployment_id,
86 litellm_logging_obj=logging_obj,
87 )
88 cost: Final = logging_obj.cost_breakdown.get("input_cost") if logging_obj.cost_breakdown is not None else None
89 return cost if cost is not None and _valid_price(cost) else None
90 except Exception: # noqa: BLE001 # the shared pricing owners raise plain Exception for unpriceable models
91 return None