Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/spend_tracking/savings.py: 51%
261 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-10-10 12:01 +0000
1"""
2Per-request cost-savings computation for the Cost Optimization dashboard.
4Turns the token-level savings recorded on a request into dollar amounts using
5the model's own pricing. Daily rollup rows are keyed by date and entity, not by
6model, so the dollars have to be computed here (where the model and its prices
7are known) and summed into the daily tables; tokens cannot be priced after they
8have been aggregated across models.
9"""
11from collections.abc import Callable, Mapping
12from datetime import datetime
13from math import isclose, isfinite
14from types import MappingProxyType
15from typing import TYPE_CHECKING, Final, Literal, NamedTuple
17from pydantic import BaseModel, ConfigDict, Field
19import litellm
20from litellm._logging import verbose_proxy_logger
21from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY
22from litellm.litellm_core_utils.llm_cost_calc.utils import (
23 _get_cost_per_unit,
24 calculate_prompt_caching_savings,
25 generic_cost_per_token,
26)
27from litellm.types.integrations.anthropic_cache_control_hook import (
28 GATEWAY_INJECTED_CACHE_METADATA_KEY,
29 GATEWAY_INJECTED_FOR_EVERY_DEPLOYMENT,
30)
32if TYPE_CHECKING: 32 ↛ 33line 32 didn't jump to line 33 because the condition on line 32 was never true
33 from litellm.router import Router
34from litellm.types.utils import ModelInfo, PromptTokensDetailsWrapper, Usage
37class SavingsSpend(NamedTuple):
38 compression: float
39 prompt_caching: float
40 autorouter: float = 0.0
41 gateway_injected_caching: float = 0.0
44def _coerce_billed_at(value: datetime | str | None) -> datetime | None:
45 if isinstance(value, datetime) or value is None: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true
46 return value
47 try:
48 return datetime.fromisoformat(value.replace("Z", "+00:00"))
49 except ValueError:
50 return None
53class _ModelIdentity(NamedTuple):
54 model: str
55 provider: str
58def _resolve_model(model: str | None, custom_llm_provider: str | None) -> _ModelIdentity | None:
59 """Canonical ``(model, provider)``, or ``None`` when the model cannot be resolved.
61 The two sides of the comparison arrive spelled differently: the spend log records a
62 normalized model name alongside its provider, while the baseline arrives as the
63 operator wrote it in config, with the provider prefixed, implied, or absent. Raw
64 string equality therefore reads `anthropic/claude-opus-5` as a switch away from
65 `claude-opus-5`, and pricing a bare name with no provider can resolve it to a
66 different vendor's rates than the deployment it names.
67 """
68 if not model:
69 return None
70 try:
71 resolved_model, provider, _, _ = litellm.get_llm_provider(model=model, custom_llm_provider=custom_llm_provider)
72 except Exception as e: # noqa: BLE001 # get_llm_provider raises for unroutable names; degrade to an unavailable estimate
73 verbose_proxy_logger.debug(
74 "savings: cannot resolve provider for model=%s custom_llm_provider=%s (%s)", model, custom_llm_provider, e
75 )
76 return None
77 return _ModelIdentity(model=resolved_model, provider=provider)
80def _effective_model_info(router: "Router | None", deployment_id: str | None, model: str) -> ModelInfo | None:
81 """What a deployment is actually charged, or ``None`` to price by name.
83 `Router.get_deployment_model_info` owns this: it merges a deployment's configured
84 prices over the built-in map, folds in `base_model` defaults for deployments whose
85 name is not a model, and falls back to the model name when nothing is overridden.
86 Resolving a name here instead reads the public rate, which a deployment with a
87 negotiated price does not pay, and an Azure deployment name prices to nothing at all.
88 """
89 if router is None or deployment_id is None: 89 ↛ 90line 89 didn't jump to line 90 because the condition on line 89 was never true
90 return None
91 try:
92 return router.get_deployment_model_info(deployment_id, model)
93 except Exception as e: # noqa: BLE001 # a dashboard metric must not fail the spend write
94 verbose_proxy_logger.debug("savings: no deployment pricing for %s (%s)", model, e)
95 return None
98def _model_info(model: _ModelIdentity) -> ModelInfo | None:
99 """The public rates for ``model``, or ``None`` when it has none."""
100 try:
101 return litellm.get_model_info(model=model.model, custom_llm_provider=model.provider)
102 except Exception as e: # noqa: BLE001 # get_model_info raises bare Exception for unmapped models
103 verbose_proxy_logger.debug("savings: no pricing for provider=%s model=%s (%s)", model.provider, model.model, e)
104 return None
107class PricingBasis(NamedTuple):
108 """The tier and region a request was priced on, as the cost calculator resolved them.
110 Read back off the request's recorded ``cost_breakdown`` rather than re-derived. The
111 tier the biller used comes from ``optional_params``, which no log record carries, and
112 the served tier that does survive on the usage object is a different fact with the
113 opposite precedence, so a spend-time re-derivation would disagree with the invoice on
114 exactly the requests where the tier changed the price.
115 """
117 service_tier: str | None = None
118 data_residency: str | None = None
119 vertex_location: str | None = None
122_STANDARD_RATES: Final = PricingBasis()
125class BaselineCostSnapshot(BaseModel):
126 model_config = ConfigDict(extra="forbid", frozen=True, strict=True)
128 model: str
129 provider: str
130 prices: ModelInfo | None
131 basis: PricingBasis = _STANDARD_RATES
132 actual_spend: float = Field(allow_inf_nan=False, ge=0)
133 actual_token_cost: float | None = Field(default=None, allow_inf_nan=False, ge=0)
134 classifier_cost: float = Field(default=0.0, allow_inf_nan=False, ge=0)
137def baseline_cost_snapshot(
138 model: str,
139 prices: ModelInfo | None,
140 actual_spend: float,
141 cost_breakdown: Mapping[str, object] | None,
142 routing_decision: Mapping[str, object] | None,
143) -> BaselineCostSnapshot:
144 return BaselineCostSnapshot(
145 model=model,
146 provider="anthropic",
147 prices=prices,
148 actual_spend=actual_spend,
149 basis=_pricing_basis(cost_breakdown),
150 actual_token_cost=_recorded_token_cost(cost_breakdown),
151 classifier_cost=classifier_cost_from_decision(routing_decision) or 0.0,
152 )
155class BaselineCosts(NamedTuple):
156 actual: float
157 baseline: float
159 @property
160 def savings(self) -> float:
161 return self.baseline - self.actual
164def price_baseline_comparison(
165 snapshot: BaselineCostSnapshot,
166 baseline_usage: Usage | None,
167 provenance: Literal["observed_identical", "modeled"] | None,
168) -> BaselineCosts | None:
169 if baseline_usage is None or provenance is None:
170 return None
171 actual: Final = snapshot.actual_spend + snapshot.classifier_cost
172 if provenance == "observed_identical":
173 return BaselineCosts(actual=actual, baseline=snapshot.actual_spend)
174 if snapshot.prices is None or snapshot.actual_token_cost is None:
175 return None
176 token_cost: Final = _cost_of_usage(
177 _ModelIdentity(snapshot.model, snapshot.provider), baseline_usage, snapshot.prices, snapshot.basis
178 )
179 if token_cost is None or not isfinite(token_cost) or token_cost < 0:
180 return None
181 baseline: Final = snapshot.actual_spend + token_cost - snapshot.actual_token_cost
182 if not isfinite(baseline) or baseline < 0:
183 return None
184 return BaselineCosts(actual=actual, baseline=baseline)
187def _pricing_basis(cost_breakdown: Mapping[str, object] | None) -> PricingBasis:
188 """The basis recorded on a request, defaulting to standard rates when absent.
190 Rows written before this field shipped carry neither key, and there is no backfill:
191 they price at standard rates, which is what they already did.
193 These values survive a JSON round trip on the way here, so none is guaranteed to be
194 a string. `generic_cost_per_token` calls `.lower()` on them without a type check, and
195 the resulting `AttributeError` would be swallowed into a silent zero by the caller's
196 `except`, so anything that is not a string is dropped here instead.
197 """
198 if not cost_breakdown: 198 ↛ 200line 198 didn't jump to line 200 because the condition on line 198 was always true
199 return _STANDARD_RATES
200 service_tier: Final = cost_breakdown.get("service_tier")
201 data_residency: Final = cost_breakdown.get("data_residency")
202 vertex_location: Final = cost_breakdown.get("vertex_location")
203 return PricingBasis(
204 service_tier=service_tier if isinstance(service_tier, str) else None,
205 data_residency=data_residency if isinstance(data_residency, str) else None,
206 vertex_location=vertex_location if isinstance(vertex_location, str) else None,
207 )
210def _recorded_token_cost(cost_breakdown: Mapping[str, object] | None) -> float | None:
211 """What the biller charged for this request's tokens, or ``None`` when unrecorded.
213 ``input_cost`` already carries the cache buckets, so it and ``output_cost`` sum to
214 exactly what `generic_cost_per_token` returns for the same request; the separate
215 ``cache_read_cost`` and ``cache_creation_cost`` entries decompose that sum rather than
216 adding to it, and including them would charge those tokens twice.
218 Built-in tool cost, discount and margin are deliberately left out. They are properties
219 of the request and the operator's contract rather than of the model the router picked,
220 so they land on both sides of the comparison or neither, and only the total the
221 counterfactual can also be priced on belongs here.
222 """
223 if not cost_breakdown:
224 return None
225 input_cost: Final = cost_breakdown.get("input_cost")
226 output_cost: Final = cost_breakdown.get("output_cost")
227 if not isinstance(input_cost, (int, float)) or not isinstance(output_cost, (int, float)):
228 return None
229 return float(input_cost) + float(output_cost)
232def _cost_of_usage(
233 model: _ModelIdentity,
234 usage: Usage,
235 model_info: ModelInfo | None = None,
236 basis: PricingBasis = _STANDARD_RATES,
237) -> float | None:
238 """What ``usage`` costs on ``model``, or ``None`` when the model has no pricing."""
239 try:
240 if model.provider == "anthropic":
241 from litellm.llms.anthropic.cost_calculation import cost_per_token
243 prompt_cost, completion_cost = cost_per_token(
244 model=model.model,
245 usage=usage,
246 service_tier=basis.service_tier,
247 model_info=model_info,
248 )
249 else:
250 prompt_cost, completion_cost = generic_cost_per_token(
251 model=model.model,
252 usage=usage,
253 custom_llm_provider=model.provider,
254 service_tier=basis.service_tier,
255 data_residency=basis.data_residency,
256 model_info=model_info,
257 vertex_location=basis.vertex_location,
258 )
259 except Exception as e: # noqa: BLE001 # get_model_info raises bare Exception for unmapped models; degrade to zero savings
260 verbose_proxy_logger.debug(
261 "savings: cannot price usage for provider=%s model=%s (%s)", model.provider, model.model, e
262 )
263 return None
264 return prompt_cost + completion_cost
267def _cache_token_split(usage: Usage) -> tuple[int, int]:
268 """``(cache_read_tokens, cache_creation_tokens)`` for a request."""
269 details: Final = usage.prompt_tokens_details
270 if details is None:
271 return 0, 0
272 read: Final = getattr(details, "cached_tokens", 0) or 0
273 created = (getattr(details, "cache_creation_tokens", 0) or 0) or (getattr(details, "cache_write_tokens", 0) or 0)
274 return int(read), int(created)
277def _baseline_cache_rate_keys(baseline_info: ModelInfo | None) -> tuple[bool, bool]:
278 """Whether the baseline model has a ``(cache read, cache write)`` rate of its own.
280 A missing rate is not a free bucket. `_get_token_base_cost` resolves an absent
281 `cache_read_input_token_cost` or `cache_creation_input_token_cost` to 0.0, so a
282 baseline whose provider prices caching implicitly, which is every OpenAI, Azure and
283 Gemini entry for cache writes, would carry the whole prompt for nothing and turn a
284 profitable route into a reported loss. Such a model pays its plain input rate for
285 those tokens, so the buckets it cannot price become ordinary input below.
286 """
287 if baseline_info is None:
288 return True, True
289 return bool(baseline_info.get("cache_read_input_token_cost")), bool(
290 baseline_info.get("cache_creation_input_token_cost")
291 )
294def _baseline_usage(usage: Usage, baseline_info: ModelInfo | None = None) -> Usage:
295 cache_read, cache_creation = _cache_token_split(usage)
296 details: Final = usage.prompt_tokens_details
297 if details is None or (cache_read <= 0 and cache_creation <= 0):
298 return usage
299 prices_reads, prices_writes = _baseline_cache_rate_keys(baseline_info)
300 reads: Final = cache_read if prices_reads else 0
301 writes: Final = cache_creation if prices_writes else 0
302 if (reads, writes) == (cache_read, cache_creation):
303 return usage
304 other_modalities: Final = sum(
305 (getattr(details, field, 0) or 0) for field in ("audio_tokens", "image_tokens", "video_tokens")
306 )
307 return Usage(
308 **{
309 **usage.model_dump(),
310 # Rebuild through Usage so private fallback counts agree with the public buckets.
311 "cache_read_input_tokens": reads,
312 "cache_creation_input_tokens": writes,
313 "prompt_tokens_details": PromptTokensDetailsWrapper(
314 **{
315 **details.model_dump(),
316 "cached_tokens": reads,
317 "cache_creation_tokens": writes,
318 "cache_write_tokens": writes,
319 "cache_creation_token_details": details.cache_creation_token_details if writes else None,
320 "text_tokens": max(usage.prompt_tokens - reads - writes - other_modalities, 0),
321 }
322 ),
323 },
324 )
327def compute_autorouter_savings(
328 baseline_model: str | None,
329 selected_model: str | None,
330 selected_provider: str | None,
331 usage: Usage,
332 conversation_continuing: bool = True,
333 selected_info: ModelInfo | None = None,
334 baseline_info: ModelInfo | None = None,
335 cost_breakdown: Mapping[str, object] | None = None,
336 baseline_deployment_id: str | None = None,
337 selected_deployment_id: str | None = None,
338 baseline_usage: Usage | None = None,
339 baseline_provenance: Literal["observed_initial", "modeled"] | None = None,
340) -> float | None:
341 """Price established baseline usage; conversation shape cannot establish cache warmth."""
342 baseline: Final = _resolve_model(baseline_model, None)
343 selected: Final = _resolve_model(selected_model, selected_provider)
344 if baseline is None or selected is None:
345 return None
346 if baseline_usage is None and any(_cache_token_split(usage)):
347 return None
348 basis: Final = _pricing_basis(cost_breakdown)
349 effective_baseline_info: Final = baseline_info if baseline_info is not None else _model_info(baseline)
350 modeled_usage: Final = baseline_usage if baseline_usage is not None else usage
351 baseline_cost: Final = _cost_of_usage(
352 baseline, _baseline_usage(modeled_usage, effective_baseline_info), effective_baseline_info, basis
353 )
354 recorded_selected_cost: Final = _recorded_token_cost(cost_breakdown)
355 selected_cost: Final = (
356 recorded_selected_cost
357 if recorded_selected_cost is not None
358 else _cost_of_usage(selected, usage, selected_info, basis)
359 )
360 if baseline_cost is None or selected_cost is None:
361 return None
362 if baseline_provenance == "observed_initial":
363 same_prices: Final = effective_baseline_info == (
364 selected_info if selected_info is not None else _model_info(selected)
365 )
366 equivalent: Final = (
367 baseline_usage is not None
368 and baseline_usage == usage
369 and baseline == selected
370 and bool(baseline_deployment_id)
371 and baseline_deployment_id == selected_deployment_id
372 and same_prices
373 and recorded_selected_cost is not None
374 and isclose(baseline_cost, recorded_selected_cost, rel_tol=1e-9, abs_tol=1e-12)
375 )
376 return 0.0 if equivalent else None
377 difference: Final = baseline_cost - selected_cost
378 return difference if isfinite(difference) else None
381def _usage_from_spend_log(usage_object: Mapping[str, object] | None) -> Usage | None:
382 """Rebuild the request's ``Usage`` from the copy the spend log recorded."""
383 if not usage_object:
384 return None
385 try:
386 return Usage(**usage_object)
387 except Exception as e: # noqa: BLE001 # a malformed usage_object must not fail the daily spend write
388 # Warning, not debug: this silently zeroes the auto-router driver for every
389 # affected row, and a shape change in Usage would otherwise show up only as a
390 # dashboard that quietly reads $0.00.
391 verbose_proxy_logger.warning("savings: unusable usage_object, auto-router savings will read zero (%s)", e)
392 return None
395def marks_gateway_injection(metadata: Mapping[str, object] | None, model_id: str | None) -> bool:
396 """Whether the gateway put cache breakpoints on the payload THIS row was billed for.
398 ``AnthropicCacheControlHook.record_gateway_injection`` stamps the deployment it
399 injected for, and a row carries the deployment it was billed for, so the two agree
400 only on the leg that was actually injected. Every retry, failover and fallback of a
401 request shares one metadata bucket and one ``litellm_call_id``, so the deployment is
402 what tells those legs apart, and a marker left by a sibling reads here as no injection
403 without anyone having to strip it. An injection that ran before any deployment was
404 chosen is in the payload every leg sends, so it is marked for all of them and credits
405 each. Absent on requests the gateway never acted on
406 (client-supplied ``cache_control``, implicit provider caching) and on rows written
407 before the marker shipped; all of it is the fail-closed direction.
408 """
409 if not metadata: 409 ↛ 410line 409 didn't jump to line 410 because the condition on line 409 was never true
410 return False
411 injected_deployment: Final = metadata.get(GATEWAY_INJECTED_CACHE_METADATA_KEY)
412 if not isinstance(injected_deployment, str): 412 ↛ 414line 412 didn't jump to line 414 because the condition on line 412 was always true
413 return False
414 return injected_deployment in (GATEWAY_INJECTED_FOR_EVERY_DEPLOYMENT, model_id)
417def extract_cache_read_tokens(usage_object: Mapping[str, object] | None) -> int:
418 """Cache-read tokens from a logged usage object, whatever shape recorded them.
420 Anthropic writes a top-level ``cache_read_input_tokens``; OpenAI-compatible
421 providers (moonshotai, openai, deepseek, etc.) write
422 ``prompt_tokens_details.cached_tokens``. This is the one owner of that
423 normalization: callers hand over the usage object rather than threading a
424 count that could disagree with it.
425 """
426 if not usage_object:
427 return 0
428 explicit: Final = usage_object.get("cache_read_input_tokens")
429 if isinstance(explicit, (int, float)) and explicit: 429 ↛ 430line 429 didn't jump to line 430 because the condition on line 429 was never true
430 return int(explicit)
431 details: Final = usage_object.get("prompt_tokens_details")
432 if not isinstance(details, Mapping): 432 ↛ 434line 432 didn't jump to line 434 because the condition on line 432 was always true
433 return 0
434 cached: Final = details.get("cached_tokens")
435 return int(cached) if isinstance(cached, (int, float)) else 0
438def extract_cache_creation_tokens(usage_object: Mapping[str, object] | None) -> int:
439 """Cache-write tokens from a logged usage object, whatever shape recorded them.
441 Anthropic writes a top-level ``cache_creation_input_tokens``; OpenAI-compatible
442 providers (kimi-k2 etc.) write ``prompt_tokens_details.cache_write_tokens`` or
443 ``prompt_tokens_details.cache_creation_tokens``.
444 """
445 if not usage_object:
446 return 0
447 explicit: Final = usage_object.get("cache_creation_input_tokens")
448 if isinstance(explicit, (int, float)) and explicit: 448 ↛ 449line 448 didn't jump to line 449 because the condition on line 448 was never true
449 return int(explicit)
450 details: Final = usage_object.get("prompt_tokens_details")
451 if not isinstance(details, Mapping): 451 ↛ 453line 451 didn't jump to line 453 because the condition on line 451 was always true
452 return 0
453 written: Final = next(
454 (
455 value
456 for value in (details.get("cache_write_tokens"), details.get("cache_creation_tokens"))
457 if isinstance(value, (int, float)) and value
458 ),
459 0,
460 )
461 return int(written)
464def _proxy_llm_router() -> "Router | None":
465 """The running proxy's router, or ``None`` outside a proxy (public rates only)."""
466 try:
467 from litellm.proxy.proxy_server import llm_router
468 except Exception: # noqa: BLE001 # SDK-only usage has no proxy module to import
469 return None
470 return llm_router
473def _numeric_savings(value: object) -> float | None:
474 """``value`` as a recorded savings figure, or ``None`` when it is not one."""
475 if isinstance(value, bool) or not isinstance(value, (int, float)) or not isfinite(value): 475 ↛ 477line 475 didn't jump to line 477 because the condition on line 475 was always true
476 return None
477 return float(value)
480def recorded_estimated_autorouter_savings(metadata: Mapping[str, object]) -> float | None:
481 estimate: Final = metadata.get("autorouter_savings_estimate")
482 if (
483 not isinstance(estimate, Mapping)
484 or type(estimate.get("version")) is not int
485 or estimate.get("version") not in (1, 2, 3)
486 or estimate.get("status") != "estimated"
487 ):
488 return None
489 return _numeric_savings(metadata.get("autorouter_savings"))
492def classifier_cost_from_decision(routing_decision: Mapping[str, object] | None) -> float | None:
493 """The LLM-classifier cost a routing decision recorded, or ``None`` when it holds none.
495 ``None`` covers the decision-less request, the heuristic short-circuit that never
496 called a classifier, the unpriced classifier model, and a malformed value alike:
497 in every one of those cases there is no dollar figure to move, so callers treat
498 ``None`` as zero rather than as an error. The one owner of that reading, shared by
499 the savings netting, the session rollup and the response header, so the three can
500 never disagree about what counts as a classifier charge.
501 """
502 decision: Final = routing_decision if isinstance(routing_decision, Mapping) else {}
503 return _numeric_savings(decision.get("classifier_cost"))
506def autorouter_savings_for_request(
507 model: str | None,
508 custom_llm_provider: str | None,
509 routing_decision: Mapping[str, object] | None,
510 usage_object: Mapping[str, object] | None,
511 model_id: str | None = None,
512 llm_router: "Callable[[], Router | None] | None" = None,
513 cost_breakdown: Mapping[str, object] | None = None,
514 baseline_usage: Usage | None = None,
515 baseline_provenance: Literal["observed_initial", "modeled"] | None = None,
516) -> float | None:
517 """Return net savings for established usage, or None when the estimate is unavailable."""
518 usage: Final = _usage_from_spend_log(usage_object)
519 if usage is None or not model:
520 return None
521 decision: Final = routing_decision if isinstance(routing_decision, Mapping) else {}
522 recorded: Final = decision.get("savings_baseline_model")
523 recorded_id: Final = decision.get("savings_baseline_deployment_id")
524 baseline_model: Final = recorded if isinstance(recorded, str) else None
525 baseline_id: Final = recorded_id if isinstance(recorded_id, str) else None
526 if not decision or not baseline_model: 526 ↛ 528line 526 didn't jump to line 528 because the condition on line 526 was always true
527 return None
528 router_instance: Final = llm_router() if llm_router else None
529 gross: Final = compute_autorouter_savings(
530 baseline_model=baseline_model,
531 selected_model=model,
532 selected_provider=custom_llm_provider,
533 usage=usage,
534 selected_info=_effective_model_info(router_instance, model_id, model or ""),
535 baseline_info=_effective_model_info(router_instance, baseline_id, baseline_model or ""),
536 cost_breakdown=cost_breakdown,
537 baseline_deployment_id=baseline_id,
538 selected_deployment_id=model_id,
539 baseline_usage=baseline_usage,
540 baseline_provenance=baseline_provenance,
541 )
542 if gross is None:
543 return None
544 classifier_cost: Final = classifier_cost_from_decision(decision)
545 return gross if classifier_cost is None else gross - classifier_cost
548def autorouter_savings_for_logging_payload(
549 request_metadata: Mapping[str, object],
550 model: str | None,
551 custom_llm_provider: str | None,
552 model_id: str | None,
553 usage_object: Mapping[str, object] | None,
554 cost_breakdown: Mapping[str, object] | None,
555 baseline_usage: Usage | None = None,
556 baseline_provenance: Literal["observed_initial", "modeled"] | None = None,
557) -> float | None:
558 """The figure the logging payload records for a request, or ``None`` when none should be.
560 Internal sub-calls (the auto-router classifier, shadow eval's shadow and judge legs)
561 are excluded here for the same reason the spend writer zeroes them: they can carry a
562 real routing decision, but they are not requests the caller made, so a figure stamped
563 on them would report savings for traffic no user sent.
564 """
565 if request_metadata.get(INTERNAL_CALL_ORIGIN_METADATA_KEY): 565 ↛ 566line 565 didn't jump to line 566 because the condition on line 565 was never true
566 return None
567 routing_decision: Final = request_metadata.get("routing_decision")
568 return autorouter_savings_for_request(
569 model=model,
570 custom_llm_provider=custom_llm_provider,
571 routing_decision=routing_decision if isinstance(routing_decision, Mapping) else None,
572 usage_object=usage_object,
573 model_id=model_id,
574 llm_router=_proxy_llm_router,
575 cost_breakdown=cost_breakdown,
576 baseline_usage=baseline_usage,
577 baseline_provenance=baseline_provenance,
578 )
581def _request_savings_pricing(
582 model: str | None,
583 custom_llm_provider: str | None,
584 model_id: str | None,
585 llm_router: "Callable[[], Router | None] | None",
586) -> tuple[str | None, ModelInfo | None]:
587 router_instance: Final = llm_router() if llm_router else None
588 identity: Final = _resolve_model(model, custom_llm_provider)
589 pricing: Final = _effective_model_info(router_instance, model_id, model or "") or (
590 _model_info(identity) if identity else None
591 )
592 return identity.provider if identity else custom_llm_provider, pricing
595def _prompt_caching_savings(
596 pricing: ModelInfo | None,
597 provider: str | None,
598 usage_object: Mapping[str, object] | None,
599 cost_breakdown: Mapping[str, object] | None,
600 billed_at: datetime | str | None,
601) -> float | None:
602 usage: Final = _usage_from_spend_log(usage_object)
603 if pricing is None or usage is None:
604 return None
605 basis: Final = _pricing_basis(cost_breakdown)
606 result: Final = calculate_prompt_caching_savings(
607 model_info=pricing,
608 usage=usage,
609 custom_llm_provider=provider,
610 service_tier=basis.service_tier,
611 data_residency=basis.data_residency,
612 vertex_location=basis.vertex_location,
613 billed_at=_coerce_billed_at(billed_at),
614 )
615 return result if isfinite(result) else None
618def prompt_caching_savings_for_request(
619 model: str | None,
620 custom_llm_provider: str | None,
621 usage_object: Mapping[str, object] | None,
622 model_id: str | None = None,
623 llm_router: "Callable[[], Router | None] | None" = None,
624 cost_breakdown: Mapping[str, object] | None = None,
625 billed_at: datetime | str | None = None,
626) -> float | None:
627 request_pricing: Final = _request_savings_pricing(model, custom_llm_provider, model_id, llm_router)
628 return _prompt_caching_savings(request_pricing[1], request_pricing[0], usage_object, cost_breakdown, billed_at)
631def compute_savings_spend(
632 model: str | None,
633 custom_llm_provider: str | None,
634 compression_saved_tokens: int,
635 gateway_injected_cache: bool,
636 routing_decision: Mapping[str, object] | None = None,
637 usage_object: Mapping[str, object] | None = None,
638 model_id: str | None = None,
639 llm_router: "Callable[[], Router | None] | None" = None,
640 cost_breakdown: Mapping[str, object] | None = None,
641 recorded_autorouter_savings: object = None,
642 recorded_autorouter_savings_estimate: Mapping[str, object] | None = None,
643 billed_at: datetime | str | None = None,
644) -> SavingsSpend:
645 """
646 Dollar savings for one request, split by optimization driver.
648 Compression savings price the tokens compression removed at the model's
649 input rate. Prompt-caching savings are NET: the cache-read discount minus the
650 premium paid to write those entries, both derived here from ``usage_object`` so no
651 caller can hand in a count that disagrees with the usage record.
653 The uncached counterfactual pays the ordinary input rate for the same prompt size
654 and tier. Cache writes subtract only the premium over that rate, split by TTL.
655 Savings stay signed: a write-only request can lose money, and daily rollups net
656 those losses against read savings.
658 Caching is reported twice. ``prompt_caching`` is every net dollar caching saved,
659 whoever caused it, which is what a customer means by "what did caching save me".
660 ``gateway_injected_caching`` is the subset the gateway can claim credit for, carrying
661 a value only when ``gateway_injected_cache`` is set, i.e. litellm itself added the
662 ``cache_control`` breakpoints (configured injection points or the auto prompt-caching
663 flag). A client that sent its own breakpoints, and a provider that
664 caches implicitly (OpenAI, Gemini), produce the same usage shape with no gateway
665 action, so they count toward the total and not toward the attributed figure.
667 Reporting both rather than gating the one column keeps the customer-facing number
668 stable across the change and leaves attribution a separate question. The attributed
669 figure is normally the smaller of the two, being a subset of the same requests, but
670 not always: a request that only writes cache and never reads it has negative net
671 savings, and dropping such a request from the attributed figure can lift it above
672 the total. Auto-router savings compare established baseline usage against the
673 recorded selected-model cost. Versioned unknown estimates contribute no dollars
674 to this subtotal and are excluded from the separately reported coverage cohort.
676 ``llm_router`` is passed as a provider rather than a router because every spend write
677 calls this and only auto-routed ones need one, so looking it up eagerly at the call
678 site would fetch and discard it on the rest.
680 ``cost_breakdown`` supplies the biller's tier and region to caching and auto-router
681 savings. Caching also uses the logged prompt size and TTL split. Compression retains
682 its flat input-rate estimate; changing that counterfactual is a separate concern.
684 ``recorded_autorouter_savings`` is the figure the logging path stamped on the spend
685 log's metadata, honoured over recomputation so the rollup, the turn table and the
686 per-request record cannot disagree; rows written before the field shipped carry
687 nothing and recompute, mirroring ``_recorded_token_cost``.
688 """
689 # Deployment rates when the request came through one, public rates otherwise --
690 # `_effective_model_info` merges a deployment's configured prices over the built-in
691 # map, so a negotiated price is not silently replaced by the list rate.
692 request_pricing: Final = _request_savings_pricing(model, custom_llm_provider, model_id, llm_router)
693 provider: Final = request_pricing[0]
694 pricing: Final = request_pricing[1]
695 input_cost: Final = (_get_cost_per_unit(pricing, "input_cost_per_token") or 0.0) if pricing else 0.0
696 compression: Final = max(compression_saved_tokens, 0) * input_cost
697 prompt_caching: Final = _prompt_caching_savings(pricing, provider, usage_object, cost_breakdown, billed_at) or 0.0
698 gateway_injected_caching: Final = prompt_caching if gateway_injected_cache else 0.0
700 # The figure the logging path recorded wins, before the usage gate on purpose: a row
701 # whose usage no longer parses still carries the number computed when it did.
702 recorded_savings: Final = (
703 recorded_estimated_autorouter_savings(
704 MappingProxyType(
705 {
706 "autorouter_savings": recorded_autorouter_savings,
707 "autorouter_savings_estimate": recorded_autorouter_savings_estimate,
708 }
709 )
710 )
711 if recorded_autorouter_savings_estimate is not None
712 else _numeric_savings(recorded_autorouter_savings)
713 )
714 autorouter: Final = (
715 recorded_savings
716 if recorded_savings is not None or recorded_autorouter_savings_estimate is not None
717 else autorouter_savings_for_request(
718 model=model,
719 custom_llm_provider=custom_llm_provider,
720 routing_decision=routing_decision,
721 usage_object=usage_object,
722 model_id=model_id,
723 llm_router=llm_router,
724 cost_breakdown=cost_breakdown,
725 )
726 )
727 return SavingsSpend(
728 compression=compression,
729 prompt_caching=prompt_caching,
730 autorouter=0.0 if autorouter is None else autorouter,
731 gateway_injected_caching=gateway_injected_caching,
732 )