Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/spend_tracking/savings.py: 51%

261 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2Per-request cost-savings computation for the Cost Optimization dashboard. 

3 

4Turns the token-level savings recorded on a request into dollar amounts using 

5the model's own pricing. Daily rollup rows are keyed by date and entity, not by 

6model, so the dollars have to be computed here (where the model and its prices 

7are known) and summed into the daily tables; tokens cannot be priced after they 

8have been aggregated across models. 

9""" 

10 

11from collections.abc import Callable, Mapping 

12from datetime import datetime 

13from math import isclose, isfinite 

14from types import MappingProxyType 

15from typing import TYPE_CHECKING, Final, Literal, NamedTuple 

16 

17from pydantic import BaseModel, ConfigDict, Field 

18 

19import litellm 

20from litellm._logging import verbose_proxy_logger 

21from litellm.constants import INTERNAL_CALL_ORIGIN_METADATA_KEY 

22from litellm.litellm_core_utils.llm_cost_calc.utils import ( 

23 _get_cost_per_unit, 

24 calculate_prompt_caching_savings, 

25 generic_cost_per_token, 

26) 

27from litellm.types.integrations.anthropic_cache_control_hook import ( 

28 GATEWAY_INJECTED_CACHE_METADATA_KEY, 

29 GATEWAY_INJECTED_FOR_EVERY_DEPLOYMENT, 

30) 

31 

32if TYPE_CHECKING: 32 ↛ 33line 32 didn't jump to line 33 because the condition on line 32 was never true

33 from litellm.router import Router 

34from litellm.types.utils import ModelInfo, PromptTokensDetailsWrapper, Usage 

35 

36 

37class SavingsSpend(NamedTuple): 

38 compression: float 

39 prompt_caching: float 

40 autorouter: float = 0.0 

41 gateway_injected_caching: float = 0.0 

42 

43 

44def _coerce_billed_at(value: datetime | str | None) -> datetime | None: 

45 if isinstance(value, datetime) or value is None: 45 ↛ 46line 45 didn't jump to line 46 because the condition on line 45 was never true

46 return value 

47 try: 

48 return datetime.fromisoformat(value.replace("Z", "+00:00")) 

49 except ValueError: 

50 return None 

51 

52 

53class _ModelIdentity(NamedTuple): 

54 model: str 

55 provider: str 

56 

57 

58def _resolve_model(model: str | None, custom_llm_provider: str | None) -> _ModelIdentity | None: 

59 """Canonical ``(model, provider)``, or ``None`` when the model cannot be resolved. 

60 

61 The two sides of the comparison arrive spelled differently: the spend log records a 

62 normalized model name alongside its provider, while the baseline arrives as the 

63 operator wrote it in config, with the provider prefixed, implied, or absent. Raw 

64 string equality therefore reads `anthropic/claude-opus-5` as a switch away from 

65 `claude-opus-5`, and pricing a bare name with no provider can resolve it to a 

66 different vendor's rates than the deployment it names. 

67 """ 

68 if not model: 

69 return None 

70 try: 

71 resolved_model, provider, _, _ = litellm.get_llm_provider(model=model, custom_llm_provider=custom_llm_provider) 

72 except Exception as e: # noqa: BLE001 # get_llm_provider raises for unroutable names; degrade to an unavailable estimate 

73 verbose_proxy_logger.debug( 

74 "savings: cannot resolve provider for model=%s custom_llm_provider=%s (%s)", model, custom_llm_provider, e 

75 ) 

76 return None 

77 return _ModelIdentity(model=resolved_model, provider=provider) 

78 

79 

80def _effective_model_info(router: "Router | None", deployment_id: str | None, model: str) -> ModelInfo | None: 

81 """What a deployment is actually charged, or ``None`` to price by name. 

82 

83 `Router.get_deployment_model_info` owns this: it merges a deployment's configured 

84 prices over the built-in map, folds in `base_model` defaults for deployments whose 

85 name is not a model, and falls back to the model name when nothing is overridden. 

86 Resolving a name here instead reads the public rate, which a deployment with a 

87 negotiated price does not pay, and an Azure deployment name prices to nothing at all. 

88 """ 

89 if router is None or deployment_id is None: 89 ↛ 90line 89 didn't jump to line 90 because the condition on line 89 was never true

90 return None 

91 try: 

92 return router.get_deployment_model_info(deployment_id, model) 

93 except Exception as e: # noqa: BLE001 # a dashboard metric must not fail the spend write 

94 verbose_proxy_logger.debug("savings: no deployment pricing for %s (%s)", model, e) 

95 return None 

96 

97 

98def _model_info(model: _ModelIdentity) -> ModelInfo | None: 

99 """The public rates for ``model``, or ``None`` when it has none.""" 

100 try: 

101 return litellm.get_model_info(model=model.model, custom_llm_provider=model.provider) 

102 except Exception as e: # noqa: BLE001 # get_model_info raises bare Exception for unmapped models 

103 verbose_proxy_logger.debug("savings: no pricing for provider=%s model=%s (%s)", model.provider, model.model, e) 

104 return None 

105 

106 

107class PricingBasis(NamedTuple): 

108 """The tier and region a request was priced on, as the cost calculator resolved them. 

109 

110 Read back off the request's recorded ``cost_breakdown`` rather than re-derived. The 

111 tier the biller used comes from ``optional_params``, which no log record carries, and 

112 the served tier that does survive on the usage object is a different fact with the 

113 opposite precedence, so a spend-time re-derivation would disagree with the invoice on 

114 exactly the requests where the tier changed the price. 

115 """ 

116 

117 service_tier: str | None = None 

118 data_residency: str | None = None 

119 vertex_location: str | None = None 

120 

121 

122_STANDARD_RATES: Final = PricingBasis() 

123 

124 

125class BaselineCostSnapshot(BaseModel): 

126 model_config = ConfigDict(extra="forbid", frozen=True, strict=True) 

127 

128 model: str 

129 provider: str 

130 prices: ModelInfo | None 

131 basis: PricingBasis = _STANDARD_RATES 

132 actual_spend: float = Field(allow_inf_nan=False, ge=0) 

133 actual_token_cost: float | None = Field(default=None, allow_inf_nan=False, ge=0) 

134 classifier_cost: float = Field(default=0.0, allow_inf_nan=False, ge=0) 

135 

136 

137def baseline_cost_snapshot( 

138 model: str, 

139 prices: ModelInfo | None, 

140 actual_spend: float, 

141 cost_breakdown: Mapping[str, object] | None, 

142 routing_decision: Mapping[str, object] | None, 

143) -> BaselineCostSnapshot: 

144 return BaselineCostSnapshot( 

145 model=model, 

146 provider="anthropic", 

147 prices=prices, 

148 actual_spend=actual_spend, 

149 basis=_pricing_basis(cost_breakdown), 

150 actual_token_cost=_recorded_token_cost(cost_breakdown), 

151 classifier_cost=classifier_cost_from_decision(routing_decision) or 0.0, 

152 ) 

153 

154 

155class BaselineCosts(NamedTuple): 

156 actual: float 

157 baseline: float 

158 

159 @property 

160 def savings(self) -> float: 

161 return self.baseline - self.actual 

162 

163 

164def price_baseline_comparison( 

165 snapshot: BaselineCostSnapshot, 

166 baseline_usage: Usage | None, 

167 provenance: Literal["observed_identical", "modeled"] | None, 

168) -> BaselineCosts | None: 

169 if baseline_usage is None or provenance is None: 

170 return None 

171 actual: Final = snapshot.actual_spend + snapshot.classifier_cost 

172 if provenance == "observed_identical": 

173 return BaselineCosts(actual=actual, baseline=snapshot.actual_spend) 

174 if snapshot.prices is None or snapshot.actual_token_cost is None: 

175 return None 

176 token_cost: Final = _cost_of_usage( 

177 _ModelIdentity(snapshot.model, snapshot.provider), baseline_usage, snapshot.prices, snapshot.basis 

178 ) 

179 if token_cost is None or not isfinite(token_cost) or token_cost < 0: 

180 return None 

181 baseline: Final = snapshot.actual_spend + token_cost - snapshot.actual_token_cost 

182 if not isfinite(baseline) or baseline < 0: 

183 return None 

184 return BaselineCosts(actual=actual, baseline=baseline) 

185 

186 

187def _pricing_basis(cost_breakdown: Mapping[str, object] | None) -> PricingBasis: 

188 """The basis recorded on a request, defaulting to standard rates when absent. 

189 

190 Rows written before this field shipped carry neither key, and there is no backfill: 

191 they price at standard rates, which is what they already did. 

192 

193 These values survive a JSON round trip on the way here, so none is guaranteed to be 

194 a string. `generic_cost_per_token` calls `.lower()` on them without a type check, and 

195 the resulting `AttributeError` would be swallowed into a silent zero by the caller's 

196 `except`, so anything that is not a string is dropped here instead. 

197 """ 

198 if not cost_breakdown: 198 ↛ 200line 198 didn't jump to line 200 because the condition on line 198 was always true

199 return _STANDARD_RATES 

200 service_tier: Final = cost_breakdown.get("service_tier") 

201 data_residency: Final = cost_breakdown.get("data_residency") 

202 vertex_location: Final = cost_breakdown.get("vertex_location") 

203 return PricingBasis( 

204 service_tier=service_tier if isinstance(service_tier, str) else None, 

205 data_residency=data_residency if isinstance(data_residency, str) else None, 

206 vertex_location=vertex_location if isinstance(vertex_location, str) else None, 

207 ) 

208 

209 

210def _recorded_token_cost(cost_breakdown: Mapping[str, object] | None) -> float | None: 

211 """What the biller charged for this request's tokens, or ``None`` when unrecorded. 

212 

213 ``input_cost`` already carries the cache buckets, so it and ``output_cost`` sum to 

214 exactly what `generic_cost_per_token` returns for the same request; the separate 

215 ``cache_read_cost`` and ``cache_creation_cost`` entries decompose that sum rather than 

216 adding to it, and including them would charge those tokens twice. 

217 

218 Built-in tool cost, discount and margin are deliberately left out. They are properties 

219 of the request and the operator's contract rather than of the model the router picked, 

220 so they land on both sides of the comparison or neither, and only the total the 

221 counterfactual can also be priced on belongs here. 

222 """ 

223 if not cost_breakdown: 

224 return None 

225 input_cost: Final = cost_breakdown.get("input_cost") 

226 output_cost: Final = cost_breakdown.get("output_cost") 

227 if not isinstance(input_cost, (int, float)) or not isinstance(output_cost, (int, float)): 

228 return None 

229 return float(input_cost) + float(output_cost) 

230 

231 

232def _cost_of_usage( 

233 model: _ModelIdentity, 

234 usage: Usage, 

235 model_info: ModelInfo | None = None, 

236 basis: PricingBasis = _STANDARD_RATES, 

237) -> float | None: 

238 """What ``usage`` costs on ``model``, or ``None`` when the model has no pricing.""" 

239 try: 

240 if model.provider == "anthropic": 

241 from litellm.llms.anthropic.cost_calculation import cost_per_token 

242 

243 prompt_cost, completion_cost = cost_per_token( 

244 model=model.model, 

245 usage=usage, 

246 service_tier=basis.service_tier, 

247 model_info=model_info, 

248 ) 

249 else: 

250 prompt_cost, completion_cost = generic_cost_per_token( 

251 model=model.model, 

252 usage=usage, 

253 custom_llm_provider=model.provider, 

254 service_tier=basis.service_tier, 

255 data_residency=basis.data_residency, 

256 model_info=model_info, 

257 vertex_location=basis.vertex_location, 

258 ) 

259 except Exception as e: # noqa: BLE001 # get_model_info raises bare Exception for unmapped models; degrade to zero savings 

260 verbose_proxy_logger.debug( 

261 "savings: cannot price usage for provider=%s model=%s (%s)", model.provider, model.model, e 

262 ) 

263 return None 

264 return prompt_cost + completion_cost 

265 

266 

267def _cache_token_split(usage: Usage) -> tuple[int, int]: 

268 """``(cache_read_tokens, cache_creation_tokens)`` for a request.""" 

269 details: Final = usage.prompt_tokens_details 

270 if details is None: 

271 return 0, 0 

272 read: Final = getattr(details, "cached_tokens", 0) or 0 

273 created = (getattr(details, "cache_creation_tokens", 0) or 0) or (getattr(details, "cache_write_tokens", 0) or 0) 

274 return int(read), int(created) 

275 

276 

277def _baseline_cache_rate_keys(baseline_info: ModelInfo | None) -> tuple[bool, bool]: 

278 """Whether the baseline model has a ``(cache read, cache write)`` rate of its own. 

279 

280 A missing rate is not a free bucket. `_get_token_base_cost` resolves an absent 

281 `cache_read_input_token_cost` or `cache_creation_input_token_cost` to 0.0, so a 

282 baseline whose provider prices caching implicitly, which is every OpenAI, Azure and 

283 Gemini entry for cache writes, would carry the whole prompt for nothing and turn a 

284 profitable route into a reported loss. Such a model pays its plain input rate for 

285 those tokens, so the buckets it cannot price become ordinary input below. 

286 """ 

287 if baseline_info is None: 

288 return True, True 

289 return bool(baseline_info.get("cache_read_input_token_cost")), bool( 

290 baseline_info.get("cache_creation_input_token_cost") 

291 ) 

292 

293 

294def _baseline_usage(usage: Usage, baseline_info: ModelInfo | None = None) -> Usage: 

295 cache_read, cache_creation = _cache_token_split(usage) 

296 details: Final = usage.prompt_tokens_details 

297 if details is None or (cache_read <= 0 and cache_creation <= 0): 

298 return usage 

299 prices_reads, prices_writes = _baseline_cache_rate_keys(baseline_info) 

300 reads: Final = cache_read if prices_reads else 0 

301 writes: Final = cache_creation if prices_writes else 0 

302 if (reads, writes) == (cache_read, cache_creation): 

303 return usage 

304 other_modalities: Final = sum( 

305 (getattr(details, field, 0) or 0) for field in ("audio_tokens", "image_tokens", "video_tokens") 

306 ) 

307 return Usage( 

308 **{ 

309 **usage.model_dump(), 

310 # Rebuild through Usage so private fallback counts agree with the public buckets. 

311 "cache_read_input_tokens": reads, 

312 "cache_creation_input_tokens": writes, 

313 "prompt_tokens_details": PromptTokensDetailsWrapper( 

314 **{ 

315 **details.model_dump(), 

316 "cached_tokens": reads, 

317 "cache_creation_tokens": writes, 

318 "cache_write_tokens": writes, 

319 "cache_creation_token_details": details.cache_creation_token_details if writes else None, 

320 "text_tokens": max(usage.prompt_tokens - reads - writes - other_modalities, 0), 

321 } 

322 ), 

323 }, 

324 ) 

325 

326 

327def compute_autorouter_savings( 

328 baseline_model: str | None, 

329 selected_model: str | None, 

330 selected_provider: str | None, 

331 usage: Usage, 

332 conversation_continuing: bool = True, 

333 selected_info: ModelInfo | None = None, 

334 baseline_info: ModelInfo | None = None, 

335 cost_breakdown: Mapping[str, object] | None = None, 

336 baseline_deployment_id: str | None = None, 

337 selected_deployment_id: str | None = None, 

338 baseline_usage: Usage | None = None, 

339 baseline_provenance: Literal["observed_initial", "modeled"] | None = None, 

340) -> float | None: 

341 """Price established baseline usage; conversation shape cannot establish cache warmth.""" 

342 baseline: Final = _resolve_model(baseline_model, None) 

343 selected: Final = _resolve_model(selected_model, selected_provider) 

344 if baseline is None or selected is None: 

345 return None 

346 if baseline_usage is None and any(_cache_token_split(usage)): 

347 return None 

348 basis: Final = _pricing_basis(cost_breakdown) 

349 effective_baseline_info: Final = baseline_info if baseline_info is not None else _model_info(baseline) 

350 modeled_usage: Final = baseline_usage if baseline_usage is not None else usage 

351 baseline_cost: Final = _cost_of_usage( 

352 baseline, _baseline_usage(modeled_usage, effective_baseline_info), effective_baseline_info, basis 

353 ) 

354 recorded_selected_cost: Final = _recorded_token_cost(cost_breakdown) 

355 selected_cost: Final = ( 

356 recorded_selected_cost 

357 if recorded_selected_cost is not None 

358 else _cost_of_usage(selected, usage, selected_info, basis) 

359 ) 

360 if baseline_cost is None or selected_cost is None: 

361 return None 

362 if baseline_provenance == "observed_initial": 

363 same_prices: Final = effective_baseline_info == ( 

364 selected_info if selected_info is not None else _model_info(selected) 

365 ) 

366 equivalent: Final = ( 

367 baseline_usage is not None 

368 and baseline_usage == usage 

369 and baseline == selected 

370 and bool(baseline_deployment_id) 

371 and baseline_deployment_id == selected_deployment_id 

372 and same_prices 

373 and recorded_selected_cost is not None 

374 and isclose(baseline_cost, recorded_selected_cost, rel_tol=1e-9, abs_tol=1e-12) 

375 ) 

376 return 0.0 if equivalent else None 

377 difference: Final = baseline_cost - selected_cost 

378 return difference if isfinite(difference) else None 

379 

380 

381def _usage_from_spend_log(usage_object: Mapping[str, object] | None) -> Usage | None: 

382 """Rebuild the request's ``Usage`` from the copy the spend log recorded.""" 

383 if not usage_object: 

384 return None 

385 try: 

386 return Usage(**usage_object) 

387 except Exception as e: # noqa: BLE001 # a malformed usage_object must not fail the daily spend write 

388 # Warning, not debug: this silently zeroes the auto-router driver for every 

389 # affected row, and a shape change in Usage would otherwise show up only as a 

390 # dashboard that quietly reads $0.00. 

391 verbose_proxy_logger.warning("savings: unusable usage_object, auto-router savings will read zero (%s)", e) 

392 return None 

393 

394 

395def marks_gateway_injection(metadata: Mapping[str, object] | None, model_id: str | None) -> bool: 

396 """Whether the gateway put cache breakpoints on the payload THIS row was billed for. 

397 

398 ``AnthropicCacheControlHook.record_gateway_injection`` stamps the deployment it 

399 injected for, and a row carries the deployment it was billed for, so the two agree 

400 only on the leg that was actually injected. Every retry, failover and fallback of a 

401 request shares one metadata bucket and one ``litellm_call_id``, so the deployment is 

402 what tells those legs apart, and a marker left by a sibling reads here as no injection 

403 without anyone having to strip it. An injection that ran before any deployment was 

404 chosen is in the payload every leg sends, so it is marked for all of them and credits 

405 each. Absent on requests the gateway never acted on 

406 (client-supplied ``cache_control``, implicit provider caching) and on rows written 

407 before the marker shipped; all of it is the fail-closed direction. 

408 """ 

409 if not metadata: 409 ↛ 410line 409 didn't jump to line 410 because the condition on line 409 was never true

410 return False 

411 injected_deployment: Final = metadata.get(GATEWAY_INJECTED_CACHE_METADATA_KEY) 

412 if not isinstance(injected_deployment, str): 412 ↛ 414line 412 didn't jump to line 414 because the condition on line 412 was always true

413 return False 

414 return injected_deployment in (GATEWAY_INJECTED_FOR_EVERY_DEPLOYMENT, model_id) 

415 

416 

417def extract_cache_read_tokens(usage_object: Mapping[str, object] | None) -> int: 

418 """Cache-read tokens from a logged usage object, whatever shape recorded them. 

419 

420 Anthropic writes a top-level ``cache_read_input_tokens``; OpenAI-compatible 

421 providers (moonshotai, openai, deepseek, etc.) write 

422 ``prompt_tokens_details.cached_tokens``. This is the one owner of that 

423 normalization: callers hand over the usage object rather than threading a 

424 count that could disagree with it. 

425 """ 

426 if not usage_object: 

427 return 0 

428 explicit: Final = usage_object.get("cache_read_input_tokens") 

429 if isinstance(explicit, (int, float)) and explicit: 429 ↛ 430line 429 didn't jump to line 430 because the condition on line 429 was never true

430 return int(explicit) 

431 details: Final = usage_object.get("prompt_tokens_details") 

432 if not isinstance(details, Mapping): 432 ↛ 434line 432 didn't jump to line 434 because the condition on line 432 was always true

433 return 0 

434 cached: Final = details.get("cached_tokens") 

435 return int(cached) if isinstance(cached, (int, float)) else 0 

436 

437 

438def extract_cache_creation_tokens(usage_object: Mapping[str, object] | None) -> int: 

439 """Cache-write tokens from a logged usage object, whatever shape recorded them. 

440 

441 Anthropic writes a top-level ``cache_creation_input_tokens``; OpenAI-compatible 

442 providers (kimi-k2 etc.) write ``prompt_tokens_details.cache_write_tokens`` or 

443 ``prompt_tokens_details.cache_creation_tokens``. 

444 """ 

445 if not usage_object: 

446 return 0 

447 explicit: Final = usage_object.get("cache_creation_input_tokens") 

448 if isinstance(explicit, (int, float)) and explicit: 448 ↛ 449line 448 didn't jump to line 449 because the condition on line 448 was never true

449 return int(explicit) 

450 details: Final = usage_object.get("prompt_tokens_details") 

451 if not isinstance(details, Mapping): 451 ↛ 453line 451 didn't jump to line 453 because the condition on line 451 was always true

452 return 0 

453 written: Final = next( 

454 ( 

455 value 

456 for value in (details.get("cache_write_tokens"), details.get("cache_creation_tokens")) 

457 if isinstance(value, (int, float)) and value 

458 ), 

459 0, 

460 ) 

461 return int(written) 

462 

463 

464def _proxy_llm_router() -> "Router | None": 

465 """The running proxy's router, or ``None`` outside a proxy (public rates only).""" 

466 try: 

467 from litellm.proxy.proxy_server import llm_router 

468 except Exception: # noqa: BLE001 # SDK-only usage has no proxy module to import 

469 return None 

470 return llm_router 

471 

472 

473def _numeric_savings(value: object) -> float | None: 

474 """``value`` as a recorded savings figure, or ``None`` when it is not one.""" 

475 if isinstance(value, bool) or not isinstance(value, (int, float)) or not isfinite(value): 475 ↛ 477line 475 didn't jump to line 477 because the condition on line 475 was always true

476 return None 

477 return float(value) 

478 

479 

480def recorded_estimated_autorouter_savings(metadata: Mapping[str, object]) -> float | None: 

481 estimate: Final = metadata.get("autorouter_savings_estimate") 

482 if ( 

483 not isinstance(estimate, Mapping) 

484 or type(estimate.get("version")) is not int 

485 or estimate.get("version") not in (1, 2, 3) 

486 or estimate.get("status") != "estimated" 

487 ): 

488 return None 

489 return _numeric_savings(metadata.get("autorouter_savings")) 

490 

491 

492def classifier_cost_from_decision(routing_decision: Mapping[str, object] | None) -> float | None: 

493 """The LLM-classifier cost a routing decision recorded, or ``None`` when it holds none. 

494 

495 ``None`` covers the decision-less request, the heuristic short-circuit that never 

496 called a classifier, the unpriced classifier model, and a malformed value alike: 

497 in every one of those cases there is no dollar figure to move, so callers treat 

498 ``None`` as zero rather than as an error. The one owner of that reading, shared by 

499 the savings netting, the session rollup and the response header, so the three can 

500 never disagree about what counts as a classifier charge. 

501 """ 

502 decision: Final = routing_decision if isinstance(routing_decision, Mapping) else {} 

503 return _numeric_savings(decision.get("classifier_cost")) 

504 

505 

506def autorouter_savings_for_request( 

507 model: str | None, 

508 custom_llm_provider: str | None, 

509 routing_decision: Mapping[str, object] | None, 

510 usage_object: Mapping[str, object] | None, 

511 model_id: str | None = None, 

512 llm_router: "Callable[[], Router | None] | None" = None, 

513 cost_breakdown: Mapping[str, object] | None = None, 

514 baseline_usage: Usage | None = None, 

515 baseline_provenance: Literal["observed_initial", "modeled"] | None = None, 

516) -> float | None: 

517 """Return net savings for established usage, or None when the estimate is unavailable.""" 

518 usage: Final = _usage_from_spend_log(usage_object) 

519 if usage is None or not model: 

520 return None 

521 decision: Final = routing_decision if isinstance(routing_decision, Mapping) else {} 

522 recorded: Final = decision.get("savings_baseline_model") 

523 recorded_id: Final = decision.get("savings_baseline_deployment_id") 

524 baseline_model: Final = recorded if isinstance(recorded, str) else None 

525 baseline_id: Final = recorded_id if isinstance(recorded_id, str) else None 

526 if not decision or not baseline_model: 526 ↛ 528line 526 didn't jump to line 528 because the condition on line 526 was always true

527 return None 

528 router_instance: Final = llm_router() if llm_router else None 

529 gross: Final = compute_autorouter_savings( 

530 baseline_model=baseline_model, 

531 selected_model=model, 

532 selected_provider=custom_llm_provider, 

533 usage=usage, 

534 selected_info=_effective_model_info(router_instance, model_id, model or ""), 

535 baseline_info=_effective_model_info(router_instance, baseline_id, baseline_model or ""), 

536 cost_breakdown=cost_breakdown, 

537 baseline_deployment_id=baseline_id, 

538 selected_deployment_id=model_id, 

539 baseline_usage=baseline_usage, 

540 baseline_provenance=baseline_provenance, 

541 ) 

542 if gross is None: 

543 return None 

544 classifier_cost: Final = classifier_cost_from_decision(decision) 

545 return gross if classifier_cost is None else gross - classifier_cost 

546 

547 

548def autorouter_savings_for_logging_payload( 

549 request_metadata: Mapping[str, object], 

550 model: str | None, 

551 custom_llm_provider: str | None, 

552 model_id: str | None, 

553 usage_object: Mapping[str, object] | None, 

554 cost_breakdown: Mapping[str, object] | None, 

555 baseline_usage: Usage | None = None, 

556 baseline_provenance: Literal["observed_initial", "modeled"] | None = None, 

557) -> float | None: 

558 """The figure the logging payload records for a request, or ``None`` when none should be. 

559 

560 Internal sub-calls (the auto-router classifier, shadow eval's shadow and judge legs) 

561 are excluded here for the same reason the spend writer zeroes them: they can carry a 

562 real routing decision, but they are not requests the caller made, so a figure stamped 

563 on them would report savings for traffic no user sent. 

564 """ 

565 if request_metadata.get(INTERNAL_CALL_ORIGIN_METADATA_KEY): 565 ↛ 566line 565 didn't jump to line 566 because the condition on line 565 was never true

566 return None 

567 routing_decision: Final = request_metadata.get("routing_decision") 

568 return autorouter_savings_for_request( 

569 model=model, 

570 custom_llm_provider=custom_llm_provider, 

571 routing_decision=routing_decision if isinstance(routing_decision, Mapping) else None, 

572 usage_object=usage_object, 

573 model_id=model_id, 

574 llm_router=_proxy_llm_router, 

575 cost_breakdown=cost_breakdown, 

576 baseline_usage=baseline_usage, 

577 baseline_provenance=baseline_provenance, 

578 ) 

579 

580 

581def _request_savings_pricing( 

582 model: str | None, 

583 custom_llm_provider: str | None, 

584 model_id: str | None, 

585 llm_router: "Callable[[], Router | None] | None", 

586) -> tuple[str | None, ModelInfo | None]: 

587 router_instance: Final = llm_router() if llm_router else None 

588 identity: Final = _resolve_model(model, custom_llm_provider) 

589 pricing: Final = _effective_model_info(router_instance, model_id, model or "") or ( 

590 _model_info(identity) if identity else None 

591 ) 

592 return identity.provider if identity else custom_llm_provider, pricing 

593 

594 

595def _prompt_caching_savings( 

596 pricing: ModelInfo | None, 

597 provider: str | None, 

598 usage_object: Mapping[str, object] | None, 

599 cost_breakdown: Mapping[str, object] | None, 

600 billed_at: datetime | str | None, 

601) -> float | None: 

602 usage: Final = _usage_from_spend_log(usage_object) 

603 if pricing is None or usage is None: 

604 return None 

605 basis: Final = _pricing_basis(cost_breakdown) 

606 result: Final = calculate_prompt_caching_savings( 

607 model_info=pricing, 

608 usage=usage, 

609 custom_llm_provider=provider, 

610 service_tier=basis.service_tier, 

611 data_residency=basis.data_residency, 

612 vertex_location=basis.vertex_location, 

613 billed_at=_coerce_billed_at(billed_at), 

614 ) 

615 return result if isfinite(result) else None 

616 

617 

618def prompt_caching_savings_for_request( 

619 model: str | None, 

620 custom_llm_provider: str | None, 

621 usage_object: Mapping[str, object] | None, 

622 model_id: str | None = None, 

623 llm_router: "Callable[[], Router | None] | None" = None, 

624 cost_breakdown: Mapping[str, object] | None = None, 

625 billed_at: datetime | str | None = None, 

626) -> float | None: 

627 request_pricing: Final = _request_savings_pricing(model, custom_llm_provider, model_id, llm_router) 

628 return _prompt_caching_savings(request_pricing[1], request_pricing[0], usage_object, cost_breakdown, billed_at) 

629 

630 

631def compute_savings_spend( 

632 model: str | None, 

633 custom_llm_provider: str | None, 

634 compression_saved_tokens: int, 

635 gateway_injected_cache: bool, 

636 routing_decision: Mapping[str, object] | None = None, 

637 usage_object: Mapping[str, object] | None = None, 

638 model_id: str | None = None, 

639 llm_router: "Callable[[], Router | None] | None" = None, 

640 cost_breakdown: Mapping[str, object] | None = None, 

641 recorded_autorouter_savings: object = None, 

642 recorded_autorouter_savings_estimate: Mapping[str, object] | None = None, 

643 billed_at: datetime | str | None = None, 

644) -> SavingsSpend: 

645 """ 

646 Dollar savings for one request, split by optimization driver. 

647 

648 Compression savings price the tokens compression removed at the model's 

649 input rate. Prompt-caching savings are NET: the cache-read discount minus the 

650 premium paid to write those entries, both derived here from ``usage_object`` so no 

651 caller can hand in a count that disagrees with the usage record. 

652 

653 The uncached counterfactual pays the ordinary input rate for the same prompt size 

654 and tier. Cache writes subtract only the premium over that rate, split by TTL. 

655 Savings stay signed: a write-only request can lose money, and daily rollups net 

656 those losses against read savings. 

657 

658 Caching is reported twice. ``prompt_caching`` is every net dollar caching saved, 

659 whoever caused it, which is what a customer means by "what did caching save me". 

660 ``gateway_injected_caching`` is the subset the gateway can claim credit for, carrying 

661 a value only when ``gateway_injected_cache`` is set, i.e. litellm itself added the 

662 ``cache_control`` breakpoints (configured injection points or the auto prompt-caching 

663 flag). A client that sent its own breakpoints, and a provider that 

664 caches implicitly (OpenAI, Gemini), produce the same usage shape with no gateway 

665 action, so they count toward the total and not toward the attributed figure. 

666 

667 Reporting both rather than gating the one column keeps the customer-facing number 

668 stable across the change and leaves attribution a separate question. The attributed 

669 figure is normally the smaller of the two, being a subset of the same requests, but 

670 not always: a request that only writes cache and never reads it has negative net 

671 savings, and dropping such a request from the attributed figure can lift it above 

672 the total. Auto-router savings compare established baseline usage against the 

673 recorded selected-model cost. Versioned unknown estimates contribute no dollars 

674 to this subtotal and are excluded from the separately reported coverage cohort. 

675 

676 ``llm_router`` is passed as a provider rather than a router because every spend write 

677 calls this and only auto-routed ones need one, so looking it up eagerly at the call 

678 site would fetch and discard it on the rest. 

679 

680 ``cost_breakdown`` supplies the biller's tier and region to caching and auto-router 

681 savings. Caching also uses the logged prompt size and TTL split. Compression retains 

682 its flat input-rate estimate; changing that counterfactual is a separate concern. 

683 

684 ``recorded_autorouter_savings`` is the figure the logging path stamped on the spend 

685 log's metadata, honoured over recomputation so the rollup, the turn table and the 

686 per-request record cannot disagree; rows written before the field shipped carry 

687 nothing and recompute, mirroring ``_recorded_token_cost``. 

688 """ 

689 # Deployment rates when the request came through one, public rates otherwise -- 

690 # `_effective_model_info` merges a deployment's configured prices over the built-in 

691 # map, so a negotiated price is not silently replaced by the list rate. 

692 request_pricing: Final = _request_savings_pricing(model, custom_llm_provider, model_id, llm_router) 

693 provider: Final = request_pricing[0] 

694 pricing: Final = request_pricing[1] 

695 input_cost: Final = (_get_cost_per_unit(pricing, "input_cost_per_token") or 0.0) if pricing else 0.0 

696 compression: Final = max(compression_saved_tokens, 0) * input_cost 

697 prompt_caching: Final = _prompt_caching_savings(pricing, provider, usage_object, cost_breakdown, billed_at) or 0.0 

698 gateway_injected_caching: Final = prompt_caching if gateway_injected_cache else 0.0 

699 

700 # The figure the logging path recorded wins, before the usage gate on purpose: a row 

701 # whose usage no longer parses still carries the number computed when it did. 

702 recorded_savings: Final = ( 

703 recorded_estimated_autorouter_savings( 

704 MappingProxyType( 

705 { 

706 "autorouter_savings": recorded_autorouter_savings, 

707 "autorouter_savings_estimate": recorded_autorouter_savings_estimate, 

708 } 

709 ) 

710 ) 

711 if recorded_autorouter_savings_estimate is not None 

712 else _numeric_savings(recorded_autorouter_savings) 

713 ) 

714 autorouter: Final = ( 

715 recorded_savings 

716 if recorded_savings is not None or recorded_autorouter_savings_estimate is not None 

717 else autorouter_savings_for_request( 

718 model=model, 

719 custom_llm_provider=custom_llm_provider, 

720 routing_decision=routing_decision, 

721 usage_object=usage_object, 

722 model_id=model_id, 

723 llm_router=llm_router, 

724 cost_breakdown=cost_breakdown, 

725 ) 

726 ) 

727 return SavingsSpend( 

728 compression=compression, 

729 prompt_caching=prompt_caching, 

730 autorouter=0.0 if autorouter is None else autorouter, 

731 gateway_injected_caching=gateway_injected_caching, 

732 )