Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/management_endpoints/cost_tracking_settings.py: 59%

229 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1""" 

2COST TRACKING SETTINGS MANAGEMENT 

3 

4Endpoints for managing cost discount and margin configuration 

5 

6GET /config/cost_discount_config - Get current cost discount configuration 

7PATCH /config/cost_discount_config - Update cost discount configuration 

8GET /config/cost_margin_config - Get current cost margin configuration 

9PATCH /config/cost_margin_config - Update cost margin configuration 

10POST /cost/estimate - Estimate cost for a given model and token counts 

11""" 

12 

13from collections.abc import Mapping 

14from dataclasses import dataclass 

15from typing import Final 

16 

17from fastapi import APIRouter, Depends, HTTPException 

18from pydantic import BaseModel 

19 

20import litellm 

21from litellm._internal_context import current_billing_time, pinned_billing_time 

22from litellm._logging import verbose_proxy_logger 

23from litellm.cost_calculator import completion_cost 

24from litellm.proxy._types import ( 

25 CommonProxyErrors, 

26 CostEstimateRequest, 

27 CostEstimateResponse, 

28 UserAPIKeyAuth, 

29) 

30from litellm.proxy.auth.user_api_key_auth import user_api_key_auth 

31from litellm.proxy.management_endpoints.prompt_cache_prediction import router as prompt_cache_prediction_router 

32from litellm.types.utils import ( 

33 CostBreakdown, 

34 CostPerToken, 

35 LlmProvidersSet, 

36 ModelInfo, 

37 ModelResponse, 

38 PromptTokensDetailsWrapper, 

39 Usage, 

40) 

41 

42router: Final = APIRouter() 

43router.include_router(prompt_cache_prediction_router) 

44 

45 

46@dataclass(frozen=True, slots=True) 

47class ResolvedCostModel: 

48 model: str 

49 provider: str | None 

50 custom_cost_per_token: CostPerToken | None 

51 

52 

53def _configured_price(key: str, sources: tuple[Mapping[str, object], ...]) -> float | None: 

54 values: Final = (source.get(key) for source in sources) 

55 numeric: Final = (float(value) for value in values if isinstance(value, (int, float))) 

56 return next(numeric, None) 

57 

58 

59def _extract_custom_pricing( 

60 litellm_params: Mapping[str, object], model_info: Mapping[str, object], builtin: ModelInfo | None 

61) -> CostPerToken | None: 

62 """ 

63 Pull per-token pricing configured on a deployment so on-prem / self-hosted 

64 models (absent from the public cost map) still estimate a real cost. 

65 Pricing may live on ``litellm_params`` or ``model_info``; ``litellm_params`` 

66 wins, matching the router's cost-map registration precedence. Cache rates the 

67 deployment leaves unset come from the backend model's built-in entry, then its 

68 own input rate, again matching what the router registers for live billing. 

69 """ 

70 sources: Final = (litellm_params, model_info) 

71 input_price: Final = _configured_price("input_cost_per_token", sources) 

72 output_price: Final = _configured_price("output_cost_per_token", sources) 

73 

74 if input_price is None and output_price is None: 

75 return None 

76 

77 input_rate: Final = input_price or 0.0 

78 cache_sources: Final = sources if builtin is None else (*sources, builtin) 

79 cache_read_price: Final = _configured_price("cache_read_input_token_cost", cache_sources) 

80 cache_creation_price: Final = _configured_price("cache_creation_input_token_cost", cache_sources) 

81 return CostPerToken( 

82 input_cost_per_token=input_rate, 

83 output_cost_per_token=output_price or 0.0, 

84 cache_read_input_token_cost=input_rate if cache_read_price is None else cache_read_price, 

85 cache_creation_input_token_cost=input_rate if cache_creation_price is None else cache_creation_price, 

86 ) 

87 

88 

89def _lookup_model_info(model: str, custom_llm_provider: str | None = None) -> ModelInfo | None: 

90 try: 

91 return litellm.get_model_info(model=model, custom_llm_provider=custom_llm_provider) 

92 except Exception: 

93 return None 

94 

95 

96def _resolve_model_for_cost_lookup(model: str) -> ResolvedCostModel: 

97 """ 

98 Resolve a model name (which may be a router alias/model_group) to the 

99 underlying litellm model name, provider, and any deployment-configured 

100 pricing used for cost lookup. 

101 

102 Args: 

103 model: The model name from the request (could be a router alias like 'e-model-router' 

104 or an actual model name like 'azure_ai/gpt-4') 

105 """ 

106 from litellm.proxy.proxy_server import llm_router 

107 

108 # Try to resolve from router if available 

109 if llm_router is not None: 109 ↛ 133line 109 didn't jump to line 133 because the condition on line 109 was always true

110 try: 

111 # Get deployments for this model name (handles aliases, wildcards, etc.) 

112 deployments: Final = llm_router.get_model_list(model_name=model) 

113 

114 if deployments and len(deployments) > 0: 114 ↛ 115line 114 didn't jump to line 115 because the condition on line 114 was never true

115 first_deployment: Final = deployments[0] 

116 litellm_params: Final = first_deployment.get("litellm_params", {}) 

117 model_info: Final = first_deployment.get("model_info", {}) 

118 custom_llm_provider: Final = litellm_params.get("custom_llm_provider") 

119 provider: Final = str(custom_llm_provider) if custom_llm_provider is not None else None 

120 # base_model wins (needed for Azure custom deployment names) 

121 base_model: Final = model_info.get("base_model") or litellm_params.get("base_model") 

122 resolved_model: Final = base_model or litellm_params.get("model") 

123 if resolved_model: 

124 verbose_proxy_logger.debug("Resolved model '%s' to '%s' from router", model, resolved_model) 

125 custom_cost_per_token: Final = _extract_custom_pricing( 

126 litellm_params, model_info, _lookup_model_info(str(resolved_model), provider) 

127 ) 

128 return ResolvedCostModel(str(resolved_model), provider, custom_cost_per_token) 

129 except Exception as e: 

130 verbose_proxy_logger.debug("Could not resolve model '%s' from router: %s", model, e) 

131 

132 # Return original model if not resolved 

133 return ResolvedCostModel(model, None, None) 

134 

135 

136@dataclass(frozen=True, slots=True) 

137class CostLines: 

138 """Cost of one request split the way the spend logs split it: the cache lines are 

139 shares of input_cost and the reasoning line is a share of output_cost.""" 

140 

141 total_cost: float 

142 input_cost: float 

143 output_cost: float 

144 margin_cost: float 

145 cache_read_cost: float 

146 cache_creation_cost: float 

147 reasoning_cost: float 

148 

149 def times(self, num_requests: int | None) -> "CostLines | None": 

150 if not num_requests: 

151 return None 

152 return CostLines( 

153 total_cost=self.total_cost * num_requests, 

154 input_cost=self.input_cost * num_requests, 

155 output_cost=self.output_cost * num_requests, 

156 margin_cost=self.margin_cost * num_requests, 

157 cache_read_cost=self.cache_read_cost * num_requests, 

158 cache_creation_cost=self.cache_creation_cost * num_requests, 

159 reasoning_cost=self.reasoning_cost * num_requests, 

160 ) 

161 

162 

163def _cost_lines(cost_per_request: float, cost_breakdown: CostBreakdown | None) -> CostLines: 

164 breakdown: Final = cost_breakdown if cost_breakdown is not None else CostBreakdown() 

165 return CostLines( 

166 total_cost=cost_per_request, 

167 input_cost=breakdown.get("input_cost", 0.0), 

168 output_cost=breakdown.get("output_cost", 0.0), 

169 margin_cost=breakdown.get("margin_total_amount", 0.0), 

170 cache_read_cost=breakdown.get("cache_read_cost", 0.0), 

171 cache_creation_cost=breakdown.get("cache_creation_cost", 0.0), 

172 reasoning_cost=breakdown.get("reasoning_cost", 0.0), 

173 ) 

174 

175 

176def _usage_for_estimate(request: CostEstimateRequest) -> Usage: 

177 cache_tokens: Final = request.cache_read_input_tokens + request.cache_creation_input_tokens 

178 return Usage( 

179 prompt_tokens=request.input_tokens, 

180 completion_tokens=request.output_tokens, 

181 total_tokens=request.input_tokens + request.output_tokens, 

182 reasoning_tokens=request.reasoning_tokens, 

183 prompt_tokens_details=PromptTokensDetailsWrapper( 

184 cached_tokens=request.cache_read_input_tokens, 

185 cache_creation_tokens=request.cache_creation_input_tokens, 

186 ) 

187 if cache_tokens 

188 else None, 

189 ) 

190 

191 

192@router.get( 

193 "/config/cost_discount_config", 

194 tags=["Cost Tracking"], 

195 dependencies=[Depends(user_api_key_auth)], 

196) 

197async def get_cost_discount_config( 

198 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

199): 

200 """ 

201 Get current cost discount configuration. 

202 

203 Returns the cost_discount_config from litellm_settings. 

204 """ 

205 from litellm.proxy.proxy_server import prisma_client, proxy_config 

206 

207 if prisma_client is None: 207 ↛ 208line 207 didn't jump to line 208 because the condition on line 207 was never true

208 raise HTTPException( 

209 status_code=500, 

210 detail={"error": CommonProxyErrors.db_not_connected_error.value}, 

211 ) 

212 

213 try: 

214 # Load config from DB 

215 config: Final = await proxy_config.get_config() 

216 

217 # Get cost_discount_config from litellm_settings 

218 litellm_settings: Final = config.get("litellm_settings", {}) 

219 cost_discount_config: Final = litellm_settings.get("cost_discount_config", {}) 

220 

221 return {"values": cost_discount_config} 

222 except Exception as e: 

223 verbose_proxy_logger.error("Error fetching cost discount config: %s", e) 

224 return {"values": {}} 

225 

226 

227@router.patch( 

228 "/config/cost_discount_config", 

229 tags=["Cost Tracking"], 

230 dependencies=[Depends(user_api_key_auth)], 

231) 

232async def update_cost_discount_config( 

233 cost_discount_config: dict[str, float], 

234 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

235): 

236 """ 

237 Update cost discount configuration. 

238 

239 Updates the cost_discount_config in litellm_settings. 

240 Discounts should be between 0 and 1 (e.g., 0.05 = 5% discount). 

241 

242 Example: 

243 ```json 

244 { 

245 "vertex_ai": 0.05, 

246 "gemini": 0.05, 

247 "openai": 0.01 

248 } 

249 ``` 

250 """ 

251 from litellm.proxy.proxy_server import ( 

252 prisma_client, 

253 proxy_config, 

254 store_model_in_db, 

255 ) 

256 

257 if prisma_client is None: 257 ↛ 258line 257 didn't jump to line 258 because the condition on line 257 was never true

258 raise HTTPException( 

259 status_code=500, 

260 detail={"error": CommonProxyErrors.db_not_connected_error.value}, 

261 ) 

262 

263 if store_model_in_db is not True: 263 ↛ 264line 263 didn't jump to line 264 because the condition on line 263 was never true

264 raise HTTPException( 

265 status_code=500, 

266 detail={"error": "Set `'STORE_MODEL_IN_DB='True'` in your env to enable this feature."}, 

267 ) 

268 

269 # Validate that all providers are valid LiteLLM providers 

270 invalid_providers: Final = [] 

271 for provider in cost_discount_config: 

272 if provider not in LlmProvidersSet: 272 ↛ 271line 272 didn't jump to line 271 because the condition on line 272 was always true

273 invalid_providers.append(provider) 

274 

275 if invalid_providers: 

276 raise HTTPException( 

277 status_code=400, 

278 detail={ 

279 "error": f"Invalid provider(s): {', '.join(invalid_providers)}. Must be valid LiteLLM providers. See https://docs.litellm.ai/docs/providers for the full list." 

280 }, 

281 ) 

282 

283 # Validate discount values are between 0 and 1 

284 for provider, discount in cost_discount_config.items(): 284 ↛ 285line 284 didn't jump to line 285 because the loop on line 284 never started

285 if not isinstance(discount, (int, float)): 

286 raise HTTPException(status_code=400, detail=f"Discount for {provider} must be a number") 

287 if not (0 <= discount <= 1): 

288 raise HTTPException( 

289 status_code=400, 

290 detail=f"Discount for {provider} must be between 0 and 1 (0% to 100%)", 

291 ) 

292 

293 try: 

294 # Load existing config 

295 config: Final = await proxy_config.get_config() 

296 

297 # Ensure litellm_settings exists 

298 if "litellm_settings" not in config: 298 ↛ 299line 298 didn't jump to line 299 because the condition on line 298 was never true

299 config["litellm_settings"] = {} 

300 

301 # Update cost_discount_config 

302 config["litellm_settings"]["cost_discount_config"] = cost_discount_config 

303 

304 # Save the updated config to DB 

305 await proxy_config.save_config(new_config=config) 

306 

307 # Update in-memory litellm.cost_discount_config 

308 litellm.cost_discount_config = cost_discount_config 

309 

310 verbose_proxy_logger.info("Updated cost_discount_config: %s", cost_discount_config) 

311 

312 return { 

313 "message": "Cost discount configuration updated successfully", 

314 "status": "success", 

315 "values": cost_discount_config, 

316 } 

317 except Exception as e: 

318 verbose_proxy_logger.error("Error updating cost discount config: %s", e) 

319 raise HTTPException( 

320 status_code=500, 

321 detail={"error": f"Failed to update cost discount config: {e}"}, 

322 ) 

323 

324 

325@router.get( 

326 "/config/cost_margin_config", 

327 tags=["Cost Tracking"], 

328 dependencies=[Depends(user_api_key_auth)], 

329) 

330async def get_cost_margin_config( 

331 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

332): 

333 """ 

334 Get current cost margin configuration. 

335 

336 Returns the cost_margin_config from litellm_settings. 

337 """ 

338 from litellm.proxy.proxy_server import prisma_client, proxy_config 

339 

340 if prisma_client is None: 340 ↛ 341line 340 didn't jump to line 341 because the condition on line 340 was never true

341 raise HTTPException( 

342 status_code=500, 

343 detail={"error": CommonProxyErrors.db_not_connected_error.value}, 

344 ) 

345 

346 try: 

347 # Load config from DB 

348 config: Final = await proxy_config.get_config() 

349 

350 # Get cost_margin_config from litellm_settings 

351 litellm_settings: Final = config.get("litellm_settings", {}) 

352 cost_margin_config: Final = litellm_settings.get("cost_margin_config", {}) 

353 

354 return {"values": cost_margin_config} 

355 except Exception as e: 

356 verbose_proxy_logger.error("Error fetching cost margin config: %s", e) 

357 return {"values": {}} 

358 

359 

360@router.patch( 

361 "/config/cost_margin_config", 

362 tags=["Cost Tracking"], 

363 dependencies=[Depends(user_api_key_auth)], 

364) 

365async def update_cost_margin_config( 

366 cost_margin_config: dict[str, float | dict[str, float]], 

367 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

368): 

369 """ 

370 Update cost margin configuration. 

371 

372 Updates the cost_margin_config in litellm_settings. 

373 Margins can be: 

374 - Percentage: {"openai": 0.10} = 10% margin 

375 - Fixed amount: {"openai": {"fixed_amount": 0.001}} = $0.001 per request 

376 - Combined: {"vertex_ai": {"percentage": 0.08, "fixed_amount": 0.0005}} 

377 - Global: {"global": 0.05} = 5% global margin on all providers 

378 

379 Example: 

380 ```json 

381 { 

382 "global": 0.05, 

383 "openai": 0.10, 

384 "anthropic": {"fixed_amount": 0.001}, 

385 "vertex_ai": {"percentage": 0.08, "fixed_amount": 0.0005} 

386 } 

387 ``` 

388 """ 

389 from litellm.proxy.proxy_server import ( 

390 prisma_client, 

391 proxy_config, 

392 store_model_in_db, 

393 ) 

394 

395 if prisma_client is None: 395 ↛ 396line 395 didn't jump to line 396 because the condition on line 395 was never true

396 raise HTTPException( 

397 status_code=500, 

398 detail={"error": CommonProxyErrors.db_not_connected_error.value}, 

399 ) 

400 

401 if store_model_in_db is not True: 401 ↛ 402line 401 didn't jump to line 402 because the condition on line 401 was never true

402 raise HTTPException( 

403 status_code=500, 

404 detail={"error": "Set `'STORE_MODEL_IN_DB='True'` in your env to enable this feature."}, 

405 ) 

406 

407 # Validate that all providers are valid LiteLLM providers (except "global") 

408 invalid_providers: Final = [] 

409 for provider in cost_margin_config: 

410 if provider != "global" and provider not in LlmProvidersSet: 410 ↛ 409line 410 didn't jump to line 409 because the condition on line 410 was always true

411 invalid_providers.append(provider) 

412 

413 if invalid_providers: 

414 raise HTTPException( 

415 status_code=400, 

416 detail={ 

417 "error": f"Invalid provider(s): {', '.join(invalid_providers)}. Must be valid LiteLLM providers or 'global'. See https://docs.litellm.ai/docs/providers for the full list." 

418 }, 

419 ) 

420 

421 # Validate margin values 

422 for provider, margin_value in cost_margin_config.items(): 422 ↛ 423line 422 didn't jump to line 423 because the loop on line 422 never started

423 if isinstance(margin_value, (int, float)): 

424 # Simple percentage format: {"openai": 0.10} 

425 if not (0 <= margin_value <= 10): # Allow up to 1000% margin 

426 raise HTTPException( 

427 status_code=400, 

428 detail=f"Margin percentage for {provider} must be between 0 and 10 (0% to 1000%)", 

429 ) 

430 elif isinstance(margin_value, dict): 

431 # Complex format: {"percentage": 0.08, "fixed_amount": 0.0005} 

432 if "percentage" in margin_value: 

433 percentage = margin_value["percentage"] 

434 if not isinstance(percentage, (int, float)): 

435 raise HTTPException( 

436 status_code=400, 

437 detail=f"Margin percentage for {provider} must be a number", 

438 ) 

439 if not (0 <= percentage <= 10): 

440 raise HTTPException( 

441 status_code=400, 

442 detail=f"Margin percentage for {provider} must be between 0 and 10 (0% to 1000%)", 

443 ) 

444 if "fixed_amount" in margin_value: 

445 fixed_amount = margin_value["fixed_amount"] 

446 if not isinstance(fixed_amount, (int, float)): 

447 raise HTTPException( 

448 status_code=400, 

449 detail=f"Fixed margin amount for {provider} must be a number", 

450 ) 

451 if fixed_amount < 0: 

452 raise HTTPException( 

453 status_code=400, 

454 detail=f"Fixed margin amount for {provider} must be non-negative", 

455 ) 

456 if not margin_value: # Empty dict 

457 raise HTTPException( 

458 status_code=400, 

459 detail=f"Margin config for {provider} cannot be empty. Must include 'percentage' and/or 'fixed_amount'", 

460 ) 

461 else: 

462 raise HTTPException( 

463 status_code=400, 

464 detail=f"Margin for {provider} must be a number (percentage) or dict with 'percentage' and/or 'fixed_amount'", 

465 ) 

466 

467 try: 

468 # Load existing config 

469 config: Final = await proxy_config.get_config() 

470 

471 # Ensure litellm_settings exists 

472 if "litellm_settings" not in config: 472 ↛ 473line 472 didn't jump to line 473 because the condition on line 472 was never true

473 config["litellm_settings"] = {} 

474 

475 # Update cost_margin_config 

476 config["litellm_settings"]["cost_margin_config"] = cost_margin_config 

477 

478 # Save the updated config to DB 

479 await proxy_config.save_config(new_config=config) 

480 

481 # Update in-memory litellm.cost_margin_config 

482 litellm.cost_margin_config = cost_margin_config 

483 

484 verbose_proxy_logger.info("Updated cost_margin_config: %s", cost_margin_config) 

485 

486 return { 

487 "message": "Cost margin configuration updated successfully", 

488 "status": "success", 

489 "values": cost_margin_config, 

490 } 

491 except Exception as e: 

492 verbose_proxy_logger.error("Error updating cost margin config: %s", e) 

493 raise HTTPException( 

494 status_code=500, 

495 detail={"error": f"Failed to update cost margin config: {e}"}, 

496 ) 

497 

498 

499class BlockUnpricedModelsRequest(BaseModel): 

500 enabled: bool 

501 

502 

503class BlockUnpricedModelsResponse(BaseModel): 

504 enabled: bool 

505 

506 

507@router.get( 

508 "/config/block_requests_for_models_without_pricing", 

509 tags=("Cost Tracking",), 

510 dependencies=(Depends(user_api_key_auth),), 

511 response_model=BlockUnpricedModelsResponse, 

512) 

513async def get_block_requests_for_models_without_pricing() -> BlockUnpricedModelsResponse: 

514 return BlockUnpricedModelsResponse(enabled=bool(litellm.block_requests_for_models_without_pricing)) 

515 

516 

517@router.patch( 

518 "/config/block_requests_for_models_without_pricing", 

519 tags=("Cost Tracking",), 

520 dependencies=(Depends(user_api_key_auth),), 

521 response_model=BlockUnpricedModelsResponse, 

522) 

523async def update_block_requests_for_models_without_pricing( 

524 request: BlockUnpricedModelsRequest, 

525) -> BlockUnpricedModelsResponse: 

526 from litellm.proxy.proxy_server import ( 

527 prisma_client, 

528 proxy_config, 

529 store_model_in_db, 

530 ) 

531 

532 if prisma_client is None: 532 ↛ 533line 532 didn't jump to line 533 because the condition on line 532 was never true

533 raise HTTPException( 

534 status_code=500, 

535 detail={ # mutable-ok: HTTPException detail must be a plain mapping 

536 "error": CommonProxyErrors.db_not_connected_error.value 

537 }, 

538 ) 

539 

540 if store_model_in_db is not True: 540 ↛ 541line 540 didn't jump to line 541 because the condition on line 540 was never true

541 raise HTTPException( 

542 status_code=500, 

543 detail={ # mutable-ok: HTTPException detail must be a plain mapping 

544 "error": "Set `'STORE_MODEL_IN_DB='True'` in your env to enable this feature." 

545 }, 

546 ) 

547 

548 try: 

549 config = await proxy_config.get_config() 

550 if "litellm_settings" not in config: 550 ↛ 551line 550 didn't jump to line 551 because the condition on line 550 was never true

551 config["litellm_settings"] = {} # mutable-ok: config is a plain-dict payload for save_config 

552 config["litellm_settings"]["block_requests_for_models_without_pricing"] = request.enabled 

553 await proxy_config.save_config(new_config=config) 

554 

555 litellm.block_requests_for_models_without_pricing = request.enabled 

556 verbose_proxy_logger.info("Updated block_requests_for_models_without_pricing: %s", request.enabled) 

557 

558 return BlockUnpricedModelsResponse(enabled=request.enabled) 

559 except Exception as e: # noqa: BLE001 # any config persistence failure must surface as a 500 response, not a crash 

560 verbose_proxy_logger.error("Error updating block_requests_for_models_without_pricing: %s", e) 

561 raise HTTPException( 

562 status_code=500, 

563 detail={ # mutable-ok: HTTPException detail must be a plain mapping 

564 "error": f"Failed to update setting: {e!s}" 

565 }, 

566 ) 

567 

568 

569@router.post( 

570 "/cost/estimate", 

571 tags=["Cost Tracking"], 

572 dependencies=[Depends(user_api_key_auth)], 

573 response_model=CostEstimateResponse, 

574) 

575async def estimate_cost( 

576 request: CostEstimateRequest, 

577 user_api_key_dict: UserAPIKeyAuth = Depends(user_api_key_auth), 

578) -> CostEstimateResponse: 

579 """ 

580 Estimate cost for a given model and token counts. 

581 

582 This endpoint uses the same cost calculation logic as actual requests, 

583 including any configured margins and discounts. 

584 

585 Parameters: 

586 - model: Model name (e.g., "gpt-4", "claude-3-opus") 

587 - input_tokens: Expected input tokens per request 

588 - output_tokens: Expected output tokens per request 

589 - cache_read_input_tokens: Cache-read tokens per request, counted within input_tokens (optional) 

590 - cache_creation_input_tokens: Cache-write tokens per request, counted within input_tokens (optional) 

591 - reasoning_tokens: Reasoning tokens per request, counted within output_tokens (optional) 

592 - num_requests_per_day: Number of requests per day (optional) 

593 - num_requests_per_month: Number of requests per month (optional) 

594 

595 Returns cost breakdown including: 

596 - Per-request costs (input, output, margin, plus the cache-read, cache-write and reasoning shares) 

597 - Daily costs (if num_requests_per_day provided) 

598 - Monthly costs (if num_requests_per_month provided) 

599 

600 Example: 

601 ```json 

602 { 

603 "model": "gpt-4", 

604 "input_tokens": 1000, 

605 "cache_read_input_tokens": 800, 

606 "output_tokens": 500, 

607 "reasoning_tokens": 200, 

608 "num_requests_per_day": 100, 

609 "num_requests_per_month": 3000 

610 } 

611 ``` 

612 """ 

613 from litellm.litellm_core_utils.litellm_logging import Logging as LiteLLMLoggingObj 

614 

615 # Resolve model name (handles router aliases like 'e-model-router' -> 'azure_ai/gpt-4') 

616 resolved: Final = _resolve_model_for_cost_lookup(request.model) 

617 resolved_model: Final = resolved.model 

618 resolved_provider: Final = resolved.provider 

619 

620 verbose_proxy_logger.debug("Cost estimate: request.model='%s' resolved to '%s'", request.model, resolved_model) 

621 

622 usage: Final = _usage_for_estimate(request) 

623 mock_response: Final = ModelResponse(model=resolved_model, usage=usage) 

624 

625 # Create a logging object to capture cost breakdown 

626 litellm_logging_obj: Final = LiteLLMLoggingObj( 

627 model=resolved_model, 

628 messages=[], 

629 stream=False, 

630 call_type="completion", 

631 start_time=None, 

632 litellm_call_id="cost-estimate", 

633 function_id="cost-estimate", 

634 ) 

635 

636 # Pinning one moment keeps an off-peak window that opens mid-quote from pricing the totals on 

637 # one side of it and the reported rates on the other. 

638 with pinned_billing_time(current_billing_time()): 

639 # Use completion_cost which handles all the logic including margins/discounts 

640 try: 

641 cost_per_request: Final = completion_cost( 

642 completion_response=mock_response, 

643 model=resolved_model, 

644 custom_llm_provider=resolved_provider, 

645 custom_cost_per_token=resolved.custom_cost_per_token, 

646 litellm_logging_obj=litellm_logging_obj, 

647 ) 

648 except Exception as e: # noqa: BLE001 # completion_cost raises a bare Exception for an unpriceable model 

649 raise HTTPException( 

650 status_code=404, 

651 detail={ 

652 "error": f"Could not calculate cost for model '{request.model}' (resolved to '{resolved_model}'): {e}" 

653 }, 

654 ) 

655 

656 # The rates come back from the pricing call itself rather than a second lookup, so they are the 

657 # ones the cost lines above billed at even when completion_cost infers a provider this endpoint 

658 # never resolved (an unrouted "xai/grok-4" prices on xai's inclusive tier thresholds; a lookup 

659 # here without that provider would report the sub-200k rate for a line billed above it). 

660 rates: Final = litellm_logging_obj.billed_token_rates 

661 per_request: Final = _cost_lines(cost_per_request, litellm_logging_obj.cost_breakdown) 

662 daily: Final = per_request.times(request.num_requests_per_day) 

663 monthly: Final = per_request.times(request.num_requests_per_month) 

664 

665 model_info: Final = _lookup_model_info(resolved_model, resolved_provider) 

666 mapped_provider: Final = model_info.get("litellm_provider") if model_info is not None else None 

667 custom_llm_provider: Final = mapped_provider if mapped_provider is not None else resolved_provider 

668 

669 return CostEstimateResponse( 

670 model=request.model, 

671 input_tokens=request.input_tokens, 

672 output_tokens=request.output_tokens, 

673 cache_read_input_tokens=request.cache_read_input_tokens, 

674 cache_creation_input_tokens=request.cache_creation_input_tokens, 

675 reasoning_tokens=request.reasoning_tokens, 

676 num_requests_per_day=request.num_requests_per_day, 

677 num_requests_per_month=request.num_requests_per_month, 

678 cost_per_request=per_request.total_cost, 

679 input_cost_per_request=per_request.input_cost, 

680 output_cost_per_request=per_request.output_cost, 

681 margin_cost_per_request=per_request.margin_cost, 

682 cache_read_cost_per_request=per_request.cache_read_cost, 

683 cache_creation_cost_per_request=per_request.cache_creation_cost, 

684 reasoning_cost_per_request=per_request.reasoning_cost, 

685 daily_cost=daily.total_cost if daily is not None else None, 

686 daily_input_cost=daily.input_cost if daily is not None else None, 

687 daily_output_cost=daily.output_cost if daily is not None else None, 

688 daily_margin_cost=daily.margin_cost if daily is not None else None, 

689 daily_cache_read_cost=daily.cache_read_cost if daily is not None else None, 

690 daily_cache_creation_cost=daily.cache_creation_cost if daily is not None else None, 

691 daily_reasoning_cost=daily.reasoning_cost if daily is not None else None, 

692 monthly_cost=monthly.total_cost if monthly is not None else None, 

693 monthly_input_cost=monthly.input_cost if monthly is not None else None, 

694 monthly_output_cost=monthly.output_cost if monthly is not None else None, 

695 monthly_margin_cost=monthly.margin_cost if monthly is not None else None, 

696 monthly_cache_read_cost=monthly.cache_read_cost if monthly is not None else None, 

697 monthly_cache_creation_cost=monthly.cache_creation_cost if monthly is not None else None, 

698 monthly_reasoning_cost=monthly.reasoning_cost if monthly is not None else None, 

699 input_cost_per_token=rates.input_cost_per_token if rates is not None else None, 

700 output_cost_per_token=rates.output_cost_per_token if rates is not None else None, 

701 cache_read_input_token_cost=rates.cache_read_input_token_cost if rates is not None else None, 

702 cache_creation_input_token_cost=rates.cache_creation_input_token_cost if rates is not None else None, 

703 output_cost_per_reasoning_token=rates.output_cost_per_reasoning_token if rates is not None else None, 

704 provider=custom_llm_provider, 

705 )