Coverage for .venv/lib/python3.13/site-packages/litellm/proxy/hooks/model_max_budget_limiter.py: 37%

201 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-10-10 12:01 +0000

1import asyncio 

2import json 

3import time 

4from collections.abc import Iterable, Mapping, Sequence 

5from dataclasses import dataclass 

6from types import MappingProxyType 

7from typing import Final 

8 

9from openai.types import Batch 

10 

11import litellm 

12from litellm._logging import verbose_proxy_logger 

13from litellm.caching.caching import DualCache 

14from litellm.integrations.custom_logger import Span 

15from litellm.litellm_core_utils.duration_parser import duration_in_seconds 

16from litellm.llms.bedrock.common_utils import get_bedrock_base_model 

17from litellm.proxy._types import Litellm_EntityType, UserAPIKeyAuth 

18from litellm.router_strategy.budget_limiter import RouterBudgetLimiting 

19from litellm.router_utils.batch_utils import is_batch_retrieve_call_type 

20from litellm.types.llms.openai import AllMessageValues 

21from litellm.types.utils import BudgetConfig, StandardLoggingPayload 

22 

23VIRTUAL_KEY_SPEND_CACHE_KEY_PREFIX: Final = "virtual_key_spend" 

24END_USER_SPEND_CACHE_KEY_PREFIX: Final = "end_user_model_spend" 

25USER_SPEND_CACHE_KEY_PREFIX: Final = "user_model_spend" 

26TEAM_SPEND_CACHE_KEY_PREFIX: Final = "team_model_spend" 

27 

28_SPEND_CACHE_KEY_PREFIXES: Final = MappingProxyType( 

29 { 

30 Litellm_EntityType.KEY: VIRTUAL_KEY_SPEND_CACHE_KEY_PREFIX, 

31 Litellm_EntityType.USER: USER_SPEND_CACHE_KEY_PREFIX, 

32 Litellm_EntityType.END_USER: END_USER_SPEND_CACHE_KEY_PREFIX, 

33 Litellm_EntityType.TEAM: TEAM_SPEND_CACHE_KEY_PREFIX, 

34 } 

35) 

36 

37_LEGACY_REQUEST_MODEL_SCOPES: Final = frozenset({Litellm_EntityType.KEY, Litellm_EntityType.END_USER}) 

38 

39_PROCESS_STARTED_AT: Final = time.monotonic() 

40 

41_BUDGET_START_TIME_KEY_PREFIXES: Final = MappingProxyType( 

42 { 

43 Litellm_EntityType.KEY: "virtual_key_budget_start_time", 

44 Litellm_EntityType.USER: "user_model_budget_start_time", 

45 Litellm_EntityType.END_USER: "end_user_budget_start_time", 

46 Litellm_EntityType.TEAM: "team_model_budget_start_time", 

47 } 

48) 

49 

50 

51@dataclass(frozen=True, slots=True) 

52class ResolvedModelBudget: 

53 """The `model_max_budget` entry a request resolved to. 

54 

55 ``budget_model`` is the key as the operator configured it, not the model 

56 name on the request. Every counter is keyed on it so enforcement, the 

57 post-call increment and the `/key/info` + `/user/info` usage reads cannot 

58 disagree about which counter a request belongs to. 

59 """ 

60 

61 budget_model: str 

62 budget_config: BudgetConfig 

63 

64 

65def model_budget_spend_cache_key( 

66 entity_type: Litellm_EntityType, 

67 entity_id: str | None, 

68 budget_model: str, 

69 budget_duration: str | None, 

70) -> str: 

71 """Sole owner of the per-model spend counter key, shared by its writer and all of its readers.""" 

72 return f"{_SPEND_CACHE_KEY_PREFIXES[entity_type]}:{entity_id}:{budget_model}:{budget_duration}" 

73 

74 

75def _legacy_request_model_spend_cache_key( 

76 entity_type: Litellm_EntityType, 

77 entity_id: str | None, 

78 model: str, 

79 resolved: ResolvedModelBudget, 

80) -> str | None: 

81 """The counter this request was billed to before the budget model owned the key, or None. 

82 

83 Upgrading proxies carry live counters keyed on the model as REQUESTED 

84 (`openai/gpt-4`) rather than as configured (`gpt-4`), and those were the 

85 counters the previous version enforced on. Nothing writes that spelling once 

86 this version is running, so the pre-upgrade and post-upgrade counters hold 

87 disjoint halves of one window and adding them is the window's real spend. 

88 

89 Only the key and end-user scopes ever had one. The user scope is introduced 

90 by this change, so it has no counter to carry. 

91 

92 The carry stops one budget window after start-up, because a legacy counter 

93 belongs to a window that was already open when this process replaced the one 

94 writing it. Past that point the lookup could only ever miss. 

95 """ 

96 budget_duration: Final = resolved.budget_config.budget_duration 

97 if entity_type not in _LEGACY_REQUEST_MODEL_SCOPES or budget_duration is None: 

98 return None 

99 if time.monotonic() - _PROCESS_STARTED_AT >= duration_in_seconds(budget_duration): 

100 return None 

101 return model_budget_spend_cache_key( 

102 entity_type=entity_type, 

103 entity_id=entity_id, 

104 budget_model=model, 

105 budget_duration=budget_duration, 

106 ) 

107 

108 

109def model_budget_start_time_cache_key( 

110 entity_type: Litellm_EntityType, 

111 entity_id: str | None, 

112 budget_model: str, 

113 budget_duration: str | None, 

114) -> str: 

115 """Window start for one (entity, budget model) pair. 

116 

117 Scoped per budget model because an entity may budget two models over 

118 different periods, and a shared start time lets the shorter period restart 

119 the longer one's window. 

120 """ 

121 return f"{_BUDGET_START_TIME_KEY_PREFIXES[entity_type]}:{entity_id}:{budget_model}:{budget_duration}" 

122 

123 

124def batch_charged_once_marker_key(spend_key: str, batch_id: str) -> str: 

125 return f"{spend_key}:batch:{batch_id}" 

126 

127 

128def batch_id_to_charge_once(call_type: object, response_obj: object, response_cost: float) -> str | None: 

129 """A finished batch reports its whole cost on every poll, so its id is charged once per counter.""" 

130 if response_cost <= 0 or not is_batch_retrieve_call_type(call_type): 

131 return None 

132 return response_obj.id if isinstance(response_obj, Batch) else None 

133 

134 

135def resolve_model_budget(model: str, model_max_budget: Mapping[str, object]) -> ResolvedModelBudget | None: 

136 """Find the `model_max_budget` entry that governs `model`, or None.""" 

137 for candidate in _budget_model_candidates(model): 

138 raw_budget_config = model_max_budget.get(candidate) 

139 if raw_budget_config is None: 

140 continue 

141 if (budget_config := _usable_budget_config(raw_budget_config)) is None: 

142 # An entry that will not validate cannot be keyed, so it cannot be 

143 # enforced or incremented. Skip to the next candidate rather than 

144 # raising: raising would abort every other scope's increment and turn 

145 # a config typo into a 500, and stopping here would let one malformed 

146 # specific entry disable a perfectly good bare-family budget beside 

147 # it. The candidate chain already falls through an ABSENT entry, and 

148 # an unparseable one is indistinguishable from absent to enforcement. 

149 # `validate_model_max_budget` rejects these on the write path, so 

150 # reaching here means config.yaml or a direct DB edit. 

151 verbose_proxy_logger.warning( 

152 "Ignoring unusable model_max_budget entry for %s; it cannot be enforced or tracked", 

153 candidate, 

154 ) 

155 continue 

156 return ResolvedModelBudget(budget_model=candidate, budget_config=budget_config) 

157 return None 

158 

159 

160def team_model_budget_applies(model: str, key_model_max_budget: Mapping[str, object] | None) -> bool: 

161 """A key entry that spend-gates `model` overrides the team cap: it is then gated on and billed to the key alone.""" 

162 if not key_model_max_budget: 162 ↛ 164line 162 didn't jump to line 164 because the condition on line 162 was always true

163 return True 

164 resolved: Final = resolve_model_budget(model=model, model_max_budget=key_model_max_budget) 

165 return resolved is None or not _spend_gated(resolved.budget_config) 

166 

167 

168def _spend_gated(budget_config: BudgetConfig) -> bool: 

169 return budget_config.max_budget is not None and budget_config.max_budget >= 0 

170 

171 

172def _budget_model_candidates(model: str) -> tuple[str, ...]: 

173 """Names a budget may be configured under for a request on `model`, most specific first. 

174 

175 Beyond the model as sent, a budget may be keyed on the model without its 

176 ``{custom_llm_provider}/`` prefix (``gpt-4o`` governs ``openai/gpt-4o``), on 

177 the Bedrock base model (``anthropic.claude-opus-4-8`` governs the 

178 cross-region ``us.anthropic.claude-opus-4-8``), or on the bare family name 

179 that Bedrock id shares with its direct-provider twin (``claude-opus-4-8``). 

180 """ 

181 return tuple(dict.fromkeys((model, model.split("/")[-1], *_bedrock_candidates(model)))) 

182 

183 

184def _bedrock_candidates(model: str) -> tuple[str, ...]: 

185 """Bedrock-only candidates, empty unless litellm prices `model` as a Bedrock model. 

186 

187 Gating on the cost map rather than on a vendor allowlist is what makes 

188 splitting the leading dotted segment safe: most dotted model ids are not 

189 Bedrock ids at all (``azure/gpt-4.1``, ``gpt-image-1.5``), and splitting one 

190 of those would produce a garbage candidate. 

191 """ 

192 base_model: Final = get_bedrock_base_model(model) 

193 cost_entry: Final = litellm.model_cost.get(base_model) 

194 if not isinstance(cost_entry, dict) or not str(cost_entry.get("litellm_provider", "")).startswith("bedrock"): 

195 return () 

196 _, _, without_vendor = base_model.partition(".") 

197 return (base_model, without_vendor) if without_vendor else (base_model,) 

198 

199 

200async def build_model_max_budget_usage( 

201 entity_type: Litellm_EntityType, 

202 entity_id: str | None, 

203 model_max_budget: Mapping[str, object] | None, 

204 cache: DualCache | None, 

205) -> dict[str, dict[str, object]]: 

206 """Current-window spend per configured budget model, as `/key/info` and `/user/info` report it. 

207 

208 `cache` must be the DualCache the limiter writes the counters to; callers 

209 read it off the limiter rather than re-deriving it, so a scope that is being 

210 blocked can never report zero usage. 

211 """ 

212 if cache is None or entity_id is None or not model_max_budget: 212 ↛ 215line 212 didn't jump to line 215 because the condition on line 212 was always true

213 return {} 

214 

215 budgets: Final = tuple( 

216 (budget_model, budget_config) 

217 for budget_model, raw_budget_config in model_max_budget.items() 

218 for budget_config in (_usable_budget_config(raw_budget_config),) 

219 if budget_config is not None 

220 ) 

221 if not budgets: 

222 return {} 

223 spend_keys: Final = tuple( 

224 model_budget_spend_cache_key( 

225 entity_type=entity_type, 

226 entity_id=entity_id, 

227 budget_model=budget_model, 

228 budget_duration=budget_config.budget_duration, 

229 ) 

230 for budget_model, budget_config in budgets 

231 ) 

232 current_spends: Final = await _current_window_spends(cache=cache, spend_keys=spend_keys) 

233 return { 

234 budget_model: { 

235 "current_spend": round(current_spend, 4), 

236 "budget_limit": budget_config.max_budget, 

237 "time_period": budget_config.budget_duration, 

238 } 

239 for (budget_model, budget_config), current_spend in zip(budgets, current_spends, strict=True) 

240 } 

241 

242 

243async def _current_window_spends(cache: DualCache, spend_keys: Sequence[str]) -> tuple[float, ...]: 

244 """Redis holds the window total across replicas; the in-memory copy is one replica's share.""" 

245 keys: Final = list(spend_keys) # mutable-ok: both batch readers annotate their key argument as list 

246 redis_cache: Final = cache.redis_cache 

247 if redis_cache is not None: 

248 shared: Final = await redis_cache.async_batch_get_cache(key_list=keys) 

249 return tuple(_as_spend(shared.get(key)) for key in keys) 

250 # async_batch_get_cache returns None if it fails internally, and its result is 

251 # index-aligned with `keys` otherwise. An unusable result reads as a miss, 

252 # which is what a never-written counter already reads as. 

253 batched: Final = await cache.async_batch_get_cache(keys=keys) 

254 if not isinstance(batched, list) or len(batched) != len(keys): 

255 return (0.0,) * len(keys) 

256 return tuple(_as_spend(current_spend) for current_spend in batched) 

257 

258 

259def _usable_budget_config(raw_budget_config: object) -> BudgetConfig | None: 

260 try: 

261 budget_config: Final = BudgetConfig.model_validate(raw_budget_config) 

262 if budget_config.budget_duration is None: 

263 return None 

264 duration_in_seconds(budget_config.budget_duration) 

265 except Exception: # noqa: BLE001 # a malformed entry must not fail the whole report 

266 return None 

267 return budget_config 

268 

269 

270def _as_spend(current_spend: object) -> float: 

271 try: 

272 return float(current_spend or 0.0) # pyright: ignore[reportArgumentType] # non-numeric falls to the except 

273 except (TypeError, ValueError): 

274 return 0.0 

275 

276 

277def _resolve_entity_model_budgets( 

278 model: str, 

279 entity_budgets: Iterable[tuple[Litellm_EntityType, str | None, object]], 

280) -> tuple[tuple[Litellm_EntityType, str, ResolvedModelBudget], ...]: 

281 """Drop the scopes that do not budget `model`, keeping only what can be incremented.""" 

282 return tuple( 

283 (entity_type, entity_id, resolved) 

284 for entity_type, entity_id, model_max_budget in entity_budgets 

285 if entity_id is not None and isinstance(model_max_budget, Mapping) and model_max_budget 

286 for resolved in (resolve_model_budget(model=model, model_max_budget=model_max_budget),) 

287 if resolved is not None and resolved.budget_config.budget_duration is not None 

288 ) 

289 

290 

291class _PROXY_VirtualKeyModelMaxBudgetLimiter(RouterBudgetLimiting): 

292 """ 

293 Handles budgets for model + virtual key 

294 

295 Example: key=sk-1234567890, model=gpt-4o, max_budget=100, time_period=1d 

296 """ 

297 

298 def __init__(self, dual_cache: DualCache): 

299 self.dual_cache = dual_cache 

300 self.redis_increment_operation_queue = [] 

301 self._redis_increment_queue_lock = asyncio.Lock() 

302 self._redis_increment_flush_lock = asyncio.Lock() 

303 self._detached_increment_operations = None 

304 self.deployment_budget_config = None 

305 

306 async def is_key_within_model_budget( 

307 self, 

308 user_api_key_dict: UserAPIKeyAuth, 

309 model: str, 

310 ) -> bool: 

311 """ 

312 Check if the user_api_key_dict is within the model budget 

313 

314 Raises: 

315 BudgetExceededError: If the user_api_key_dict has exceeded the model budget 

316 """ 

317 return await self._is_entity_within_model_budget( 

318 entity_type=Litellm_EntityType.KEY, 

319 entity_id=user_api_key_dict.token, 

320 model_max_budget=user_api_key_dict.model_max_budget, 

321 model=model, 

322 exceeded_message=( 

323 f"LiteLLM Virtual Key: {user_api_key_dict.token}, key_alias: {user_api_key_dict.key_alias}, " 

324 f"exceeded budget for model={model}" 

325 ), 

326 ) 

327 

328 async def get_fallback_model_within_budget( 

329 self, 

330 user_api_key_dict: UserAPIKeyAuth, 

331 model: str, 

332 ) -> str | None: 

333 budget_fallbacks: Final[dict[str, list[str]]] = user_api_key_dict.budget_fallbacks or {} 

334 for fallback_model in budget_fallbacks.get(model, []): 

335 try: 

336 await self.is_key_within_model_budget(user_api_key_dict=user_api_key_dict, model=fallback_model) 

337 return fallback_model 

338 except litellm.BudgetExceededError: 

339 continue 

340 return None 

341 

342 async def is_user_within_model_budget( 

343 self, 

344 user_id: str, 

345 user_model_max_budget: Mapping[str, object], 

346 model: str, 

347 ) -> bool: 

348 """ 

349 Check if the internal user is within the model budget 

350 

351 Raises: 

352 BudgetExceededError: If the user has exceeded the model budget 

353 """ 

354 return await self._is_entity_within_model_budget( 

355 entity_type=Litellm_EntityType.USER, 

356 entity_id=user_id, 

357 model_max_budget=user_model_max_budget, 

358 model=model, 

359 exceeded_message=f"LiteLLM User: {user_id}, exceeded budget for model={model}", 

360 ) 

361 

362 async def is_end_user_within_model_budget( 

363 self, 

364 end_user_id: str, 

365 end_user_model_max_budget: Mapping[str, object], 

366 model: str, 

367 ) -> bool: 

368 """ 

369 Check if the end_user is within the model budget 

370 

371 Raises: 

372 BudgetExceededError: If the end_user has exceeded the model budget 

373 """ 

374 return await self._is_entity_within_model_budget( 

375 entity_type=Litellm_EntityType.END_USER, 

376 entity_id=end_user_id, 

377 model_max_budget=end_user_model_max_budget, 

378 model=model, 

379 exceeded_message=f"LiteLLM End User: {end_user_id}, exceeded budget for model={model}", 

380 ) 

381 

382 async def is_team_within_model_budget( 

383 self, 

384 team_id: str, 

385 team_model_max_budget: Mapping[str, object], 

386 key_model_max_budget: Mapping[str, object] | None, 

387 model: str, 

388 ) -> bool: 

389 """ 

390 Check if the team is within the model budget, unless the key's own 

391 `model_max_budget` overrides it for `model` 

392 

393 Raises: 

394 BudgetExceededError: If the team has exceeded the model budget 

395 """ 

396 if not team_model_budget_applies(model=model, key_model_max_budget=key_model_max_budget): 

397 return True 

398 return await self._is_entity_within_model_budget( 

399 entity_type=Litellm_EntityType.TEAM, 

400 entity_id=team_id, 

401 model_max_budget=team_model_max_budget, 

402 model=model, 

403 exceeded_message=f"LiteLLM Team: {team_id}, exceeded budget for model={model}", 

404 ) 

405 

406 async def _is_entity_within_model_budget( 

407 self, 

408 entity_type: Litellm_EntityType, 

409 entity_id: str | None, 

410 model_max_budget: Mapping[str, object] | None, 

411 model: str, 

412 exceeded_message: str, 

413 ) -> bool: 

414 if not model_max_budget: 

415 return True 

416 resolved: Final = resolve_model_budget(model=model, model_max_budget=model_max_budget) 

417 if resolved is None: 

418 verbose_proxy_logger.debug("Model %s not found in %s model_max_budget", model, entity_type.value) 

419 return True 

420 

421 max_budget: Final = resolved.budget_config.max_budget 

422 if max_budget is None or max_budget < 0: 

423 return True 

424 

425 current_spend: Final = await self._get_spend_for_model_budget( 

426 entity_type=entity_type, 

427 entity_id=entity_id, 

428 model=model, 

429 resolved=resolved, 

430 ) 

431 if current_spend >= max_budget: 

432 raise litellm.BudgetExceededError( 

433 message=exceeded_message, 

434 current_cost=current_spend, 

435 max_budget=max_budget, 

436 entity_type=entity_type.value, 

437 entity_id=entity_id, 

438 ) 

439 return True 

440 

441 async def _get_spend_for_model_budget( 

442 self, 

443 entity_type: Litellm_EntityType, 

444 entity_id: str | None, 

445 model: str, 

446 resolved: ResolvedModelBudget, 

447 ) -> float: 

448 """Spend charged to this budget in the current window, legacy counter included. 

449 

450 A counter that was never written is zero spend, not unknown spend. The 

451 distinction only shows up at a zero-dollar cap, where skipping the 

452 comparison would let the strictest possible limit admit every request. 

453 """ 

454 spend_key: Final = model_budget_spend_cache_key( 

455 entity_type=entity_type, 

456 entity_id=entity_id, 

457 budget_model=resolved.budget_model, 

458 budget_duration=resolved.budget_config.budget_duration, 

459 ) 

460 legacy_spend_key: Final = _legacy_request_model_spend_cache_key( 

461 entity_type=entity_type, 

462 entity_id=entity_id, 

463 model=model, 

464 resolved=resolved, 

465 ) 

466 current_spend: Final = _as_spend(await self._cached_spend(spend_key)) 

467 if legacy_spend_key is None or legacy_spend_key == spend_key: 

468 return current_spend 

469 return current_spend + _as_spend(await self._cached_spend(legacy_spend_key)) 

470 

471 async def _cached_spend(self, spend_key: str) -> float | None: 

472 redis_cache: Final = self.dual_cache.redis_cache 

473 if redis_cache is None: 

474 return await self.dual_cache.async_get_cache(key=spend_key) 

475 return await redis_cache.async_get_cache(key=spend_key) 

476 

477 async def async_filter_deployments( 

478 self, 

479 model: str, 

480 healthy_deployments: list, 

481 messages: list[AllMessageValues] | None, 

482 request_kwargs: dict | None = None, 

483 parent_otel_span: Span | None = None, 

484 ) -> list[dict]: 

485 return healthy_deployments 

486 

487 async def async_log_success_event(self, kwargs, response_obj, start_time, end_time): 

488 """ 

489 Track spend for virtual key + model in DualCache 

490 

491 Example: key=sk-1234567890, model=gpt-4o, max_budget=100, time_period=1d 

492 """ 

493 verbose_proxy_logger.debug("in RouterBudgetLimiting.async_log_success_event") 

494 standard_logging_payload: Final[StandardLoggingPayload | None] = kwargs.get("standard_logging_object", None) 

495 if standard_logging_payload is None: 

496 verbose_proxy_logger.debug( 

497 "Skipping _PROXY_VirtualKeyModelMaxBudgetLimiter.async_log_success_event: standard_logging_payload is None" 

498 ) 

499 return 

500 

501 _litellm_params: Final[dict] = kwargs.get("litellm_params", {}) or {} 

502 _metadata: Final[dict] = _litellm_params.get("metadata", {}) or {} 

503 payload_metadata: Final = standard_logging_payload.get("metadata") or {} 

504 

505 # Use model_group (the user-facing model alias, e.g. "gpt-4o") when 

506 # available. The enforcement path receives the model name from 

507 # request_data["model"] which is the model group alias, so the spend 

508 # tracking cache key must resolve from the same name. Falling back to 

509 # the deployment-level "model" field preserves behaviour for non-proxy 

510 # or non-router deployments where model_group is None. 

511 model: Final = standard_logging_payload.get("model_group") or standard_logging_payload.get("model") 

512 if model is None: 512 ↛ 513line 512 didn't jump to line 513 because the condition on line 512 was never true

513 return 

514 

515 response_cost: Final[float] = standard_logging_payload.get("response_cost", 0) 

516 key_model_max_budget: Final = _metadata.get("user_api_key_model_max_budget") 

517 entity_budgets: Final = ( 

518 ( 

519 Litellm_EntityType.KEY, 

520 payload_metadata.get("user_api_key_hash"), 

521 key_model_max_budget, 

522 ), 

523 ( 

524 Litellm_EntityType.TEAM, 

525 payload_metadata.get("user_api_key_team_id"), 

526 ( 

527 _metadata.get("user_api_key_team_model_max_budget") 

528 if team_model_budget_applies( 

529 model=model, 

530 key_model_max_budget=( 

531 key_model_max_budget if isinstance(key_model_max_budget, Mapping) else None 

532 ), 

533 ) 

534 else None 

535 ), 

536 ), 

537 ( 

538 Litellm_EntityType.USER, 

539 payload_metadata.get("user_api_key_user_id"), 

540 _metadata.get("user_api_key_user_model_max_budget"), 

541 ), 

542 ( 

543 Litellm_EntityType.END_USER, 

544 standard_logging_payload.get("end_user") or payload_metadata.get("user_api_key_end_user_id"), 

545 _metadata.get("user_api_key_end_user_model_max_budget"), 

546 ), 

547 ) 

548 

549 resolved_budgets: Final = _resolve_entity_model_budgets(model=model, entity_budgets=entity_budgets) 

550 if not resolved_budgets: 550 ↛ 558line 550 didn't jump to line 558 because the condition on line 550 was always true

551 verbose_proxy_logger.debug( 

552 "Not running _PROXY_VirtualKeyModelMaxBudgetLimiter.async_log_success_event: " 

553 "no key, team, user or end-user model_max_budget covers model=%s", 

554 model, 

555 ) 

556 return 

557 

558 batch_id: Final = batch_id_to_charge_once( 

559 call_type=kwargs.get("call_type"), 

560 response_obj=response_obj, 

561 response_cost=response_cost, 

562 ) 

563 for entity_type, entity_id, resolved in resolved_budgets: 

564 await self._charge_entity( 

565 entity_type=entity_type, 

566 entity_id=entity_id, 

567 resolved=resolved, 

568 response_cost=response_cost, 

569 batch_id=batch_id, 

570 ) 

571 

572 if self.dual_cache.redis_cache is not None: 

573 await self._push_in_memory_increments_to_redis() 

574 

575 verbose_proxy_logger.debug( 

576 "current state of in memory cache %s", 

577 json.dumps(self.dual_cache.in_memory_cache.cache_dict, indent=4, default=str), 

578 ) 

579 

580 async def _charge_entity( 

581 self, 

582 entity_type: Litellm_EntityType, 

583 entity_id: str | None, 

584 resolved: ResolvedModelBudget, 

585 response_cost: float, 

586 batch_id: str | None, 

587 ) -> None: 

588 budget_duration: Final = resolved.budget_config.budget_duration 

589 if budget_duration is None: 

590 return 

591 spend_key: Final = model_budget_spend_cache_key( 

592 entity_type=entity_type, 

593 entity_id=entity_id, 

594 budget_model=resolved.budget_model, 

595 budget_duration=budget_duration, 

596 ) 

597 if batch_id is not None and not await self._claim_batch_charge( 

598 spend_key=spend_key, 

599 batch_id=batch_id, 

600 ttl_seconds=duration_in_seconds(budget_duration), 

601 ): 

602 return 

603 await self._increment_spend_for_key( 

604 budget_config=resolved.budget_config, 

605 spend_key=spend_key, 

606 start_time_key=model_budget_start_time_cache_key( 

607 entity_type=entity_type, 

608 entity_id=entity_id, 

609 budget_model=resolved.budget_model, 

610 budget_duration=budget_duration, 

611 ), 

612 response_cost=response_cost, 

613 ) 

614 

615 async def _claim_batch_charge(self, spend_key: str, batch_id: str, ttl_seconds: int) -> bool: 

616 marker_key: Final = batch_charged_once_marker_key(spend_key=spend_key, batch_id=batch_id) 

617 polls: Final = await self.dual_cache.async_increment_cache( 

618 key=marker_key, value=1, ttl=ttl_seconds, refresh_ttl=True 

619 ) 

620 return polls == 1